mirror of
https://github.com/Ed94/pikuma_ps1.git
synced 2026-08-05 23:28:47 +00:00
Compare commits
127
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
27a9038e0d | ||
|
|
8c8d2e54aa | ||
|
|
80a35aa23a | ||
|
|
f247d56c32 | ||
|
|
590ff1e2ec | ||
|
|
653e18ee28 | ||
|
|
ebb876fe89 | ||
|
|
1b40b16c0e | ||
|
|
9ffd6592bc | ||
|
|
d56adab38f | ||
|
|
08af73d0d2 | ||
|
|
67d54debfa | ||
|
|
3c25306070 | ||
|
|
c3cf05950e | ||
|
|
f6b4d9895e | ||
|
|
e70361b548 | ||
|
|
ed3eb45b1d | ||
|
|
d7770b6e1d | ||
|
|
137549b1c8 | ||
|
|
7d5b13aadb | ||
|
|
2d901003f9 | ||
|
|
b43d22008e | ||
|
|
904889b483 | ||
|
|
f7aa7b75e7 | ||
|
|
aca6e30e20 | ||
|
|
8b0fb1d4e4 | ||
|
|
9f7a4a00ce | ||
|
|
277af1c901 | ||
|
|
97d2f66c5a | ||
|
|
d9406553b3 | ||
|
|
e662d175ab | ||
|
|
5387a07b84 | ||
|
|
65d805e3ba | ||
|
|
987f4dee1e | ||
|
|
df723c691d | ||
|
|
45ac85c038 | ||
|
|
072231c46b | ||
|
|
2b00956862 | ||
|
|
1ffad6cf98 | ||
|
|
318516a354 | ||
|
|
91a91b3495 | ||
|
|
a0d22700db | ||
|
|
51bdf7106b | ||
|
|
531e1cbd58 | ||
|
|
541e52de2b | ||
|
|
eccf17d21c | ||
|
|
0d94632edf | ||
|
|
798807a9c2 | ||
|
|
e9f26f89b8 | ||
|
|
a226b45d18 | ||
|
|
c22e4baa41 | ||
|
|
fa598a41c6 | ||
|
|
a928d06ac9 | ||
|
|
91c2218471 | ||
|
|
27a5f8029f | ||
|
|
7a168137fc | ||
|
|
2ceb2f2a05 | ||
|
|
6103f47f05 | ||
|
|
c824c998eb | ||
|
|
9d066ae292 | ||
|
|
c9b7f8c08b | ||
|
|
59903546d7 | ||
|
|
1ffdda45e5 | ||
|
|
1209172649 | ||
|
|
98e27c2815 | ||
|
|
88aa1b8b59 | ||
|
|
ca3dc4aff0 | ||
|
|
1fb4883138 | ||
|
|
0ad609e7c2 | ||
|
|
ccdf1b832b | ||
|
|
4d177bc34d | ||
|
|
32a754cd06 | ||
|
|
407c7d352a | ||
|
|
8541713d0c | ||
|
|
602a0b46d8 | ||
|
|
74f390c3b1 | ||
|
|
5e7da32387 | ||
|
|
5375478044 | ||
|
|
10c8dcdc07 | ||
|
|
d0b1bae896 | ||
|
|
0b147a8b0c | ||
|
|
101b07fe71 | ||
|
|
a01e724211 | ||
|
|
7a1def6c46 | ||
|
|
6eee0249ff | ||
|
|
93bab89f76 | ||
|
|
b17a653d20 | ||
|
|
26a63ed908 | ||
|
|
c293e35cb4 | ||
|
|
b29fc7dc02 | ||
|
|
ffcd43c1ea | ||
|
|
352a8c9c25 | ||
|
|
c3a7e22743 | ||
|
|
54968d20e0 | ||
|
|
13a0b9dca4 | ||
|
|
b1982ba862 | ||
|
|
288f92ff5b | ||
|
|
68fd4ca791 | ||
|
|
e7d0b7a4b3 | ||
|
|
47806e3d21 | ||
|
|
c4e25daa9b | ||
|
|
2e4912b1e2 | ||
|
|
1b77d8bae3 | ||
|
|
2c3d0c4af7 | ||
|
|
0040f6326e | ||
|
|
30e2a84ebf | ||
|
|
66facd79dd | ||
|
|
4603a3bb9a | ||
|
|
35c9278ebb | ||
|
|
915b14ef31 | ||
|
|
8d03366d92 | ||
|
|
1b1c926318 | ||
|
|
0573644494 | ||
|
|
bcb9d9a77c | ||
|
|
912fdcde30 | ||
|
|
28bb20d6fe | ||
|
|
d776d71574 | ||
|
|
690d184acf | ||
|
|
5cc30bacc6 | ||
|
|
d89a29c941 | ||
|
|
e178743ffb | ||
|
|
27667a4232 | ||
|
|
11cc936d2d | ||
|
|
679c022625 | ||
|
|
b81221cbf5 | ||
|
|
fe4bd6d86d | ||
|
|
8b5fb7a5a1 |
+9
-1
@@ -16,7 +16,11 @@ indent_size = 4
|
||||
indent_style = space
|
||||
indent_size = 4
|
||||
|
||||
[*{.h, .c, .hpp, .cpp}]
|
||||
[*.h]
|
||||
indent_style = tab
|
||||
indent_size = 2
|
||||
|
||||
[*.c]
|
||||
indent_style = tab
|
||||
indent_size = 2
|
||||
|
||||
@@ -32,3 +36,7 @@ charset = utf-8
|
||||
[*.{natvis, natstepfilter}]
|
||||
indent_style = tab
|
||||
indent_size = 4
|
||||
|
||||
[*.lua]
|
||||
indent_style = tab
|
||||
indent_size = 2
|
||||
|
||||
+5
-2
@@ -1,8 +1,9 @@
|
||||
build
|
||||
toolchain/armips
|
||||
toolchain/luajit-2.1
|
||||
toolchain/pcsx-redux
|
||||
# toolchain/psyq_iwyu
|
||||
# toolchain/PSn00bSDK
|
||||
toolchain/psyq_iwyu
|
||||
toolchain/PSn00bSDK
|
||||
|
||||
*.exe
|
||||
*.elf
|
||||
@@ -14,3 +15,5 @@ toolchain/pcsx-redux
|
||||
*.a
|
||||
.sentry-native
|
||||
.vscode/settings.json
|
||||
toolchain/lfs
|
||||
toolchain/lpeg
|
||||
|
||||
Vendored
+73
-5
@@ -4,7 +4,7 @@
|
||||
// For more information, visit: https://go.microsoft.com/fwlink/?linkid=830387
|
||||
"version": "0.2.0",
|
||||
"configurations": [
|
||||
{
|
||||
{
|
||||
"name": "Debug: Hello Psy-Q!",
|
||||
"type": "gdb",
|
||||
"request": "attach",
|
||||
@@ -12,6 +12,10 @@
|
||||
"remote": true,
|
||||
"cwd": "${workspaceRoot}/build",
|
||||
"valuesFormatting": "parseText",
|
||||
"registerLimit": "1-32",
|
||||
"frameFilters": false,
|
||||
"showDevDebugOutput": false,
|
||||
"printCalls": false,
|
||||
"stopAtConnect": true,
|
||||
"gdbpath": "gdb-multiarch",
|
||||
"windows": {
|
||||
@@ -20,10 +24,17 @@
|
||||
"osx": {
|
||||
"gdbpath": "gdb"
|
||||
},
|
||||
"executable": "${workspaceRoot}/build/hello_psyq.elf",
|
||||
"executable": "${workspaceRoot}/build/hello_gte.elf",
|
||||
"setupCommands": [
|
||||
{ "text": "set mi-async off" },
|
||||
{ "text": "set remotetimeout 0" },
|
||||
{ "text": "set logging file build/gen/hello_gte.gdb.log" },
|
||||
{ "text": "set logging redirect on" }
|
||||
],
|
||||
"autorun": [
|
||||
"monitor reset shellhalt",
|
||||
"load hello_psyq.elf",
|
||||
"load hello_gte.elf",
|
||||
"source scripts/gdb/gdb_tape_atoms.gdb",
|
||||
"tbreak main",
|
||||
"continue"
|
||||
]
|
||||
@@ -36,6 +47,10 @@
|
||||
"remote": true,
|
||||
"cwd": "${workspaceRoot}/build",
|
||||
"valuesFormatting": "parseText",
|
||||
"registerLimit": "1-32",
|
||||
"frameFilters": false,
|
||||
"showDevDebugOutput": false,
|
||||
"printCalls": false,
|
||||
"stopAtConnect": true,
|
||||
"gdbpath": "gdb-multiarch",
|
||||
"windows": {
|
||||
@@ -44,10 +59,16 @@
|
||||
"osx": {
|
||||
"gdbpath": "gdb"
|
||||
},
|
||||
"executable": "${workspaceRoot}/build/hello_gpu.elf",
|
||||
"executable": "${workspaceRoot}/build/hello_gte.elf",
|
||||
"setupCommands": [
|
||||
{ "text": "set mi-async off" },
|
||||
{ "text": "set remotetimeout 0" },
|
||||
{ "text": "set logging file build/gen/hello_gte.gdb.log" },
|
||||
{ "text": "set logging redirect on" }
|
||||
],
|
||||
"autorun": [
|
||||
"monitor reset shellhalt",
|
||||
"load hello_gpu.elf",
|
||||
"load hello_gte.elf",
|
||||
"tbreak main",
|
||||
"continue"
|
||||
]
|
||||
@@ -60,6 +81,10 @@
|
||||
"remote": true,
|
||||
"cwd": "${workspaceRoot}/build",
|
||||
"valuesFormatting": "parseText",
|
||||
"registerLimit": "1-32",
|
||||
"frameFilters": false,
|
||||
"showDevDebugOutput": false,
|
||||
"printCalls": false,
|
||||
"stopAtConnect": true,
|
||||
"gdbpath": "gdb-multiarch",
|
||||
"windows": {
|
||||
@@ -69,12 +94,55 @@
|
||||
"gdbpath": "gdb"
|
||||
},
|
||||
"executable": "${workspaceRoot}/build/hello_gte.elf",
|
||||
"setupCommands": [
|
||||
{ "text": "set mi-async off" },
|
||||
{ "text": "set remotetimeout 0" },
|
||||
{ "text": "set logging file build/gen/hello_gte.gdb.log" },
|
||||
{ "text": "set logging redirect on" }
|
||||
],
|
||||
"autorun": [
|
||||
"monitor reset shellhalt",
|
||||
"load hello_gte.elf",
|
||||
"tbreak main",
|
||||
"continue"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "Debug: Hello GTE Psy-Q! (atoms debug — DWARF-injected)",
|
||||
"type": "gdb",
|
||||
"request": "attach",
|
||||
"target": "localhost:3333",
|
||||
"remote": true,
|
||||
"cwd": "${workspaceRoot}",
|
||||
"valuesFormatting": "parseText",
|
||||
"registerLimit": "1-32",
|
||||
"frameFilters": false,
|
||||
"showDevDebugOutput": false,
|
||||
"printCalls": false,
|
||||
"stopAtConnect": true,
|
||||
"gdbpath": "gdb-multiarch",
|
||||
"windows": {
|
||||
"gdbpath": "gdb-multiarch.exe"
|
||||
},
|
||||
"osx": {
|
||||
"gdbpath": "gdb"
|
||||
},
|
||||
"executable": "${workspaceRoot}/build/hello_gte.dwarf-injected.elf",
|
||||
"setupCommands": [
|
||||
{ "text": "set mi-async off" },
|
||||
{ "text": "set remotetimeout 0" },
|
||||
{ "text": "set logging file build/gen/hello_gte.gdb.log" },
|
||||
{ "text": "set logging redirect on" }
|
||||
],
|
||||
"autorun": [
|
||||
"monitor reset shellhalt",
|
||||
"load build/hello_gte.dwarf-injected.elf",
|
||||
"source scripts/gdb/gdb_tape_atoms.gdb",
|
||||
"source build/gen/hello_gte.gdbinit",
|
||||
"tbreak main",
|
||||
"continue"
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,160 @@
|
||||
/*
|
||||
* atom_dsl.h
|
||||
* ============================================================================
|
||||
*
|
||||
* ATOM DSL: Annotation layer for tape atoms (lottes_tape.h).
|
||||
* The metaprogram (scripts/passes/annotation.lua) reads source-as-written and validates:
|
||||
* - atom_info(...) shape: up to three sub-calls (atom_bind(Binds_X), atom_reads(...), atom_writes(...)) in any order and are optional.
|
||||
* - rbind atoms (atom_info(..., atom_bind(Binds_X), ...)) reference a real Binds_* struct declaration.
|
||||
* - atom word-counts in word_counts.metadata.h match the body's actual .word count.
|
||||
*
|
||||
* Pure macro anntation.
|
||||
* ---------------
|
||||
* Don't want to constraint the macro usage to some attribute placment constraint, etc, don't want ot dela with the compiler.
|
||||
* atom_info, atom_bind, atom_reads, atom_writes, atom_label, atom_dbg_skip each expand to a C comment or to nothing
|
||||
* (C preprocessor strips them to whitespace).
|
||||
*
|
||||
* ============================================================================
|
||||
* Usage:
|
||||
* MipsAtom_(cube_tri) atom_info(
|
||||
* atom_reads (R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
* , atom_writes(R_PrimCursor, R_FaceCursor)
|
||||
* ){
|
||||
* atom_label(culling),
|
||||
* // ... atom body ...
|
||||
* atom_offset(culling, bounds_chk) // branch target, validated
|
||||
* // ... atom body ...
|
||||
* atom_label(bounds_chk),
|
||||
* };
|
||||
*
|
||||
*
|
||||
* Data Binding pattern -- atom_bind as a sub-call of atom_info
|
||||
*
|
||||
* // Wave-context register layout (declarative):
|
||||
* typedef Struct_(Binds_TrackFaceBatch) {
|
||||
* U4 PrimCursor;
|
||||
* U4 FaceCursor;
|
||||
* U4 VertBase;
|
||||
* U4 OtBase;
|
||||
* };
|
||||
* MipsAtom_(rbind_track_face_batch) atom_info(
|
||||
* atom_bind(Binds_TrackFaceBatch)
|
||||
* , atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
* ){ ... };
|
||||
*
|
||||
* Annotation rules
|
||||
* ----------------
|
||||
* 1. atom_info(...) is OPTIONAL. Atoms without atom_info are silently skipped by the metaprogram.
|
||||
* 2. If present, atom_info takes up to three sub-calls, all order-independent within the arg list:
|
||||
* - atom_bind(Binds_X)
|
||||
* - atom_reads(...)
|
||||
* - atom_writes(...)
|
||||
* 3. atom_bind(Binds_X): metaprogram cross-references Binds_X against the `typedef struct Binds_X { ... } Binds_X;` declaration.
|
||||
* 4. atom_reads(...) and atom_writes(...): Used to to check if registers are used correctly in macros: R_PrimCursor / R_FaceCursor / R_VertBase / R_OtBase.
|
||||
* 5. atom_label(name: Utilize with atom_offset as a target location.
|
||||
* 6. atom_offset(F, T): Resolved by gen/atom_offsets.h, generated from the atom_label markers. Calculated during the offset pass of the lua metaprogram.
|
||||
*/
|
||||
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
#pragma once
|
||||
// #include <stdint.h>
|
||||
#endif
|
||||
|
||||
/* ============================================================================
|
||||
* atom_reads(...) / atom_writes(...)
|
||||
*
|
||||
* Used during the static analysis pass of the metaprogram to do
|
||||
* ============================================================================*/
|
||||
#define atom_reads(...) (__VA_ARGS__)
|
||||
#define atom_writes(...) (__VA_ARGS__)
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* atom_reg (per-enum opt-in marker for the DWARF register-alias registry)
|
||||
*
|
||||
* The bare `atom_reg` token adjacent to an enum entry in mips.h / lottes_tape.h flags that alias as debug-visible for scan_source's register_alias_registry.
|
||||
* The C preprocessor strips it to a comment so no runtime symbol is created; the Lua scanner reads the bare token.
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_reg /* atom_reg: opt the preceding enum entry into the DWARF registry */
|
||||
|
||||
/* ============================================================================
|
||||
* atom_info :
|
||||
* MipsAtom_(cube_tri) atom_info(
|
||||
* atom_reads (R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
* , atom_writes(R_PrimCursor, R_FaceCursor)
|
||||
* ){ ... };
|
||||
*
|
||||
* - atom_bind(Binds_X): metaprogram cross-references Binds_X against the `typedef struct Binds_X { ... } Binds_X;` declaration.
|
||||
* - atom_reads(...): comma-list of registers
|
||||
* - atom_writes(...): comma-list of registers
|
||||
* ============================================================================*/
|
||||
#define atom_info(...) /* atom_info(__VA_ARGS__) */
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* DEBUG SOURCE-STEP MARKER
|
||||
*
|
||||
* Place `atom_dbg_skip` (BARE) before a MipsAtom_, MipsAtomComp_, or MipsAtomComp_Proc_.
|
||||
* The following declaration kind determines whether the marker selects a whole atom or a component inline view.
|
||||
* The source scanner associates the marker with that declaration; placement diagnostics are handled by the annotation pass.
|
||||
*
|
||||
* Example:
|
||||
* atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(rret_addr), nop };
|
||||
* atom_dbg_skip MipsAtomComp_(ac_yield) { ... };
|
||||
* atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, { ... });
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_dbg_skip /* atom_dbg_skip: skip the following atom or component source view */
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* Typed-view annotations (Registry for DWARF RR_<R_X> chain resolution)
|
||||
* atom_type(<T>) -- overloaded:
|
||||
* (a) enum-site default: `R_Foo = R_Tn, atom_reg atom_type(T)`
|
||||
* Sets the per-alias default typed view in the register_alias_registry.
|
||||
* Consumed by the DWARF chain step (e) when no per-atom atom_ctx / atom_phase / atom_type callsite provides a stronger resolution.
|
||||
* (b) callsite override: `atom_reads(R_Foo atom_type(T), ...)` Overrides the per-alias default for THIS atom only.
|
||||
* Last-write-wins per R_Name; conflict -> error.
|
||||
* atom_ctx(<atom_name>) -- atom-info sub-call:
|
||||
* Propagate another atom's atom.rbind.fields (its Binds_* typed fields) into THIS atom's typed-view resolution.
|
||||
* The named atom must be an rbind atom (have `atom_bind(Binds_X)` in its `atom_info`).
|
||||
* Used as the escape hatch when atom_phase is not the natural correlation.
|
||||
* atom_phase(<label>) -- atom-info sub-call:
|
||||
* Free-form C-identifier label for grouping atoms.
|
||||
* Within a phase, the FIRST atom in source-order that owns its own atom.rbind provides
|
||||
* the Binds_* field types used by all other atoms in the same phase.
|
||||
* The preferred correlation mechanism; atom_ctx is the escape hatch for non-natural cases.
|
||||
*
|
||||
* All three expand to C comments
|
||||
* (the bare-token convention matching `atom_reg` and `atom_dbg_skip`).
|
||||
* The Lua scanner reads the bare tokens in source-as-written; the C preprocessor strips them.
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_type(T) /* atom_type: associate <T> with the preceding enum entry (enum site) or this register (atom-info site) */
|
||||
#define atom_ctx(atom_name) /* atom_ctx: propagate <atom_name>'s Binds_* field types into this atom's typed views */
|
||||
#define atom_phase(label) /* atom_phase: tag this atom with <label> for grouped typed-view resolution */
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* atom_bind(Binds_X) -- rbind sub-call of atom_info
|
||||
*
|
||||
* MipsAtom_(rbind_cube_tri) atom_info(
|
||||
* atom_bind(Binds_CubeTri)
|
||||
* , atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
* ){ ... };
|
||||
*
|
||||
* The Binds_X MUST be a typedef'd type (declared via `typedef struct Binds_X { ... } Binds_X;` somewhere in the source).
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_bind(binds_struct) /* atom_bind(binds_struct) */
|
||||
|
||||
/* ============================================================================
|
||||
* atom_label / atom_offset — branch target machinery
|
||||
*
|
||||
* atom_label(culling) ← nothing in C; anchor only
|
||||
* ... body ...
|
||||
* atom_label(bounds_chk) ← another anchor
|
||||
*
|
||||
* atom_offset(culling, bounds_chk) ← resolved by gen/.offsets.h
|
||||
*
|
||||
* The metaprogram generates gen/atom_offsets.h with one #define with the offset value per atom_offset(F, T) call.
|
||||
* The preprocessor then expands the call to the right immediate value.
|
||||
*
|
||||
* If gen/atom_offsets.h is stale (or atom_label(name) is undefined), `atom_offset_F_T` becomes an undefined macro and the C build fails.
|
||||
* ============================================================================*/
|
||||
#define atom_offset(F, T) atom_offset_ ## F ## _ ## T
|
||||
// atom_label is a pure annotation for the metaprogram's offset calculations.
|
||||
#define atom_label(name) /* atom_label anchor: name */
|
||||
+185
-124
@@ -3,105 +3,166 @@
|
||||
# include "assert.h"
|
||||
#endif
|
||||
|
||||
#define LP_ static // local_persist
|
||||
#define internal static // internal
|
||||
#define global
|
||||
#define gknown
|
||||
|
||||
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
||||
#define expect_(x, y) __builtin_expect(x, y) // so compiler knows the common path
|
||||
#define FI_ static inline __attribute__((always_inline)) // force inline
|
||||
#define NI_ static __attribute__((noinline)) // force no inline [used in thread api]
|
||||
#define R_ __restrict // pointers are either restricted or volatile and nothing else
|
||||
#define V_ volatile // pointers are either restricted or volatile and nothing else
|
||||
|
||||
#define glue_impl(A, B) A ## B
|
||||
#define glue(A, B) glue_impl(A, B)
|
||||
#define stringify_impl(S) #S
|
||||
#define stringify(S) stringify_impl(S)
|
||||
#define tmpl(prefix, type) prefix ## _ ## type
|
||||
|
||||
#define offset_of(type, member) cast(U8,__builtin_offsetof(type,member))
|
||||
#define static_assert _Static_assert
|
||||
#define typeof __typeof__
|
||||
#define typeof_ptr(ptr) typeof((ptr)[0])
|
||||
#define typeof_same(a, b) _Generic((a), typeof((b)): 1, default: 0)
|
||||
|
||||
#define def_R_(type) type*restrict type ## _R
|
||||
#define def_V_(type) type*volatile type ## _V
|
||||
#define def_ptr_set(type) def_R_(type); typedef def_V_(type)
|
||||
#define def_tset(type) type; typedef def_ptr_set(type)
|
||||
#define m_expand(...) __VA_ARGS__
|
||||
#define glue_impl(A, B) A ## B
|
||||
#define glue(A, B) glue_impl(A, B)
|
||||
#define tmpl(prefix, type) prefix ## _ ## type
|
||||
|
||||
typedef __UINT8_TYPE__ def_tset(U1);
|
||||
typedef __UINT16_TYPE__ def_tset(U2);
|
||||
typedef __UINT32_TYPE__ def_tset(U4);
|
||||
typedef __INT8_TYPE__ def_tset(S1);
|
||||
typedef __INT16_TYPE__ def_tset(S2);
|
||||
typedef __INT32_TYPE__ def_tset(S4);
|
||||
typedef unsigned char def_tset(B1);
|
||||
typedef __UINT16_TYPE__ def_tset(B2);
|
||||
typedef __UINT32_TYPE__ def_tset(B4);
|
||||
typedef __UINT64_TYPE__ def_tset(B8);
|
||||
#define stringify_impl(S) #S
|
||||
#define stringify(S) stringify_impl(S)
|
||||
|
||||
#define VA_Sel_1( _1, ... ) _1 // <-- Of all th args passed pick _1.
|
||||
#define VA_Sel_2( _1, _2, ... ) _2 // <-- Of all the args passed pick _2.
|
||||
#define VA_Sel_3( _1, _2, _3, ... ) _3 // etc..
|
||||
|
||||
#define global static // Mark global data
|
||||
#define gknown // Mark global data used in procedure
|
||||
|
||||
#define LP_ static // static data within procedure scope
|
||||
#define internal static // internal
|
||||
|
||||
#define asm __asm__
|
||||
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
||||
|
||||
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
||||
#define C_(type,data) ((type)(data)) // for enforced precedence
|
||||
#define expect_(x, y) __builtin_expect(x, y) // so compiler knows the common path
|
||||
#define cexpr_ __builtin_constant_p
|
||||
#define I_ internal inline
|
||||
#define FI_ inline __attribute__((always_inline)) // inline always
|
||||
#define NI_ internal __attribute__((noinline)) // inline never
|
||||
#define RO_ __attribute__((section(".rodata"))) // Read only data allocation
|
||||
#define T_ typeof //
|
||||
#define T_same(a,b) _Generic((a), typeof((b)): 1, default: 0)
|
||||
|
||||
#define R_ restrict
|
||||
#define V_ volatile
|
||||
// Fictional, used for intiution.
|
||||
#define EUB_ restrict // Execute Unit Bound: Data is siloed in the ALU Register File. The Load/Store Unit is bypassed. (Route to Execution Unit. Keep in registers)
|
||||
#define ISO_ restrict // Isolated Provenance: Alternative to Exu_. Guarantees electrical memory isolation,
|
||||
// unlocking the compiler’s ability to safely pack data across multiple parallel SIMD lanes (vectorization).
|
||||
#define LSU_ volatile // Load/Store Unit Bound: The compiler is forbidden from caching in registers. Forces physical L1 Cache matrix sampling.
|
||||
#define LIVE_ volatile // Live External Data: Alternative to Lsu_ emphasizing the memory is tapped by an external electrical actor.
|
||||
|
||||
#define latch_store /* ~: atomic_store*/ // Blasts voltages from the Store Buffer into the L1 SRAM, physically flipping the cross-coupled inverters to lock the state.
|
||||
#define pulse_rfo /* ~: atomic_xchg*/ // Broadcasts an electrical RFO (Request For Ownership) pulse across the CPU mesh network to invalidate other L1 caches.
|
||||
#define tact_acquire /* ~: memory_order_acquire*/ // Clamp. Sends a voltage signal to the instruction decoder to halt the Out-of-Order engine until the load resolves.
|
||||
#define tact_release /* ~: memory_order_release*/ // Drain. Forces the Store Buffer flip-flops to completely empty into the L1 cache before proceeding.
|
||||
|
||||
// -----------------------------------------------------------------------------
|
||||
// Out-of-Order (OoO) Pipeline Modifiers
|
||||
// -----------------------------------------------------------------------------
|
||||
#define ooo_drift_ __ATOMIC_RELAXED // OoO engine allowed to drift
|
||||
#define ooo_anchor_ __ATOMIC_ACQUIRE // Anchor the Load Queue (halt spec lookahead)
|
||||
#define ooo_drain_ __ATOMIC_RELEASE // Drain the Store Buffer (force writeback)
|
||||
#define ooo_weld_ __ATOMIC_SEQ_CST // Weld pipeline (total order bus lock)
|
||||
|
||||
// Latch operations with physical queue modifiers
|
||||
#define latch_load_anchor(ptr) //__atomic_load_n(ptr, ooo_anchor_)
|
||||
#define latch_store_drain(ptr, val) //__atomic_store_n(ptr, val, ooo_drain_)
|
||||
#define pulse_xchg_weld(ptr, val) //__atomic_exchange_n(ptr, val, ooo_weld_)
|
||||
//end of: Fictional.
|
||||
|
||||
|
||||
// R_ (restrict) establishes an "Eigen" or "Proprius" mapping.
|
||||
// Unlike volatile (V_), which assumes the memory can be changed by anything,
|
||||
// R_ tells the compiler that this pointer holds the *sole*, private (idios)
|
||||
// ownership of the memory slice. Writes to this memory are exclusively bound
|
||||
// to this single symbolic mapping for the duration of the scope, guaranteeing
|
||||
// zero aliasing.
|
||||
|
||||
#define r_(ptr) C_(T_(ptr[0])*R_, ptr) // Constrain pointer to restrict
|
||||
#define v_(ptr) C_(T_(ptr[0])V_*, ptr) //
|
||||
#define tr_(type, ptr) C_(type *R_, ptr)
|
||||
#define tv_(type, ptr) C_(type V_*, ptr)
|
||||
|
||||
#define TypeR_(type) type *R_ type ## _R // type *restrict type_R
|
||||
#define TypeV_(type) type V_* type ## _V // type volatile* type_V
|
||||
#define PtrSet_(type) TypeR_(type); typedef TypeV_(type)
|
||||
#define TSet_(type) type; typedef PtrSet_(type)
|
||||
|
||||
#define array_len(a) (U4)(sizeof(a) / sizeof(typeof((a)[0])))
|
||||
#define array_decl(type, ...) (type[]){__VA_ARGS__}
|
||||
#define Array_sym(type,len) A ## len ## _ ## type
|
||||
#define Array_expand(type,len) type Array_sym(type, len)[len]; typedef PtrSet_(Array_sym(type, len))
|
||||
#define Array_(type,len) Array_expand(type,len)
|
||||
#define Bit_(id,b) id = (1 << b), tmpl(id,pos) = b
|
||||
#define Enum_(underlying_type, symbol) underlying_type TSet_(symbol); enum symbol
|
||||
#define Proc_(symbol) symbol
|
||||
#define Relative_(symbol) // Does nothing but annotate that a symbol is associated with another.
|
||||
#define Struct_(symbol) struct symbol TSet_(symbol); struct symbol
|
||||
#define Union_(symbol) union symbol TSet_(symbol); union symbol
|
||||
|
||||
#define Opt_(proc) Struct_(tmpl(Opt,proc))
|
||||
#define opt_(symbol, ...) (tmpl(Opt,symbol)){__VA_ARGS__}
|
||||
#define Ret_(proc) Struct_(tmpl(Ret,proc))
|
||||
#define ret_(proc) tmpl(Ret,proc) proc
|
||||
|
||||
// Using Byte-Width convention for the fundamental types.
|
||||
typedef __UINT8_TYPE__ TSet_(U1);
|
||||
typedef __UINT16_TYPE__ TSet_(U2);
|
||||
typedef __UINT32_TYPE__ TSet_(U4);
|
||||
typedef __INT8_TYPE__ TSet_(S1);
|
||||
typedef __INT16_TYPE__ TSet_(S2);
|
||||
typedef __INT32_TYPE__ TSet_(S4);
|
||||
typedef unsigned char TSet_(B1);
|
||||
typedef __UINT16_TYPE__ TSet_(B2);
|
||||
typedef __UINT32_TYPE__ TSet_(B4);
|
||||
|
||||
#define u1_(value) C_(U1, value)
|
||||
#define u2_(value) C_(U2, value)
|
||||
#define u4_(value) C_(U4, value)
|
||||
#define s1_(value) C_(S1, value)
|
||||
#define s2_(value) C_(S2, value)
|
||||
#define s4_(value) C_(S4, value)
|
||||
|
||||
#define u1_r(value) C_(U1 *R_, value)
|
||||
#define u2_r(value) C_(U2 *R_, value)
|
||||
#define u4_r(value) C_(U4 *R_, value)
|
||||
#define u1_v(value) C_(U1 V_*, value)
|
||||
#define u2_v(value) C_(U2 V_*, value)
|
||||
#define u4_v(value) C_(U4 V_*, value)
|
||||
enum { false = 0, true = 1, true_overflow, };
|
||||
|
||||
#define u1_r(value) cast(U1_R, value)
|
||||
#define u2_r(value) cast(U2_R, value)
|
||||
#define u4_r(value) cast(U4_R, value)
|
||||
#define u1_v(value) cast(U1_V, value)
|
||||
#define u2_v(value) cast(U2_V, value)
|
||||
#define u4_v(value) cast(U4_V, value)
|
||||
#define u4_lo(value) ((value) & 0xFFFFU)
|
||||
#define u4_hi(value) ((value) >> 12)
|
||||
|
||||
#define u1_(value) cast(U1, value)
|
||||
#define u2_(value) cast(U2, value)
|
||||
#define u4_(value) cast(U4, value)
|
||||
#define s1_(value) cast(S1, value)
|
||||
#define s2_(value) cast(S2, value)
|
||||
#define s4_(value) cast(S4, value)
|
||||
typedef void Proc_(VoidFn) (void);
|
||||
|
||||
#define farray_len(array) (SSIZE)sizeof(array) / size_of( typeof((array)[0]))
|
||||
#define farray_init(type, ...) (type[]){__VA_ARGS__}
|
||||
#define def_farray_sym(_type, _len) A ## _len ## _ ## _type
|
||||
#define def_farray_impl(_type, _len) _type def_farray_sym(_type, _len)[_len]; typedef def_ptr_set(def_farray_sym(_type, _len))
|
||||
#define def_farray(type, len) def_farray_impl(type, len)
|
||||
#define def_enum(underlying_type, symbol) underlying_type def_tset(symbol); enum symbol
|
||||
#define def_struct(symbol) struct symbol def_tset(symbol); struct symbol
|
||||
#define def_union(symbol) union symbol def_tset(symbol); union symbol
|
||||
#define def_proc(symbol) symbol
|
||||
#define opt_args(symbol, ...) &(symbol){__VA_ARGS__}
|
||||
#define ret_type(type) type
|
||||
#define kilo(n) (C_(U4, n) << 10)
|
||||
#define mega(n) (C_(U4, n) << 20)
|
||||
#define giga(n) (C_(U4, n) << 30)
|
||||
#define tera(n) (C_(U4, n) << 40)
|
||||
#define null C_(U4, 0)
|
||||
#define nullptr C_(void*, 0)
|
||||
#define O_(type, field) (C_(U4, & C_(type*,0)->field))
|
||||
|
||||
#define OT_(field) O_(typeof_ptr(& field), filed))
|
||||
#define S_(data) C_(U4, sizeof(data))
|
||||
|
||||
#define o_(field) offset_of(typeof_ptr(& field), filed))
|
||||
#define sop_1(op,a,b) C_(U1, s1_(a) op s1_(b))
|
||||
#define sop_2(op,a,b) C_(U2, s2_(a) op s2_(b))
|
||||
#define sop_4(op,a,b) C_(U4, s4_(a) op s4_(b))
|
||||
|
||||
#define alignas _Alignas
|
||||
#define alignof _Alignof
|
||||
#define byte_pad(amount, ...) B1 glue(_PAD_, __VA_ARGS__) [amount]
|
||||
#define cast(type, data) ((type)(data))
|
||||
#define pcast(type, data) (cast(type*, & (data)) [0])
|
||||
#define nullptr cast(void*, 0)
|
||||
#define size_of(data) cast(U4, sizeof(data))
|
||||
|
||||
#define r_(ptr) cast(typeof_ptr(ptr)*R_, ptr)
|
||||
#define v_(ptr) cast(typeof_ptr(ptr)*V_, ptr)
|
||||
#define tr_(type, ptr) cast(type*R_, ptr)
|
||||
#define tv_(type, ptr) cast(type*V_, ptr)
|
||||
|
||||
#define kilo(n) (cast(U4, n) << 10)
|
||||
#define mega(n) (cast(U4, n) << 20)
|
||||
#define giga(n) (cast(U4, n) << 30)
|
||||
#define tera(n) (cast(U4, n) << 40)
|
||||
|
||||
#define dbg_args(...) __VA_ARGS__
|
||||
|
||||
#define sop_1(op, a, b) cast(U1, s1_(a) op s1_(b))
|
||||
#define sop_2(op, a, b) cast(U2, s2_(a) op s2_(b))
|
||||
#define sop_4(op, a, b) cast(U4, s4_(a) op s4_(b))
|
||||
|
||||
#define def_signed_op(id, op, width) FI_ U ## width id ## _s ## width(U ## width a, U ## width b) {return sop_ ## width(op, a, b); }
|
||||
#define def_signed_ops(id, op) def_signed_op(id, op, 1) def_signed_op(id, op, 2) def_signed_op(id, op, 4)
|
||||
def_signed_ops(add, +) def_signed_ops(sub, -)
|
||||
def_signed_ops(mut, *) def_signed_ops(div, /)
|
||||
def_signed_ops(gt, >) def_signed_ops(lt, <)
|
||||
def_signed_ops(ge, >=) def_signed_ops(le, <=)
|
||||
#undef def_signed_op
|
||||
#define def_signed_op(id,op,width) FI_ U ## width id ## _s ## width(U ## width a, U ## width b) {return sop_ ## width(op, a, b); }
|
||||
#define def_signed_ops(id,op) def_signed_op(id, op, 1) def_signed_op(id, op, 2) def_signed_op(id, op, 4)
|
||||
def_signed_ops(add, +)
|
||||
def_signed_ops(sub, -)
|
||||
def_signed_ops(mut, *)
|
||||
def_signed_ops(div, /)
|
||||
def_signed_ops(gt, >)
|
||||
def_signed_ops(lt, <)
|
||||
def_signed_ops(ge, >=)
|
||||
def_signed_ops(le, <=)
|
||||
#undef def_signed_ops
|
||||
#undef def_signed_op
|
||||
|
||||
#define def_generic_sop(op, a, ...) _Generic((a), U1: op ## _s1, U2: op ## _s2, U4: op ## _s4) (a, __VA_ARGS__)
|
||||
#define add_s(a,b) def_generic_sop(add,a,b)
|
||||
@@ -111,6 +172,27 @@ def_signed_ops(ge, >=) def_signed_ops(le, <=)
|
||||
#define lt_s(a,b) def_generic_sop(lt, a,b)
|
||||
#define ge_s(a,b) def_generic_sop(ge, a,b)
|
||||
#define le_s(a,b) def_generic_sop(le, a,b)
|
||||
#undef def_generic_sop
|
||||
|
||||
#define alignas _Alignas
|
||||
#define alignof _Alignof
|
||||
#define byte_pad(amount, ...) B1 glue(_PAD_, __VA_ARGS__) [amount]
|
||||
#define pcast(type, data) (C_(type*, & (data)) [0])
|
||||
|
||||
#define dbg_args(...) __VA_ARGS__
|
||||
|
||||
#pragma region Control Flow & Iteration
|
||||
#define each_iter(type, iter, end) (type iter = 0; iter < end; ++ iter)
|
||||
#define index_iter(type, iter, begin, op, end) (type iter = begin; iter op end; (begin < end ? ++ iter : -- iter))
|
||||
#define range_iter(iter,op,range) (T_((range).p0) iter = (range).p0; iter op (range).p1; ((range).p0 < (range).p1 ? ++ iter : -- iter))
|
||||
|
||||
#define defer(expr) for(U4 once= 1; once!=1;++ once,(expr)) // Basic do something after body
|
||||
#define scope(begin,end) for(U4 once=(1,(begin)); once!=1;++ once,(end )) // Do things before or after a scope
|
||||
#define defer_rewind(cursor) for(T_(cursor) sp=cursor,once=0; once!=1;++ once,cursor=sp) // Used with arenas/stacks
|
||||
#define defer_info(type,expr, ...) for(type info= {__VA_ARGS__}; info.once!=1;++info.once,(expr)) // Defer with tracked state
|
||||
|
||||
#define do_while(cond) for (U8 once=0; once!=1 || (cond); ++once)
|
||||
#pragma endregion Control Flow & Iteration
|
||||
|
||||
#define span_iter(type, iter, m_begin, op, m_end) ( \
|
||||
tmpl(Iter_Span,type) iter = { \
|
||||
@@ -119,44 +201,23 @@ def_signed_ops(ge, >=) def_signed_ops(le, <=)
|
||||
iter.cursor op iter.r.end; \
|
||||
++ iter.cursor \
|
||||
)
|
||||
#define def_span(type) \
|
||||
def_struct(tmpl( Span,type)) { type begin; type end; }; \
|
||||
typedef def_struct(tmpl(Iter_Span,type)) { tmpl(Span,type) r; type cursor; }
|
||||
#define Span_(type) \
|
||||
Struct_(tmpl( Span,type)) { type begin; type end; }; \
|
||||
typedef Struct_(tmpl(Iter_Span,type)) { tmpl(Span,type) r; type cursor; }
|
||||
|
||||
typedef def_span(S4);
|
||||
typedef def_span(U4);
|
||||
typedef Span_(S4);
|
||||
typedef Span_(U4);
|
||||
|
||||
typedef void def_proc(VoidFn) (void);
|
||||
#if 0
|
||||
#pragma region Debug
|
||||
#define debug_trap() __builtin_debugtrap()
|
||||
#if BUILD_DEBUG
|
||||
IA_ void assert(U8 cond) { if(cond){return;} else{debug_trap(); ms_exit_process(1);} }
|
||||
#else
|
||||
#define assert(cond)
|
||||
#endif
|
||||
#pragma endregion Debug
|
||||
#endif
|
||||
|
||||
typedef unsigned char def_tset(UTF8);
|
||||
typedef def_struct(Str8) { UTF8* ptr; U4 len; }; typedef Str8 def_tset(Slice_UTF8);
|
||||
typedef def_struct(Slice_Str8) { Str8* ptr; U4 len; };
|
||||
#define txt(string_literal) (Str8){ (UTF8*) string_literal, size_of(string_literal) - 1 }
|
||||
|
||||
#define def_Slice(type) def_struct(tmpl(Slice,type)) { type* ptr; U4 len; }
|
||||
#define slice_assert(slice) do { assert((slice).ptr != nullptr); assert((slice).len > 0); } while(0)
|
||||
#define slice_end(slice) ((slice).ptr + (slice).len)
|
||||
#define size_of_slice_type(slice) size_of((slice).ptr[0])
|
||||
|
||||
typedef def_Slice(void);
|
||||
typedef def_Slice(B1);
|
||||
#define slice_byte(slice) ((Slice_B1){cast(B1*, (slice).ptr), (slice).len * size_of_slice_type(slice)})
|
||||
#define slice_fmem(mem) ((Slice_B1){ mem, size_of(mem) })
|
||||
|
||||
void slice__copy(Slice_B1 dest, U4 dest_typewidth, Slice_B1 src, U4 src_typewidth);
|
||||
void slice__zero(Slice_B1 mem, U4 typewidth);
|
||||
#define slice_copy(dest, src) do { \
|
||||
static_assert(typeof_same(dest, src)); \
|
||||
slice__copy(slice_byte(dest), size_of_slice_type(dest), slice_byte(src), size_of_slice_type(src)); \
|
||||
} while (0)
|
||||
#define slice_zero(slice) slice__zero(slice_byte(slice), size_of_slice_type(slice))
|
||||
|
||||
#define slice_iter(container, iter) ( \
|
||||
typeof((container).ptr) iter = (container).ptr; \
|
||||
iter != slice_end(container); \
|
||||
++ iter \
|
||||
)
|
||||
#define slice_from_farray(type, ...) & (tmpl(Slice,type)) { \
|
||||
.ptr = farray_init(type, __VA_ARGS__), \
|
||||
.len = farray_len( farray_init(type, __VA_ARGS__)) \
|
||||
}
|
||||
#define GCC_OPTIMIZATION_DISABLE _Pragma("GCC push_options") _Pragma("GCC optimize(\"O0\")")
|
||||
#define GCC_OPTIMIZATION_ENABLE _Pragma("GCC pop_options")
|
||||
|
||||
@@ -1,31 +0,0 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "dsl.h"
|
||||
# include "memory.h"
|
||||
# include "strings.h"
|
||||
#endif
|
||||
|
||||
typedef def_struct(Opts_farena) {
|
||||
Str8 type_name;
|
||||
U4 alignment;
|
||||
};
|
||||
typedef def_struct(FArena) {
|
||||
void* start;
|
||||
U4 capacity;
|
||||
U4 used;
|
||||
};
|
||||
FArena farena_make (Slice_B1 mem);
|
||||
void farena_init (FArena* arena, Slice_B1 byte);
|
||||
Slice_B1 farena__push (FArena* arena, U4 amount, U4 type_width, Opts_farena* opts);
|
||||
void farena_reset (FArena* arena);
|
||||
void farena_rewind(FArena* arena, AllocatorSP save_point);
|
||||
AllocatorSP farena_save (FArena arena);
|
||||
|
||||
// void farena_allocator_proc(AllocatorProc_In in, AllocatorProc_Out* out);
|
||||
// #define ainfo_farena(arena) (AllocatorInfo){ .proc = farena_allocator_proc, .data = & arena }
|
||||
|
||||
#define farena_push(arena, type, ...) \
|
||||
cast(type*, farena__push(arena, size_of(type), 1, opt_args(Opts_farena_push, lit(stringify(type)), __VA_ARGS__))).ptr
|
||||
|
||||
#define farena_push_array(arena, type, amount, ...) \
|
||||
(Slice ## type){ farena__push(arena, size_of(type), amount, opt_args(Opts_farena_push, lit(stringify(type)), __VA_ARGS__)).ptr, amount }
|
||||
@@ -0,0 +1,457 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "dsl.h"
|
||||
#endif
|
||||
|
||||
/* ============================================================================
|
||||
* GCC INLINE ASM STATEMENT DSL
|
||||
* ============================================================================
|
||||
* A complete GCC inline-asm statement has up to 5 sections separated by `:`
|
||||
* asm volatile ( "code template" : OUTPUTS : INPUTS : CLOBBERS : GOTO_LABELS );
|
||||
*/
|
||||
|
||||
// Below are used purely for annotation.
|
||||
#define asm_out // OUTPUTS section /* cannot be used with asm_words */
|
||||
#define asm_in // INPUTS section /* can be appended onto after asm_words for pinned registers */
|
||||
#define asm_clobber // CLOBBERS section
|
||||
|
||||
// Pinned Registers after asm_words list (Semantic marker)
|
||||
// We aren't starting a new offical section, its just a continuation of the input section.
|
||||
// asm_words(...) // ".words " code word ids... : : code_words...
|
||||
// asm_rpins, r_use(r0), ... // , pinned registers...
|
||||
// asm_clobber:
|
||||
#define asm_rpins
|
||||
|
||||
/* --- Logic & Control Flow --- */
|
||||
|
||||
/* Annotation for the 'Goto' section of 'asm volatile goto'.
|
||||
* Allows you to jump from assembly directly to a C label. */
|
||||
#define asm_goto // Annotate the last `:` in an asm expression.
|
||||
|
||||
/* `asm_words(...)` dispatches into `_INL_<count>` to emit up to 99 encoded
|
||||
* instruction words. This is the "compiled-instruction" form of `asm_code`.
|
||||
*
|
||||
* Result is a 2-colon body WITHOUT the final clobber section:
|
||||
* ".word %c0, %c1, ..." : --- empty --- : "i"(p0), "i"(p1), ...
|
||||
* |------ code --------| |--- outputs ---| |------- inputs -------|
|
||||
*
|
||||
* Use it inside `asm volatile( ... )` like so:
|
||||
* asm volatile(
|
||||
* asm_words(w0, w1, w3)
|
||||
* asm_clobber: clobbers
|
||||
* )
|
||||
* which expands to:
|
||||
* asm volatile(".word %c0, %c1, %c2"
|
||||
* asm_out: // empty outputs
|
||||
* asm_in: "i"(w0), "i"(w1), "i"(w2)
|
||||
* asm_clobber: "$2", "$8", ...
|
||||
* )
|
||||
*/
|
||||
#define asm_words(...) m_expand(glue(GCC_ASM_INL_, GCC_ASM_COUNT_ARGS(__VA_ARGS__))(__VA_ARGS__))
|
||||
// Very nasty macro expansion. See the Cruft pragma region after all the DSL defines
|
||||
|
||||
/* reg_str(n) — Stringify an integer register id into the GCC asm
|
||||
* string form (e.g. 12 → "$12"). Use this anywhere GCC's parser
|
||||
* expects a literal string identifying a register: clobber lists,
|
||||
* asm templates, etc. The two-level macro is the standard preprocessor
|
||||
* idiom for forcing one level of expansion before stringify — without
|
||||
* it, `#n` would stringify the macro name `R_T4` to `"R_T4"` instead
|
||||
* of expanding `R_T4` to its value first.
|
||||
*
|
||||
* For declaring a register variable bound to a specific GPR, use the
|
||||
* `rgcc(n)` bundle from gcc_asm.h instead — it adds the `__asm__()`
|
||||
* qualifier around the string.
|
||||
*
|
||||
* register V3_S2* p0 __asm__(reg_str(R_T4)) = ...; // verbose
|
||||
* register V3_S2* p0 rgcc(R_T4) = ...; // bundled
|
||||
*
|
||||
* asm volatile("nop" : : : reg_str(R_RA), "memory"); // clobber list */
|
||||
#define rlit_stringfy(n) "$" stringify(n)
|
||||
#define rlit_tmpl(n) rlit_stringfy(tmpl(n,Code))
|
||||
#define rlit(n) rlit_tmpl(n)
|
||||
|
||||
/* ------------------------------------------------------------------------ *
|
||||
* rgcc(n) — GCC-specific bundle for register-variable declarations.
|
||||
*
|
||||
* Produces `__asm__(reg_str(tmpl(n, MipsCode)))` at expansion time.
|
||||
* The `tmpl(n, MipsCode)` indirection derives the preprocessor-visible `_Code`
|
||||
* form from the enum name (which the preprocessor can't expand on its own).
|
||||
* So a call is: register V3_S2* p rgcc(R_T4) = verts[0].ptr;
|
||||
* expands (via tmpl) to: register V3_S2* p __asm__(rlit(R_T4_Code)) = verts[0].ptr;
|
||||
* which (via reg_str) becomes: register V3_S2* p __asm__("$12") = verts[0].ptr;
|
||||
*
|
||||
* Why bundle the `__asm__()` wrapper?
|
||||
* - The integer R_T4 (= 12, via R_T4_Code) already indicates the register.
|
||||
* - The string "$12" is derived from it via reg_str, so they cannot drift apart.
|
||||
* - Spelling `__asm__(reg_str(R_T4_Code))` at every call site is noise.
|
||||
*
|
||||
* tmpl defined in dsl.h (the token-paste glue).
|
||||
* rgcc define here (gcc_asm.h) because the `__asm__` keyword is GCC-specific.
|
||||
* Anyone porting to a different compiler's asm dialect overrides rgcc,
|
||||
* and the integer→string derivation in rlit can be retargeted in one place.
|
||||
*
|
||||
* For clobber lists and asm-template strings, use the bare `rlit(R_T4_Code)`.
|
||||
* ------------------------------------------------------------------------ */
|
||||
#define rgcc(n) __asm__(rlit(n))
|
||||
|
||||
/* rgcc_ref(n) — GCC operand-reference form "%N". Not currently used
|
||||
* by the placeholder-pun macros (the .word bodies are fully baked
|
||||
* at compile time and have no runtime operand references), but kept
|
||||
* here for completeness in case a future asm template needs to refer
|
||||
* to a runtime input by position. Mirror of rgcc but produces "%N"
|
||||
* instead of "$N". */
|
||||
#define rgcc_ref_(n) "%" #n
|
||||
#define rgcc_ref(n) rgcc_ref_(n)
|
||||
|
||||
/* --- Register Constraint Aliases (for Pinned Variables) --- */
|
||||
|
||||
#define r_use(var) "r"(var) /* General Purpose Register */
|
||||
#define r_set(var) "=r"(var) /* Write-only output */
|
||||
#define r_mod(var) "+r"(var) /* Read-write */
|
||||
#define r_imm(val) "i"(val) /* Immediate / Constant */
|
||||
/* Memory: Forces GCC to sync the variable to RAM before the asm runs.
|
||||
* Essential for DMA buffers or when the hardware reads from memory. */
|
||||
#define r_mem(var) "m"(var)
|
||||
#define r_imm(val) "i"(val) /* Immediate: Forces a compile-time constant. */
|
||||
#define r_fpu(var) "f"(var) /* FPU (PS2/MIPS III/IV): Use for COP1 floating point registers. */
|
||||
#define r_acc(var) "a"(var) /* Accumulator: Use for HI/LO register results (multiplication/division). */
|
||||
|
||||
#define clb_mem_drain "memory"
|
||||
|
||||
// C Preprocessor Iterative Expansion Jank
|
||||
#pragma region Cruft
|
||||
|
||||
/* --- 1. The Argument Counter --- */
|
||||
#define GCC_ASM_COUNT_ARGS_IMPL( \
|
||||
_1, _2, _3, _4, _5, _6, _7, _8, _9, _10, \
|
||||
_11,_12,_13,_14,_15,_16,_17,_18,_19,_20, \
|
||||
_21,_22,_23,_24,_25,_26,_27,_28,_29,_30, \
|
||||
_31,_32,_33,_34,_35,_36,_37,_38,_39,_40, \
|
||||
_41,_42,_43,_44,_45,_46,_47,_48,_49,_50, \
|
||||
_51,_52,_53,_54,_55,_56,_57,_58,_59,_60, \
|
||||
_61,_62,_63,_64,_65,_66,_67,_68,_69,_70, \
|
||||
_71,_72,_73,_74,_75,_76,_77,_78,_79,_80, \
|
||||
_81,_82,_83,_84,_85,_86,_87,_88,_89,_90, \
|
||||
_91,_92,_93,_94,_95,_96,_97,_98,_99, N, ...) N
|
||||
|
||||
#define GCC_ASM_COUNT_ARGS(...) m_expand(GCC_ASM_COUNT_ARGS_IMPL(__VA_ARGS__, \
|
||||
99, 98, 97, 96, 95, 94, 93, 92, 91, 90, \
|
||||
89, 88, 87, 86, 85, 84, 83, 82, 81, 80, \
|
||||
79, 78, 77, 76, 75, 74, 73, 72, 71, 70, \
|
||||
69, 68, 67, 66, 65, 64, 63, 62, 61, 60, \
|
||||
59, 58, 57, 56, 55, 54, 53, 52, 51, 50, \
|
||||
49, 48, 47, 46, 45, 44, 43, 42, 41, 40, \
|
||||
39, 38, 37, 36, 35, 34, 33, 32, 31, 30, \
|
||||
29, 28, 27, 26, 25, 24, 23, 22, 21, 20, \
|
||||
19, 18, 17, 16, 15, 14, 13, 12, 11, 10, \
|
||||
9, 8, 7, 6, 5, 4, 3, 2, 1, 0))
|
||||
|
||||
/* --- 2. String Concatenation Helpers --- *
|
||||
* NOTE: we use `%0`, `%1`, ... not `%c0`, `%c1`, ... because GCC's
|
||||
* asm-parser rejects `%cN` in this position with "invalid use of '%c'".
|
||||
* The `%cN` form is for printing *character* constants; for arbitrary
|
||||
* integer immediates (the only kind `"i"(...)` produces), the plain
|
||||
* `%N` form is the right one. Both expand to the bare immediate.
|
||||
*/
|
||||
#define GCC_ASM_W1 "%0"
|
||||
#define GCC_ASM_W2 GCC_ASM_W1 ", %1"
|
||||
#define GCC_ASM_W3 GCC_ASM_W2 ", %2"
|
||||
#define GCC_ASM_W4 GCC_ASM_W3 ", %3"
|
||||
#define GCC_ASM_W5 GCC_ASM_W4 ", %4"
|
||||
#define GCC_ASM_W6 GCC_ASM_W5 ", %5"
|
||||
#define GCC_ASM_W7 GCC_ASM_W6 ", %6"
|
||||
#define GCC_ASM_W8 GCC_ASM_W7 ", %7"
|
||||
#define GCC_ASM_W9 GCC_ASM_W8 ", %8"
|
||||
#define GCC_ASM_W10 GCC_ASM_W9 ", %9"
|
||||
#define GCC_ASM_W11 GCC_ASM_W10 ", %10"
|
||||
#define GCC_ASM_W12 GCC_ASM_W11 ", %11"
|
||||
#define GCC_ASM_W13 GCC_ASM_W12 ", %12"
|
||||
#define GCC_ASM_W14 GCC_ASM_W13 ", %13"
|
||||
#define GCC_ASM_W15 GCC_ASM_W14 ", %14"
|
||||
#define GCC_ASM_W16 GCC_ASM_W15 ", %15"
|
||||
#define GCC_ASM_W17 GCC_ASM_W16 ", %16"
|
||||
#define GCC_ASM_W18 GCC_ASM_W17 ", %17"
|
||||
#define GCC_ASM_W19 GCC_ASM_W18 ", %18"
|
||||
#define GCC_ASM_W20 GCC_ASM_W19 ", %19"
|
||||
#define GCC_ASM_W21 GCC_ASM_W20 ", %20"
|
||||
#define GCC_ASM_W22 GCC_ASM_W21 ", %21"
|
||||
#define GCC_ASM_W23 GCC_ASM_W22 ", %22"
|
||||
#define GCC_ASM_W24 GCC_ASM_W23 ", %23"
|
||||
#define GCC_ASM_W25 GCC_ASM_W24 ", %24"
|
||||
#define GCC_ASM_W26 GCC_ASM_W25 ", %25"
|
||||
#define GCC_ASM_W27 GCC_ASM_W26 ", %26"
|
||||
#define GCC_ASM_W28 GCC_ASM_W27 ", %27"
|
||||
#define GCC_ASM_W29 GCC_ASM_W28 ", %28"
|
||||
#define GCC_ASM_W30 GCC_ASM_W29 ", %29"
|
||||
#define GCC_ASM_W31 GCC_ASM_W30 ", %30"
|
||||
#define GCC_ASM_W32 GCC_ASM_W31 ", %31"
|
||||
#define GCC_ASM_W33 GCC_ASM_W32 ", %32"
|
||||
#define GCC_ASM_W34 GCC_ASM_W33 ", %33"
|
||||
#define GCC_ASM_W35 GCC_ASM_W34 ", %34"
|
||||
#define GCC_ASM_W36 GCC_ASM_W35 ", %35"
|
||||
#define GCC_ASM_W37 GCC_ASM_W36 ", %36"
|
||||
#define GCC_ASM_W38 GCC_ASM_W37 ", %37"
|
||||
#define GCC_ASM_W39 GCC_ASM_W38 ", %38"
|
||||
#define GCC_ASM_W40 GCC_ASM_W39 ", %39"
|
||||
#define GCC_ASM_W41 GCC_ASM_W40 ", %40"
|
||||
#define GCC_ASM_W42 GCC_ASM_W41 ", %41"
|
||||
#define GCC_ASM_W43 GCC_ASM_W42 ", %42"
|
||||
#define GCC_ASM_W44 GCC_ASM_W43 ", %43"
|
||||
#define GCC_ASM_W45 GCC_ASM_W44 ", %44"
|
||||
#define GCC_ASM_W46 GCC_ASM_W45 ", %45"
|
||||
#define GCC_ASM_W47 GCC_ASM_W46 ", %46"
|
||||
#define GCC_ASM_W48 GCC_ASM_W47 ", %47"
|
||||
#define GCC_ASM_W49 GCC_ASM_W48 ", %48"
|
||||
#define GCC_ASM_W50 GCC_ASM_W49 ", %49"
|
||||
#define GCC_ASM_W51 GCC_ASM_W50 ", %50"
|
||||
#define GCC_ASM_W52 GCC_ASM_W51 ", %51"
|
||||
#define GCC_ASM_W53 GCC_ASM_W52 ", %52"
|
||||
#define GCC_ASM_W54 GCC_ASM_W53 ", %53"
|
||||
#define GCC_ASM_W55 GCC_ASM_W54 ", %54"
|
||||
#define GCC_ASM_W56 GCC_ASM_W55 ", %55"
|
||||
#define GCC_ASM_W57 GCC_ASM_W56 ", %56"
|
||||
#define GCC_ASM_W58 GCC_ASM_W57 ", %57"
|
||||
#define GCC_ASM_W59 GCC_ASM_W58 ", %58"
|
||||
#define GCC_ASM_W60 GCC_ASM_W59 ", %59"
|
||||
#define GCC_ASM_W61 GCC_ASM_W60 ", %60"
|
||||
#define GCC_ASM_W62 GCC_ASM_W61 ", %61"
|
||||
#define GCC_ASM_W63 GCC_ASM_W62 ", %62"
|
||||
#define GCC_ASM_W64 GCC_ASM_W63 ", %63"
|
||||
#define GCC_ASM_W65 GCC_ASM_W64 ", %64"
|
||||
#define GCC_ASM_W66 GCC_ASM_W65 ", %65"
|
||||
#define GCC_ASM_W67 GCC_ASM_W66 ", %66"
|
||||
#define GCC_ASM_W68 GCC_ASM_W67 ", %67"
|
||||
#define GCC_ASM_W69 GCC_ASM_W68 ", %68"
|
||||
#define GCC_ASM_W70 GCC_ASM_W69 ", %69"
|
||||
#define GCC_ASM_W71 GCC_ASM_W70 ", %70"
|
||||
#define GCC_ASM_W72 GCC_ASM_W71 ", %71"
|
||||
#define GCC_ASM_W73 GCC_ASM_W72 ", %72"
|
||||
#define GCC_ASM_W74 GCC_ASM_W73 ", %73"
|
||||
#define GCC_ASM_W75 GCC_ASM_W74 ", %74"
|
||||
#define GCC_ASM_W76 GCC_ASM_W75 ", %75"
|
||||
#define GCC_ASM_W77 GCC_ASM_W76 ", %76"
|
||||
#define GCC_ASM_W78 GCC_ASM_W77 ", %77"
|
||||
#define GCC_ASM_W79 GCC_ASM_W78 ", %78"
|
||||
#define GCC_ASM_W80 GCC_ASM_W79 ", %79"
|
||||
#define GCC_ASM_W81 GCC_ASM_W80 ", %80"
|
||||
#define GCC_ASM_W82 GCC_ASM_W81 ", %81"
|
||||
#define GCC_ASM_W83 GCC_ASM_W82 ", %82"
|
||||
#define GCC_ASM_W84 GCC_ASM_W83 ", %83"
|
||||
#define GCC_ASM_W85 GCC_ASM_W84 ", %84"
|
||||
#define GCC_ASM_W86 GCC_ASM_W85 ", %85"
|
||||
#define GCC_ASM_W87 GCC_ASM_W86 ", %86"
|
||||
#define GCC_ASM_W88 GCC_ASM_W87 ", %87"
|
||||
#define GCC_ASM_W89 GCC_ASM_W88 ", %88"
|
||||
#define GCC_ASM_W90 GCC_ASM_W89 ", %89"
|
||||
#define GCC_ASM_W91 GCC_ASM_W90 ", %90"
|
||||
#define GCC_ASM_W92 GCC_ASM_W91 ", %91"
|
||||
#define GCC_ASM_W93 GCC_ASM_W92 ", %92"
|
||||
#define GCC_ASM_W94 GCC_ASM_W93 ", %93"
|
||||
#define GCC_ASM_W95 GCC_ASM_W94 ", %94"
|
||||
#define GCC_ASM_W96 GCC_ASM_W95 ", %95"
|
||||
#define GCC_ASM_W97 GCC_ASM_W96 ", %96"
|
||||
#define GCC_ASM_W98 GCC_ASM_W97 ", %97"
|
||||
#define GCC_ASM_W99 GCC_ASM_W98 ", %98"
|
||||
|
||||
/* Utilizing cascading operand strings to compress the payload */
|
||||
#define GCC_ASM_I1(p0) "i"(p0)
|
||||
#define GCC_ASM_I2(p0, ...) "i"(p0), GCC_ASM_I1( __VA_ARGS__)
|
||||
#define GCC_ASM_I3(p0, ...) "i"(p0), GCC_ASM_I2( __VA_ARGS__)
|
||||
#define GCC_ASM_I4(p0, ...) "i"(p0), GCC_ASM_I3( __VA_ARGS__)
|
||||
#define GCC_ASM_I5(p0, ...) "i"(p0), GCC_ASM_I4( __VA_ARGS__)
|
||||
#define GCC_ASM_I6(p0, ...) "i"(p0), GCC_ASM_I5( __VA_ARGS__)
|
||||
#define GCC_ASM_I7(p0, ...) "i"(p0), GCC_ASM_I6( __VA_ARGS__)
|
||||
#define GCC_ASM_I8(p0, ...) "i"(p0), GCC_ASM_I7( __VA_ARGS__)
|
||||
#define GCC_ASM_I9(p0, ...) "i"(p0), GCC_ASM_I8( __VA_ARGS__)
|
||||
#define GCC_ASM_I10(p0, ...) "i"(p0), GCC_ASM_I9( __VA_ARGS__)
|
||||
#define GCC_ASM_I11(p0, ...) "i"(p0), GCC_ASM_I10(__VA_ARGS__)
|
||||
#define GCC_ASM_I12(p0, ...) "i"(p0), GCC_ASM_I11(__VA_ARGS__)
|
||||
#define GCC_ASM_I13(p0, ...) "i"(p0), GCC_ASM_I12(__VA_ARGS__)
|
||||
#define GCC_ASM_I14(p0, ...) "i"(p0), GCC_ASM_I13(__VA_ARGS__)
|
||||
#define GCC_ASM_I15(p0, ...) "i"(p0), GCC_ASM_I14(__VA_ARGS__)
|
||||
#define GCC_ASM_I16(p0, ...) "i"(p0), GCC_ASM_I15(__VA_ARGS__)
|
||||
#define GCC_ASM_I17(p0, ...) "i"(p0), GCC_ASM_I16(__VA_ARGS__)
|
||||
#define GCC_ASM_I18(p0, ...) "i"(p0), GCC_ASM_I17(__VA_ARGS__)
|
||||
#define GCC_ASM_I19(p0, ...) "i"(p0), GCC_ASM_I18(__VA_ARGS__)
|
||||
#define GCC_ASM_I20(p0, ...) "i"(p0), GCC_ASM_I19(__VA_ARGS__)
|
||||
#define GCC_ASM_I21(p0, ...) "i"(p0), GCC_ASM_I20(__VA_ARGS__)
|
||||
#define GCC_ASM_I22(p0, ...) "i"(p0), GCC_ASM_I21(__VA_ARGS__)
|
||||
#define GCC_ASM_I23(p0, ...) "i"(p0), GCC_ASM_I22(__VA_ARGS__)
|
||||
#define GCC_ASM_I24(p0, ...) "i"(p0), GCC_ASM_I23(__VA_ARGS__)
|
||||
#define GCC_ASM_I25(p0, ...) "i"(p0), GCC_ASM_I24(__VA_ARGS__)
|
||||
#define GCC_ASM_I26(p0, ...) "i"(p0), GCC_ASM_I25(__VA_ARGS__)
|
||||
#define GCC_ASM_I27(p0, ...) "i"(p0), GCC_ASM_I26(__VA_ARGS__)
|
||||
#define GCC_ASM_I28(p0, ...) "i"(p0), GCC_ASM_I27(__VA_ARGS__)
|
||||
#define GCC_ASM_I29(p0, ...) "i"(p0), GCC_ASM_I28(__VA_ARGS__)
|
||||
#define GCC_ASM_I30(p0, ...) "i"(p0), GCC_ASM_I29(__VA_ARGS__)
|
||||
#define GCC_ASM_I31(p0, ...) "i"(p0), GCC_ASM_I30(__VA_ARGS__)
|
||||
#define GCC_ASM_I32(p0, ...) "i"(p0), GCC_ASM_I31(__VA_ARGS__)
|
||||
#define GCC_ASM_I33(p0, ...) "i"(p0), GCC_ASM_I32(__VA_ARGS__)
|
||||
#define GCC_ASM_I34(p0, ...) "i"(p0), GCC_ASM_I33(__VA_ARGS__)
|
||||
#define GCC_ASM_I35(p0, ...) "i"(p0), GCC_ASM_I34(__VA_ARGS__)
|
||||
#define GCC_ASM_I36(p0, ...) "i"(p0), GCC_ASM_I35(__VA_ARGS__)
|
||||
#define GCC_ASM_I37(p0, ...) "i"(p0), GCC_ASM_I36(__VA_ARGS__)
|
||||
#define GCC_ASM_I38(p0, ...) "i"(p0), GCC_ASM_I37(__VA_ARGS__)
|
||||
#define GCC_ASM_I39(p0, ...) "i"(p0), GCC_ASM_I38(__VA_ARGS__)
|
||||
#define GCC_ASM_I40(p0, ...) "i"(p0), GCC_ASM_I39(__VA_ARGS__)
|
||||
#define GCC_ASM_I41(p0, ...) "i"(p0), GCC_ASM_I40(__VA_ARGS__)
|
||||
#define GCC_ASM_I42(p0, ...) "i"(p0), GCC_ASM_I41(__VA_ARGS__)
|
||||
#define GCC_ASM_I43(p0, ...) "i"(p0), GCC_ASM_I42(__VA_ARGS__)
|
||||
#define GCC_ASM_I44(p0, ...) "i"(p0), GCC_ASM_I43(__VA_ARGS__)
|
||||
#define GCC_ASM_I45(p0, ...) "i"(p0), GCC_ASM_I44(__VA_ARGS__)
|
||||
#define GCC_ASM_I46(p0, ...) "i"(p0), GCC_ASM_I45(__VA_ARGS__)
|
||||
#define GCC_ASM_I47(p0, ...) "i"(p0), GCC_ASM_I46(__VA_ARGS__)
|
||||
#define GCC_ASM_I48(p0, ...) "i"(p0), GCC_ASM_I47(__VA_ARGS__)
|
||||
#define GCC_ASM_I49(p0, ...) "i"(p0), GCC_ASM_I48(__VA_ARGS__)
|
||||
#define GCC_ASM_I50(p0, ...) "i"(p0), GCC_ASM_I49(__VA_ARGS__)
|
||||
#define GCC_ASM_I51(p0, ...) "i"(p0), GCC_ASM_I50(__VA_ARGS__)
|
||||
#define GCC_ASM_I52(p0, ...) "i"(p0), GCC_ASM_I51(__VA_ARGS__)
|
||||
#define GCC_ASM_I53(p0, ...) "i"(p0), GCC_ASM_I52(__VA_ARGS__)
|
||||
#define GCC_ASM_I54(p0, ...) "i"(p0), GCC_ASM_I53(__VA_ARGS__)
|
||||
#define GCC_ASM_I55(p0, ...) "i"(p0), GCC_ASM_I54(__VA_ARGS__)
|
||||
#define GCC_ASM_I56(p0, ...) "i"(p0), GCC_ASM_I55(__VA_ARGS__)
|
||||
#define GCC_ASM_I57(p0, ...) "i"(p0), GCC_ASM_I56(__VA_ARGS__)
|
||||
#define GCC_ASM_I58(p0, ...) "i"(p0), GCC_ASM_I57(__VA_ARGS__)
|
||||
#define GCC_ASM_I59(p0, ...) "i"(p0), GCC_ASM_I58(__VA_ARGS__)
|
||||
#define GCC_ASM_I60(p0, ...) "i"(p0), GCC_ASM_I59(__VA_ARGS__)
|
||||
#define GCC_ASM_I61(p0, ...) "i"(p0), GCC_ASM_I60(__VA_ARGS__)
|
||||
#define GCC_ASM_I62(p0, ...) "i"(p0), GCC_ASM_I61(__VA_ARGS__)
|
||||
#define GCC_ASM_I63(p0, ...) "i"(p0), GCC_ASM_I62(__VA_ARGS__)
|
||||
#define GCC_ASM_I64(p0, ...) "i"(p0), GCC_ASM_I63(__VA_ARGS__)
|
||||
#define GCC_ASM_I65(p0, ...) "i"(p0), GCC_ASM_I64(__VA_ARGS__)
|
||||
#define GCC_ASM_I66(p0, ...) "i"(p0), GCC_ASM_I65(__VA_ARGS__)
|
||||
#define GCC_ASM_I67(p0, ...) "i"(p0), GCC_ASM_I66(__VA_ARGS__)
|
||||
#define GCC_ASM_I68(p0, ...) "i"(p0), GCC_ASM_I67(__VA_ARGS__)
|
||||
#define GCC_ASM_I69(p0, ...) "i"(p0), GCC_ASM_I68(__VA_ARGS__)
|
||||
#define GCC_ASM_I70(p0, ...) "i"(p0), GCC_ASM_I69(__VA_ARGS__)
|
||||
#define GCC_ASM_I71(p0, ...) "i"(p0), GCC_ASM_I70(__VA_ARGS__)
|
||||
#define GCC_ASM_I72(p0, ...) "i"(p0), GCC_ASM_I71(__VA_ARGS__)
|
||||
#define GCC_ASM_I73(p0, ...) "i"(p0), GCC_ASM_I72(__VA_ARGS__)
|
||||
#define GCC_ASM_I74(p0, ...) "i"(p0), GCC_ASM_I73(__VA_ARGS__)
|
||||
#define GCC_ASM_I75(p0, ...) "i"(p0), GCC_ASM_I74(__VA_ARGS__)
|
||||
#define GCC_ASM_I76(p0, ...) "i"(p0), GCC_ASM_I75(__VA_ARGS__)
|
||||
#define GCC_ASM_I77(p0, ...) "i"(p0), GCC_ASM_I76(__VA_ARGS__)
|
||||
#define GCC_ASM_I78(p0, ...) "i"(p0), GCC_ASM_I77(__VA_ARGS__)
|
||||
#define GCC_ASM_I79(p0, ...) "i"(p0), GCC_ASM_I78(__VA_ARGS__)
|
||||
#define GCC_ASM_I80(p0, ...) "i"(p0), GCC_ASM_I79(__VA_ARGS__)
|
||||
#define GCC_ASM_I81(p0, ...) "i"(p0), GCC_ASM_I80(__VA_ARGS__)
|
||||
#define GCC_ASM_I82(p0, ...) "i"(p0), GCC_ASM_I81(__VA_ARGS__)
|
||||
#define GCC_ASM_I83(p0, ...) "i"(p0), GCC_ASM_I82(__VA_ARGS__)
|
||||
#define GCC_ASM_I84(p0, ...) "i"(p0), GCC_ASM_I83(__VA_ARGS__)
|
||||
#define GCC_ASM_I85(p0, ...) "i"(p0), GCC_ASM_I84(__VA_ARGS__)
|
||||
#define GCC_ASM_I86(p0, ...) "i"(p0), GCC_ASM_I85(__VA_ARGS__)
|
||||
#define GCC_ASM_I87(p0, ...) "i"(p0), GCC_ASM_I86(__VA_ARGS__)
|
||||
#define GCC_ASM_I88(p0, ...) "i"(p0), GCC_ASM_I87(__VA_ARGS__)
|
||||
#define GCC_ASM_I89(p0, ...) "i"(p0), GCC_ASM_I88(__VA_ARGS__)
|
||||
#define GCC_ASM_I90(p0, ...) "i"(p0), GCC_ASM_I89(__VA_ARGS__)
|
||||
#define GCC_ASM_I91(p0, ...) "i"(p0), GCC_ASM_I90(__VA_ARGS__)
|
||||
#define GCC_ASM_I92(p0, ...) "i"(p0), GCC_ASM_I91(__VA_ARGS__)
|
||||
#define GCC_ASM_I93(p0, ...) "i"(p0), GCC_ASM_I92(__VA_ARGS__)
|
||||
#define GCC_ASM_I94(p0, ...) "i"(p0), GCC_ASM_I93(__VA_ARGS__)
|
||||
#define GCC_ASM_I95(p0, ...) "i"(p0), GCC_ASM_I94(__VA_ARGS__)
|
||||
#define GCC_ASM_I96(p0, ...) "i"(p0), GCC_ASM_I95(__VA_ARGS__)
|
||||
#define GCC_ASM_I97(p0, ...) "i"(p0), GCC_ASM_I96(__VA_ARGS__)
|
||||
#define GCC_ASM_I98(p0, ...) "i"(p0), GCC_ASM_I97(__VA_ARGS__)
|
||||
#define GCC_ASM_I99(p0, ...) "i"(p0), GCC_ASM_I98(__VA_ARGS__)
|
||||
|
||||
#define GCC_ASM_INL_1( a) ".word " GCC_ASM_W1 : : GCC_ASM_I1( a)
|
||||
#define GCC_ASM_INL_2( a, ...) ".word " GCC_ASM_W2 : : GCC_ASM_I2( a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_3( a, ...) ".word " GCC_ASM_W3 : : GCC_ASM_I3( a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_4( a, ...) ".word " GCC_ASM_W4 : : GCC_ASM_I4( a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_5( a, ...) ".word " GCC_ASM_W5 : : GCC_ASM_I5( a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_6( a, ...) ".word " GCC_ASM_W6 : : GCC_ASM_I6( a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_7( a, ...) ".word " GCC_ASM_W7 : : GCC_ASM_I7( a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_8( a, ...) ".word " GCC_ASM_W8 : : GCC_ASM_I8( a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_9( a, ...) ".word " GCC_ASM_W9 : : GCC_ASM_I9( a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_10(a, ...) ".word " GCC_ASM_W10 : : GCC_ASM_I10(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_11(a, ...) ".word " GCC_ASM_W11 : : GCC_ASM_I11(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_12(a, ...) ".word " GCC_ASM_W12 : : GCC_ASM_I12(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_13(a, ...) ".word " GCC_ASM_W13 : : GCC_ASM_I13(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_14(a, ...) ".word " GCC_ASM_W14 : : GCC_ASM_I14(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_15(a, ...) ".word " GCC_ASM_W15 : : GCC_ASM_I15(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_16(a, ...) ".word " GCC_ASM_W16 : : GCC_ASM_I16(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_17(a, ...) ".word " GCC_ASM_W17 : : GCC_ASM_I17(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_18(a, ...) ".word " GCC_ASM_W18 : : GCC_ASM_I18(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_19(a, ...) ".word " GCC_ASM_W19 : : GCC_ASM_I19(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_20(a, ...) ".word " GCC_ASM_W20 : : GCC_ASM_I20(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_21(a, ...) ".word " GCC_ASM_W21 : : GCC_ASM_I21(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_22(a, ...) ".word " GCC_ASM_W22 : : GCC_ASM_I22(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_23(a, ...) ".word " GCC_ASM_W23 : : GCC_ASM_I23(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_24(a, ...) ".word " GCC_ASM_W24 : : GCC_ASM_I24(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_25(a, ...) ".word " GCC_ASM_W25 : : GCC_ASM_I25(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_26(a, ...) ".word " GCC_ASM_W26 : : GCC_ASM_I26(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_27(a, ...) ".word " GCC_ASM_W27 : : GCC_ASM_I27(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_28(a, ...) ".word " GCC_ASM_W28 : : GCC_ASM_I28(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_29(a, ...) ".word " GCC_ASM_W29 : : GCC_ASM_I29(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_30(a, ...) ".word " GCC_ASM_W30 : : GCC_ASM_I30(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_31(a, ...) ".word " GCC_ASM_W31 : : GCC_ASM_I31(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_32(a, ...) ".word " GCC_ASM_W32 : : GCC_ASM_I32(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_33(a, ...) ".word " GCC_ASM_W33 : : GCC_ASM_I33(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_34(a, ...) ".word " GCC_ASM_W34 : : GCC_ASM_I34(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_35(a, ...) ".word " GCC_ASM_W35 : : GCC_ASM_I35(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_36(a, ...) ".word " GCC_ASM_W36 : : GCC_ASM_I36(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_37(a, ...) ".word " GCC_ASM_W37 : : GCC_ASM_I37(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_38(a, ...) ".word " GCC_ASM_W38 : : GCC_ASM_I38(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_39(a, ...) ".word " GCC_ASM_W39 : : GCC_ASM_I39(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_40(a, ...) ".word " GCC_ASM_W40 : : GCC_ASM_I40(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_41(a, ...) ".word " GCC_ASM_W41 : : GCC_ASM_I41(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_42(a, ...) ".word " GCC_ASM_W42 : : GCC_ASM_I42(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_43(a, ...) ".word " GCC_ASM_W43 : : GCC_ASM_I43(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_44(a, ...) ".word " GCC_ASM_W44 : : GCC_ASM_I44(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_45(a, ...) ".word " GCC_ASM_W45 : : GCC_ASM_I45(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_46(a, ...) ".word " GCC_ASM_W46 : : GCC_ASM_I46(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_47(a, ...) ".word " GCC_ASM_W47 : : GCC_ASM_I47(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_48(a, ...) ".word " GCC_ASM_W48 : : GCC_ASM_I48(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_49(a, ...) ".word " GCC_ASM_W49 : : GCC_ASM_I49(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_50(a, ...) ".word " GCC_ASM_W50 : : GCC_ASM_I50(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_51(a, ...) ".word " GCC_ASM_W51 : : GCC_ASM_I51(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_52(a, ...) ".word " GCC_ASM_W52 : : GCC_ASM_I52(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_53(a, ...) ".word " GCC_ASM_W53 : : GCC_ASM_I53(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_54(a, ...) ".word " GCC_ASM_W54 : : GCC_ASM_I54(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_55(a, ...) ".word " GCC_ASM_W55 : : GCC_ASM_I55(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_56(a, ...) ".word " GCC_ASM_W56 : : GCC_ASM_I56(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_57(a, ...) ".word " GCC_ASM_W57 : : GCC_ASM_I57(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_58(a, ...) ".word " GCC_ASM_W58 : : GCC_ASM_I58(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_59(a, ...) ".word " GCC_ASM_W59 : : GCC_ASM_I59(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_60(a, ...) ".word " GCC_ASM_W60 : : GCC_ASM_I60(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_61(a, ...) ".word " GCC_ASM_W61 : : GCC_ASM_I61(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_62(a, ...) ".word " GCC_ASM_W62 : : GCC_ASM_I62(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_63(a, ...) ".word " GCC_ASM_W63 : : GCC_ASM_I63(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_64(a, ...) ".word " GCC_ASM_W64 : : GCC_ASM_I64(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_65(a, ...) ".word " GCC_ASM_W65 : : GCC_ASM_I65(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_66(a, ...) ".word " GCC_ASM_W66 : : GCC_ASM_I66(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_67(a, ...) ".word " GCC_ASM_W67 : : GCC_ASM_I67(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_68(a, ...) ".word " GCC_ASM_W68 : : GCC_ASM_I68(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_69(a, ...) ".word " GCC_ASM_W69 : : GCC_ASM_I69(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_70(a, ...) ".word " GCC_ASM_W70 : : GCC_ASM_I70(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_71(a, ...) ".word " GCC_ASM_W71 : : GCC_ASM_I71(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_72(a, ...) ".word " GCC_ASM_W72 : : GCC_ASM_I72(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_73(a, ...) ".word " GCC_ASM_W73 : : GCC_ASM_I73(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_74(a, ...) ".word " GCC_ASM_W74 : : GCC_ASM_I74(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_75(a, ...) ".word " GCC_ASM_W75 : : GCC_ASM_I75(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_76(a, ...) ".word " GCC_ASM_W76 : : GCC_ASM_I76(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_77(a, ...) ".word " GCC_ASM_W77 : : GCC_ASM_I77(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_78(a, ...) ".word " GCC_ASM_W78 : : GCC_ASM_I78(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_79(a, ...) ".word " GCC_ASM_W79 : : GCC_ASM_I79(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_80(a, ...) ".word " GCC_ASM_W80 : : GCC_ASM_I80(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_81(a, ...) ".word " GCC_ASM_W81 : : GCC_ASM_I81(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_82(a, ...) ".word " GCC_ASM_W82 : : GCC_ASM_I82(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_83(a, ...) ".word " GCC_ASM_W83 : : GCC_ASM_I83(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_84(a, ...) ".word " GCC_ASM_W84 : : GCC_ASM_I84(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_85(a, ...) ".word " GCC_ASM_W85 : : GCC_ASM_I85(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_86(a, ...) ".word " GCC_ASM_W86 : : GCC_ASM_I86(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_87(a, ...) ".word " GCC_ASM_W87 : : GCC_ASM_I87(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_88(a, ...) ".word " GCC_ASM_W88 : : GCC_ASM_I88(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_89(a, ...) ".word " GCC_ASM_W89 : : GCC_ASM_I89(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_90(a, ...) ".word " GCC_ASM_W90 : : GCC_ASM_I90(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_91(a, ...) ".word " GCC_ASM_W91 : : GCC_ASM_I91(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_92(a, ...) ".word " GCC_ASM_W92 : : GCC_ASM_I92(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_93(a, ...) ".word " GCC_ASM_W93 : : GCC_ASM_I93(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_94(a, ...) ".word " GCC_ASM_W94 : : GCC_ASM_I94(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_95(a, ...) ".word " GCC_ASM_W95 : : GCC_ASM_I95(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_96(a, ...) ".word " GCC_ASM_W96 : : GCC_ASM_I96(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_97(a, ...) ".word " GCC_ASM_W97 : : GCC_ASM_I97(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_98(a, ...) ".word " GCC_ASM_W98 : : GCC_ASM_I98(a, __VA_ARGS__)
|
||||
#define GCC_ASM_INL_99(a, ...) ".word " GCC_ASM_W99 : : GCC_ASM_I99(a, __VA_ARGS__)
|
||||
|
||||
#pragma endregion Cruft
|
||||
@@ -0,0 +1,139 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
#pragma once
|
||||
#endif
|
||||
// Auto-generated by ps1_meta.lua — DO NOT EDIT
|
||||
// Source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
||||
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
|
||||
|
||||
#ifndef WORD_COUNT
|
||||
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
||||
#endif
|
||||
|
||||
/* atom_dbg_skip */
|
||||
/* ---------------------------------------------------------------------------
|
||||
* MACRO ATOM Components (Reusable Assembly Components)
|
||||
* These do NOT yield. They are expanded inline inside Tape Atoms.
|
||||
* ---------------------------------------------------------------------------*/
|
||||
// The 'Yield' sequence for Tape Atoms (mac_yield).
|
||||
#define mac_yield(...) \
|
||||
load_word(R_AtomJmp, R_TapePtr, 0) \
|
||||
, add_ui_self( R_TapePtr, S_(MipsCode)) \
|
||||
, jump_reg( R_AtomJmp) \
|
||||
, nop
|
||||
WORD_COUNT(mac_yield, 4)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
/* Words: 3; Loads 3 S2 indices from the face array */
|
||||
#define mac_load_tri_indices(...) \
|
||||
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)) \
|
||||
, load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)) \
|
||||
, load_half_u(R_T2, R_FaceCursor, 2 * S_(S2))
|
||||
WORD_COUNT(mac_load_tri_indices, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
|
||||
#define mac_gte_load_tri_verts(...) \
|
||||
shift_lleft(R_AT, R_T0, v3s2_byteoff) \
|
||||
, add_u_self(R_AT, R_VertBase) \
|
||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||
, gte_mv_to_data_r(R_V0, C2_VXY0) \
|
||||
, gte_mv_to_data_r(R_V1, C2_VZ0) \
|
||||
, shift_lleft(R_AT, R_T1, v3s2_byteoff) \
|
||||
, add_u_self(R_AT, R_VertBase) \
|
||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||
, gte_mv_to_data_r(R_V0, C2_VXY1) \
|
||||
, gte_mv_to_data_r(R_V1, C2_VZ1) \
|
||||
, shift_lleft(R_AT, R_T2, v3s2_byteoff) \
|
||||
, add_u_self(R_AT, R_VertBase) \
|
||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||
, gte_mv_to_data_r(R_V0, C2_VXY2) \
|
||||
, gte_mv_to_data_r(R_V1, C2_VZ2)
|
||||
WORD_COUNT(mac_gte_load_tri_verts, 18)
|
||||
|
||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
|
||||
* Hardcoded for Poly_F3 (5 words). For Poly_G4, use ac_insert_ot_tag_g4. */
|
||||
#define mac_insert_ot_tag_f3(...) \
|
||||
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
|
||||
, add_u_self( R_T1, R_OtBase) /* T1 = & OrderingTable[OTZ] */ \
|
||||
, load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \
|
||||
, load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) /* V0 = (5 - 1) << 24 = 4 << 24 */ \
|
||||
, mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
|
||||
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
|
||||
, store_word( R_AT, R_PrimCursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \
|
||||
, shift_lleft( R_AT, R_PrimCursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
|
||||
, shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \
|
||||
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
|
||||
WORD_COUNT(mac_insert_ot_tag_f3, 11)
|
||||
|
||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
|
||||
* Hardcoded for Poly_G4 (9 words). For Poly_F3, use ac_insert_ot_tag_f3. */
|
||||
#define mac_insert_ot_tag_g4(...) \
|
||||
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
|
||||
, add_u_self( R_T1, R_OtBase) /* T1 = & OrderingTable[OTZ] */ \
|
||||
, load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \
|
||||
, load_upper_i(R_V0, (S_(Poly_G4)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) /* V0 = (9 - 1) << 24 = 8 << 24 */ \
|
||||
, mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
|
||||
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
|
||||
, store_word( R_AT, R_PrimCursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \
|
||||
, shift_lleft( R_AT, R_PrimCursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
|
||||
, shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \
|
||||
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
|
||||
WORD_COUNT(mac_insert_ot_tag_g4, 11)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_pack_color_word(off, cmd, r, g, b) \
|
||||
load_upper_i(R_AT, (cmd) << 8 | (b)) \
|
||||
, or_i_self( R_AT, ((g) << 8) | (r)) \
|
||||
, store_word( R_AT, R_PrimCursor, (off))
|
||||
WORD_COUNT(mac_pack_color_word, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_format_f3_color(r, g, b) \
|
||||
mac_pack_color_word(O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b)
|
||||
WORD_COUNT(mac_format_f3_color, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
|
||||
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
|
||||
#define mac_gte_store_f3_post_rtpt(...) \
|
||||
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_F3,p0)) \
|
||||
, gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_F3,p1)) \
|
||||
, gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_F3,p2))
|
||||
WORD_COUNT(mac_gte_store_f3_post_rtpt, 3)
|
||||
|
||||
#define mac_format_g4_color(r0, g0, b0, r1, g1, b1, r2, g2, b2, r3, g3, b3) \
|
||||
mac_pack_color_word(O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0) \
|
||||
, mac_pack_color_word(O_(Poly_G4,c1), 0, r1,g1,b1) \
|
||||
, mac_pack_color_word(O_(Poly_G4,c2), 0, r2,g2,b2) \
|
||||
, mac_pack_color_word(O_(Poly_G4,c3), 0, r3,g3,b3)
|
||||
WORD_COUNT(mac_format_g4_color, 12)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
|
||||
* G4 triangle portion to p0/p1/p2.
|
||||
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
|
||||
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2
|
||||
* get overwritten with v3 (RTPS writes only to SXY2, but to keep the
|
||||
* three registers aligned with v0/v1/v2 you must store before RTPS).
|
||||
* The macro name declares the pipeline position; check #6 (GTE state-
|
||||
* machine validation) verifies the call site matches the declaration. */
|
||||
#define mac_gte_store_g4_p012_post_rtpt_pre_rtps(...) \
|
||||
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_G4,p0)) \
|
||||
, gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_G4,p1)) \
|
||||
, gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2))
|
||||
WORD_COUNT(mac_gte_store_g4_p012_post_rtpt_pre_rtps, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
/* Words: 1; Stores the V3 screen coord to the G4's p3 slot.
|
||||
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its
|
||||
* single-vertex result to SXY2; SXY0 still holds v0.screen from the
|
||||
* earlier RTPT — DO NOT read SXY0 here, that's the bug this name
|
||||
* prevents).
|
||||
*/
|
||||
#define mac_gte_store_g4_p3_post_rtps(...) \
|
||||
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p3))
|
||||
WORD_COUNT(mac_gte_store_g4_p3_post_rtps, 1)
|
||||
|
||||
@@ -0,0 +1,9 @@
|
||||
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
|
||||
// Source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
||||
#pragma once
|
||||
|
||||
#pragma region lottes_tape
|
||||
|
||||
|
||||
#pragma endregion lottes_tape
|
||||
|
||||
+649
-80
@@ -1,97 +1,666 @@
|
||||
/* ============================================================================
|
||||
* duffle DSL Suffix Conventions
|
||||
* ============================================================================
|
||||
* Every mnemonic in this header follows the same suffix grammar:
|
||||
*
|
||||
* Primitive commands: gp0_cmd_poly_f3 = 0x20 (byte opcode)
|
||||
* Packed 32-bit cmd: gp0_word_poly_f3(r, g, b) (32-bit, shifted)
|
||||
*
|
||||
* Type ordering: domain?_(direction)?_action_target_modifier_type?
|
||||
* Examples: add_ui (add + unsigned + immediate)
|
||||
* add_s (add + signed, R-type implicit)
|
||||
* shift_lleft (shift + logical + left)
|
||||
* shift_aright (shift + arithmetic + right)
|
||||
* call_reg(rs) (call + register, $ra implicit)
|
||||
* gte_mv_to_data_r (gte + mv + to + data + register)
|
||||
* gte_lw_v0_xy(base) (gte + lw + v0 + xy)
|
||||
* load_upper_i (load-upper + immediate, unique verb)
|
||||
*
|
||||
* --- GPU-domain layer cake ---
|
||||
* Every gp.h macro follows the same 4-layer composition as mips.h and gte.h:
|
||||
* 4. Semantic encoders gp0_word_poly_f3(r,g,b)
|
||||
* 3. Composite encoders enc_color_word(cmd, r, g, b)
|
||||
* 2. Per-field encoders enc_gp0_color_r(r), enc_gp0_color_g(g), ...
|
||||
* 1. Bitfield layout consts gp0_color_red_shift = 0, gp0_color_red_mask = 0xFF
|
||||
* 0. Opcode IDs gp0_cmd_poly_f3 = 0x20
|
||||
*
|
||||
* Vendor mnemonics (gte_mtc2, gte_mfc2, etc.) are NOT in this header.
|
||||
* They live in the opt-in `gp_vendor_sym.h` for users who prefer the PSYQ-style names.
|
||||
* ============================================================================ */
|
||||
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "dsl.h"
|
||||
# include "math.h"
|
||||
# include "mips.h"
|
||||
#endif
|
||||
|
||||
typedef def_enum(U4, gp_Commands) {
|
||||
gcmd_Reset = 0b000,
|
||||
gcmd_Polygon = 0b001,
|
||||
gcmd_Line = 0b010,
|
||||
gcmd_Rect = 0b011,
|
||||
gcmd_VM_to_VM = 0b100,
|
||||
gcmd_CPU_to_VM = 0b101,
|
||||
gcmd_VM_to_CPU = 0b110,
|
||||
gcmd_Environment = 0b111,
|
||||
|
||||
gcmd_SetDrawMode = 0xE1,
|
||||
gcmd_SetTextureWindow = 0xE2,
|
||||
gcmd_SetDrawArea_TopLeft = 0xE3,
|
||||
gcmd_SetDrawArea_BotRight = 0xE4,
|
||||
gcmd_SetDrawOffset = 0xE5,
|
||||
gcmd_SetMaskBit = 0xE6,
|
||||
|
||||
gcmd_ResetCommandBuffer = 0x01,
|
||||
gcmd_AcknowledgeGPUInterrupt = 0x02,
|
||||
gcmd_DisplayEnable = 0x03,
|
||||
gcmd_DMA_Request = 0x04,
|
||||
gcmd_DispArea_Start = 0x05,
|
||||
gcmd_HorizontalDisplayRange = 0x06,
|
||||
gcmd_VerticalDisplayRange = 0x07,
|
||||
gcmd_DisplayMode = 0x08,
|
||||
|
||||
gcmd_SetVramSize = 0x09,
|
||||
};
|
||||
|
||||
#pragma region GPU Ports & Commands
|
||||
/* ============================================================================
|
||||
* Hardware MMIO Addresses
|
||||
* ============================================================================
|
||||
* PSX GPU has two 32-bit ports in the I/O register region at KSEG2
|
||||
* 0x1F800000+. GP0 (offset 0x10) is the data port (commands + params).
|
||||
* GP1 (offset 0x14) is the control port (status, ctrl writes).
|
||||
* ============================================================================ */
|
||||
/* IO base address (KSEG2 0x1F800000+ for the I/O register region).
|
||||
* The 16-bit upper half `IO_BASE_ADDR_HI16` is the form used by tape-side macros that pin a register
|
||||
* to hold the IO base and access ports via offsets:
|
||||
* `lui $reg, 0x1F80` (1 word) then `sw $data, GPIO_PORT*_OFFSET($reg)` (1 word).
|
||||
* Mirrors the `IO_BASE_ADDR equ 0x1F80` + `gpio_port0 equ 0x1810` pattern from graphics_hello/gp.s. */
|
||||
enum {
|
||||
gpio_port_0 = 0x1810,
|
||||
gpio_port_1 = 0x1814,
|
||||
IO_BASE_ADDR = 0x1F800000, /* full 32-bit I/O region base */
|
||||
IO_BASE_ADDR_HI16 = 0x1F80, /* fits in a single `lui $reg, 0x1F80` */
|
||||
|
||||
gcmd_offset = 24,
|
||||
/* Offsets from IO_BASE_ADDR to each port. Used by tape-side macros
|
||||
* that pin a register to IO_BASE_ADDR and access ports via offsets:
|
||||
* sw $data, GPIO_PORT0_OFFSET($io_base) ; write GP0
|
||||
* sw $data, GPIO_PORT1_OFFSET($io_base) ; write GP1 */
|
||||
GPIO_PORT0_OFFSET = 0x1810,
|
||||
GPIO_PORT1_OFFSET = 0x1814,
|
||||
|
||||
gp_Reset = (gcmd_Reset << gcmd_offset),
|
||||
|
||||
gp_DisplayEnabled = (gcmd_DisplayEnable << gcmd_offset | 0x0),
|
||||
gp_DisplayDisabled = (gcmd_DisplayEnable << gcmd_offset | 0x1),
|
||||
|
||||
gp_DMA_FIFO = 1,
|
||||
gp_DMA_CPU_to_GPU = 2,
|
||||
gp_DMA_GPU_to_CPU = 3,
|
||||
gp_DMA_Request = (gcmd_DMA_Request << gcmd_offset),
|
||||
|
||||
gp_HorizontalDisplayRange_3168_608 = (gcmd_HorizontalDisplayRange << gcmd_offset | 0xC60 << 12 | 0x260),
|
||||
|
||||
gp_VerticalDiplayRange = (gcmd_VerticalDisplayRange << gcmd_offset),
|
||||
gp_VerticalDisplayRange_264_24 = (gp_VerticalDiplayRange | 264 << 10 | 24),
|
||||
gp_VerticalDisplayRange_504_24 = (gp_VerticalDiplayRange | 504 << 10 | 24),
|
||||
|
||||
gp_DisplayMode = (gcmd_DisplayMode << gcmd_offset),
|
||||
gp_Disp_HRes_256 = (0x0),
|
||||
gp_Disp_HRes_320 = (0x1),
|
||||
gp_Disp_HRes_512 = (0x2),
|
||||
gp_Disp_HRes_640 = (0x3),
|
||||
gp_Disp_VRes_240 = (0x0 << 2),
|
||||
gp_Disp_VRes_480 = (0x1 << 2),
|
||||
gp_Disp_Color15 = (0x0 << 4),
|
||||
gp_Disp_Color24 = (0x1 << 4),
|
||||
gp_Disp_VInterlace = (0x1 << 5),
|
||||
gp_DisplayMode_320x240_15bit_NTSC = (gp_DisplayMode | gp_Disp_HRes_320 | gp_Disp_VRes_240 | gp_Disp_Color15),
|
||||
gp_DisplayMOde_640x480_24bbp_NTSC = (gp_DisplayMode | gp_Disp_HRes_640 | gp_Disp_VRes_480 | gp_Disp_Color24 | gp_Disp_VInterlace),
|
||||
|
||||
gp_DrawMode_DrawAllowed = 10,
|
||||
gp_SetDrawMode_DrawAllowed = (gcmd_SetDrawMode << gcmd_offset | 0x1 << gp_DrawMode_DrawAllowed),
|
||||
|
||||
gp_SetArea_TopLeft = (gcmd_SetDrawArea_TopLeft << gcmd_offset),
|
||||
gp_SetArea_BottomRight = (gcmd_SetDrawArea_BotRight << gcmd_offset),
|
||||
HW_GP0_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT0_OFFSET,
|
||||
HW_GP1_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT1_OFFSET,
|
||||
};
|
||||
|
||||
typedef def_struct(RGB8) { B1 r; B1 g; B1 b; };
|
||||
#define rgb8(r, g, b) (RGB8){ r, g, b }
|
||||
#define HW_GP0 C_(U4 V_*, HW_GP0_ADDR)
|
||||
#define HW_GP1 C_(U4 V_*, HW_GP1_ADDR)
|
||||
|
||||
typedef B1 gp_Pixel16[1];
|
||||
typedef B1 gp_Pixel24[3];
|
||||
#define gp0_send(word) (HW_GP0[0] = (word))
|
||||
#define gp1_send(word) (HW_GP1[0] = (word))
|
||||
|
||||
/* ============================================================================
|
||||
* GP0 command byte constants + Layer 1 (GPU bitfield shifts)
|
||||
* ============================================================================
|
||||
* 8-bit GP0 opcodes (the upper byte of a primitive's first word). These are the BYTE only.
|
||||
* The layer-1 bitfield-layout constants live in the same enum block so the encoder can reference them by name.
|
||||
* NO macro body past this point uses a raw shift or raw mask.
|
||||
* Every shift/width/mask is named here, named once.
|
||||
* Mirrors the OPCODE_SHIFT / RS_SHIFT / REG_MASK convention from mips.h.
|
||||
* ============================================================================ */
|
||||
enum {
|
||||
gp_b10_X = 0,
|
||||
gp_b10_Y = 10,
|
||||
gp_b16_X = 0,
|
||||
gp_b16_Y = 16,
|
||||
gp0_cmd_Nop = 0x00,
|
||||
|
||||
/* Cache management */
|
||||
gp0_cmd_ClearCache = 0x01,
|
||||
gp0_cmd_FillVram = 0x02,
|
||||
gp0_cmd_CopyVram = 0x80,
|
||||
gp0_cmd_CopyVramChained = 0x81,
|
||||
gp0_cmd_ReadVram = 0xC0,
|
||||
|
||||
/* Polygons */
|
||||
gp0_cmd_poly_f3 = 0x20, /* Flat Triangle */
|
||||
gp0_cmd_poly_ft3 = 0x24, /* Flat Textured Triangle */
|
||||
gp0_cmd_poly_g3 = 0x30, /* Gouraud Triangle */
|
||||
gp0_cmd_poly_gt3 = 0x34, /* Gouraud Textured Tri */
|
||||
gp0_cmd_poly_f4 = 0x28, /* Flat Quad */
|
||||
gp0_cmd_poly_ft4 = 0x2C, /* Flat Textured Quad */
|
||||
gp0_cmd_poly_g4 = 0x38, /* Gouraud Quad */
|
||||
gp0_cmd_poly_gt4 = 0x3C, /* Gouraud Textured Quad */
|
||||
|
||||
/* Lines */
|
||||
gp0_cmd_line_f2 = 0x40,
|
||||
gp0_cmd_line_g2 = 0x50,
|
||||
|
||||
/* Sprites + Tiles + Rects */
|
||||
gp0_cmd_sprt_1 = 0x64,
|
||||
gp0_cmd_sprt_8 = 0x74,
|
||||
gp0_cmd_sprt_16 = 0x7C,
|
||||
gp0_cmd_tile_1 = 0x60,
|
||||
gp0_cmd_tile_8 = 0x68,
|
||||
gp0_cmd_tile_16 = 0x70,
|
||||
|
||||
/* State setters (not drawing primitives; set render context). */
|
||||
gp0_cmd_DrawModeSetting = 0xE1, /* TPage / draw-mode (semi-trans, dither, etc.) */
|
||||
gp0_cmd_SetTextureWindow = 0xE2,
|
||||
gp0_cmd_SetDrawArea_TopLeft = 0xE3,
|
||||
gp0_cmd_SetDrawArea_BotRight = 0xE4,
|
||||
gp0_cmd_SetDrawOffset = 0xE5,
|
||||
gp0_cmd_SetMaskBit = 0xE6,
|
||||
|
||||
/* bitfield shifts / widths / masks ----
|
||||
* Generic GP0/GP1 command byte (upper 8 bits of every word sent to either port). */
|
||||
gp0_cmd_shift = 24,
|
||||
gp0_cmd_width = 8,
|
||||
gp0_cmd_mask = 0xFF,
|
||||
|
||||
/* Color word layout (lives in Poly_F3.color, Poly_G4.c0..c3, etc.):
|
||||
* bits 31..24 = command byte
|
||||
* bits 23..16 = BLUE
|
||||
* bits 15..08 = GREEN
|
||||
* bits 07..00 = RED (PSX GPU is BGR, NOT RGB) */
|
||||
gp0_color_cmd_shift = 24, gp0_color_cmd_width = 8, gp0_color_cmd_mask = 0xFF,
|
||||
gp0_color_blue_shift = 16, gp0_color_blue_width = 8, gp0_color_blue_mask = 0xFF,
|
||||
gp0_color_green_shift = 8, gp0_color_green_width = 8, gp0_color_green_mask = 0xFF,
|
||||
gp0_color_red_shift = 0, gp0_color_red_width = 8, gp0_color_red_mask = 0xFF,
|
||||
};
|
||||
|
||||
typedef def_struct(gp_Vec2) { U2 y; U2 x; };
|
||||
/* ============================================================================
|
||||
* Layer 1.5 (per-field encoders) + Layer 2 (composite) + Layer 3 (semantic GP0 word builders)
|
||||
* ============================================================================
|
||||
* Layer 1.5 encoders take one field's value, mask it to its own width, and shift it to its own position.
|
||||
* Mirrors `enc_op` / `enc_rs` / `enc_rt` in mips.h and `enc_gte_sf` / `enc_gte_mx` in gte.h.
|
||||
* Layer-2 composite encoders OR the per-field encoders together; layer-3 semantic macros delegate to the composites.
|
||||
* No raw shifts or magic numbers in any macro body below this point.
|
||||
* ============================================================================ */
|
||||
|
||||
#if 1
|
||||
void gp_screen_init(void) __asm__("gp_screen_init_asm");
|
||||
#else
|
||||
#define gp_screen_init() gp_screen_init_c11()
|
||||
#endif
|
||||
/* ---- Layer 1.5: per-field encoders ---- */
|
||||
#define enc_gp0_cmd(cmd) (((cmd) & gp0_cmd_mask) << gp0_cmd_shift)
|
||||
|
||||
#define enc_gp0_color_cmd(cmd) (((cmd) & gp0_color_cmd_mask) << gp0_color_cmd_shift)
|
||||
#define enc_gp0_color_r(r) (((r) & gp0_color_red_mask) << gp0_color_red_shift)
|
||||
#define enc_gp0_color_g(g) (((g) & gp0_color_green_mask) << gp0_color_green_shift)
|
||||
#define enc_gp0_color_b(b) (((b) & gp0_color_blue_mask) << gp0_color_blue_shift)
|
||||
|
||||
/* ---- Layer 2: composite encoders ---- */
|
||||
#define enc_color_word(cmd, r, g, b) (enc_gp0_color_cmd(cmd) | enc_gp0_color_r(r) | enc_gp0_color_g(g) | enc_gp0_color_b(b))
|
||||
|
||||
#define enc_gp0_cmd_word(cmd) (enc_gp0_cmd(cmd))
|
||||
|
||||
/* ---- Layer 3: semantic GP0 word builders ---- */
|
||||
|
||||
/* Pre-baked color+command words for all 8 polygon variants.
|
||||
* Mirrors `load_word` / `add_ui` / `jump_reg` style in mips.h. */
|
||||
#define gp0_word_poly_f3(r,g,b) enc_color_word(gp0_cmd_poly_f3, (r),(g),(b))
|
||||
#define gp0_word_poly_ft3(r,g,b) enc_color_word(gp0_cmd_poly_ft3, (r),(g),(b))
|
||||
#define gp0_word_poly_g3(r,g,b) enc_color_word(gp0_cmd_poly_g3, (r),(g),(b))
|
||||
#define gp0_word_poly_gt3(r,g,b) enc_color_word(gp0_cmd_poly_gt3, (r),(g),(b))
|
||||
#define gp0_word_poly_f4(r,g,b) enc_color_word(gp0_cmd_poly_f4, (r),(g),(b))
|
||||
#define gp0_word_poly_ft4(r,g,b) enc_color_word(gp0_cmd_poly_ft4, (r),(g),(b))
|
||||
#define gp0_word_poly_g4(r,g,b) enc_color_word(gp0_cmd_poly_g4, (r),(g),(b))
|
||||
#define gp0_word_poly_gt4(r,g,b) enc_color_word(gp0_cmd_poly_gt4, (r),(g),(b))
|
||||
|
||||
/* Cache management — bare-cmd words (no color/range payload). */
|
||||
#define gp0_word_clear_cache() enc_gp0_cmd_word(gp0_cmd_ClearCache)
|
||||
#define gp0_word_fill_vram() enc_gp0_cmd_word(gp0_cmd_FillVram)
|
||||
#define gp0_word_copy_vram() enc_gp0_cmd_word(gp0_cmd_CopyVram)
|
||||
#define gp0_word_read_vram() enc_gp0_cmd_word(gp0_cmd_ReadVram)
|
||||
|
||||
/* ============================================================================
|
||||
* GP1 command byte constants + Layer 1 (display-mode + range + draw-area bitfield shifts)
|
||||
* ============================================================================
|
||||
* GP1 status bits are read from HW_GP1;
|
||||
* ctrl writes use GP1 commands packed into 32-bit words
|
||||
* (cmd byte in the upper 8 bits via `enc_gp0_cmd(cmd)`).
|
||||
* ============================================================================ */
|
||||
enum {
|
||||
gp1_cmd_Reset = 0x00,
|
||||
gp1_cmd_ResetCmdBuffer = 0x01,
|
||||
gp1_cmd_AcknowledgeIRQ = 0x02,
|
||||
gp1_cmd_DisplayEnable = 0x03,
|
||||
gp1_cmd_DMADirection = 0x04,
|
||||
gp1_cmd_StartDisplayArea = 0x05,
|
||||
gp1_cmd_HorizontalDisplayRange = 0x06,
|
||||
gp1_cmd_VerticalDisplayRange = 0x07,
|
||||
gp1_cmd_DisplayMode = 0x08,
|
||||
/* Note: GP1 only has commands 0x00..0x08.
|
||||
* The state-setter commands (SetTextureWindow, * SetDrawArea*, SetDrawOffset, SetMaskBit)
|
||||
* live in the GP0 enum as * 0xE1..0xE6.
|
||||
* DrawArea word builders are below as GP0s * macros (since they emit GP0 commands). */
|
||||
|
||||
/* ---- Display-mode payload flags (per PSX-SPX §"GP1 Display Mode").
|
||||
* Bit positions match the encoder shifts below; values are the
|
||||
* *payload* bits only (the cmd byte is OR'd in by enc_gp1_disp_mode_word). */
|
||||
gp1_disp_HRes_256 = 0x0,
|
||||
gp1_disp_HRes_320 = 0x1,
|
||||
gp1_disp_HRes_512 = 0x2,
|
||||
gp1_disp_HRes_640 = 0x3,
|
||||
gp1_disp_VRes_240 = 0x0,
|
||||
gp1_disp_VRes_480 = 0x1,
|
||||
gp1_disp_Color15 = 0x0,
|
||||
gp1_disp_Color24 = 0x1,
|
||||
gp1_disp_VInterlace = 0x1,
|
||||
|
||||
/* ---- Layer 1: GP1 display-mode + range + draw-area shifts/masks ---- */
|
||||
gp1_disp_hres_shift = 0, gp1_disp_hres_width = 2, gp1_disp_hres_mask = 0x3,
|
||||
gp1_disp_vres_shift = 2, gp1_disp_vres_width = 1, gp1_disp_vres_mask = 0x1,
|
||||
gp1_disp_color_shift = 4, gp1_disp_color_width = 1, gp1_disp_color_mask = 0x1,
|
||||
gp1_disp_interlace_shift = 5, gp1_disp_interlace_width = 1, gp1_disp_interlace_mask = 0x1,
|
||||
|
||||
/* GP1 horizontal display range: bits 0..11 = X2, bits 12..23 = X1 */
|
||||
gp1_hrange_x1_shift = 12, gp1_hrange_x1_width = 12, gp1_hrange_x1_mask = 0xFFF,
|
||||
gp1_hrange_x2_shift = 0, gp1_hrange_x2_width = 12, gp1_hrange_x2_mask = 0xFFF,
|
||||
|
||||
/* GP1 vertical display range: bits 0..9 = Y2, bits 10..19 = Y1 */
|
||||
gp1_vrange_y1_shift = 10, gp1_vrange_y1_width = 10, gp1_vrange_y1_mask = 0x3FF,
|
||||
gp1_vrange_y2_shift = 0, gp1_vrange_y2_width = 10, gp1_vrange_y2_mask = 0x3FF,
|
||||
|
||||
/* GP1 draw area (top-left or bottom-right): bits 0..9 = X, bits 10..19 = Y
|
||||
* (10-bit signed — caller pre-signs and masks with the named mask) */
|
||||
gp1_draw_x_shift = 0, gp1_draw_x_width = 10, gp1_draw_x_mask = 0x3FF,
|
||||
gp1_draw_y_shift = 10, gp1_draw_y_width = 10, gp1_draw_y_mask = 0x3FF,
|
||||
};
|
||||
|
||||
/* ---- Layer 1.5: GP1 per-field encoders ---- */
|
||||
#define enc_gp1_disp_hres(h) (((h) & gp1_disp_hres_mask) << gp1_disp_hres_shift)
|
||||
#define enc_gp1_disp_vres(v) (((v) & gp1_disp_vres_mask) << gp1_disp_vres_shift)
|
||||
#define enc_gp1_disp_color(c) (((c) & gp1_disp_color_mask) << gp1_disp_color_shift)
|
||||
#define enc_gp1_disp_interlace(i) (((i) & gp1_disp_interlace_mask << gp1_disp_interlace_shift)
|
||||
|
||||
#define enc_gp1_hrange_x1(x1) (((x1) & gp1_hrange_x1_mask) << gp1_hrange_x1_shift)
|
||||
#define enc_gp1_hrange_x2(x2) (((x2) & gp1_hrange_x2_mask) << gp1_hrange_x2_shift)
|
||||
#define enc_gp1_vrange_y1(y1) (((y1) & gp1_vrange_y1_mask) << gp1_vrange_y1_shift)
|
||||
#define enc_gp1_vrange_y2(y2) (((y2) & gp1_vrange_y2_mask) << gp1_vrange_y2_shift)
|
||||
#define enc_gp1_draw_x(x) (((x) & gp1_draw_x_mask) << gp1_draw_x_shift)
|
||||
#define enc_gp1_draw_y(y) (((y) & gp1_draw_y_mask) << gp1_draw_y_shift)
|
||||
|
||||
/* ---- Layer 2: GP1 composite encoders ---- */
|
||||
#define enc_gp1_disp_mode_word(h, v, c, i) (enc_gp0_cmd(gp1_cmd_DisplayMode) | enc_gp1_disp_hres(h) | enc_gp1_disp_vres(v) | enc_gp1_disp_color(c) | enc_gp1_disp_interlace(i))
|
||||
#define enc_gp1_hrange_word(x1, x2) (enc_gp0_cmd(gp1_cmd_HorizontalDisplayRange) | enc_gp1_hrange_x1(x1) | enc_gp1_hrange_x2(x2))
|
||||
#define enc_gp1_vrange_word(y1, y2) (enc_gp0_cmd(gp1_cmd_VerticalDisplayRange) | enc_gp1_vrange_y1(y1) | enc_gp1_vrange_y2(y2))
|
||||
|
||||
/* ---- Layer 2: GP0 state-setter composite encoders ----
|
||||
* GP0(0xE3) SetDrawArea top-left and GP0(0xE4) SetDrawArea bottom-right both use the same X/Y 10-bit signed payload as GP1 DisplayRange. */
|
||||
#define enc_gp0_draw_area_tl_word(x, y) (enc_gp0_cmd(gp0_cmd_SetDrawArea_TopLeft) | enc_gp1_draw_x(x) | enc_gp1_draw_y(y))
|
||||
#define enc_gp0_draw_area_br_word(x, y) (enc_gp0_cmd(gp0_cmd_SetDrawArea_BotRight) | enc_gp1_draw_x(x) | enc_gp1_draw_y(y))
|
||||
|
||||
/* ---- Layer 3: GP1 semantic word builders ---- */
|
||||
#define gp1_word_display_enable(on) (enc_gp0_cmd(gp1_cmd_DisplayEnable) | ((on) & 1))
|
||||
#define gp1_word_display_disable() gp1_word_display_enable(0)
|
||||
#define gp1_word_display_mode_320x240_15bit_ntsc enc_gp1_disp_mode_word(gp1_disp_HRes_320, gp1_disp_VRes_240, gp1_disp_Color15, 0)
|
||||
#define gp1_word_display_mode_640x480_24bit_ntsc_interlaced enc_gp1_disp_mode_word(gp1_disp_HRes_640, gp1_disp_VRes_480, gp1_disp_Color24, gp1_disp_VInterlace)
|
||||
|
||||
#define gp1_word_horizontal_range(x1, x2) enc_gp1_hrange_word((x1), (x2))
|
||||
#define gp1_word_vertical_range(y1, y2) enc_gp1_vrange_word((y1), (y2))
|
||||
|
||||
/* ---- Layer 3: GP0 state-setter semantic word builders ---- */
|
||||
/* DrawArea: top-left = (X, Y), bottom-right = (X, Y) — X/Y in 10-bit signed.
|
||||
* Caller is responsible for sign-conversion before passing in. */
|
||||
#define gp0_word_draw_area_top_left(x, y) enc_gp0_draw_area_tl_word((x), (y))
|
||||
#define gp0_word_draw_area_bottom_right(x, y) enc_gp0_draw_area_br_word((x), (y))
|
||||
|
||||
/* ============================================================================
|
||||
* Pre-baked GPU state words
|
||||
* ============================================================================
|
||||
* Common command words for boot-time GPU init and standard display configurations.
|
||||
* ============================================================================ */
|
||||
|
||||
/* ---- Display enable (1-bit payload on DisplayEnable cmd) ---- */
|
||||
#define gp1_word_display_enabled enc_gp0_cmd_word(gp1_cmd_DisplayEnable)
|
||||
#define gp1_word_display_disabled (enc_gp0_cmd_word(gp1_cmd_DisplayEnable) | 1)
|
||||
|
||||
/* ---- DMA direction (2-bit payload on DMADirection cmd 0x04) ---- */
|
||||
enum {
|
||||
gp1_dma_dir_Off = 0,
|
||||
gp1_dma_dir_FIFO = 1,
|
||||
gp1_dma_dir_CPU_to_GPU = 2,
|
||||
gp1_dma_dir_GPUREAD_to_CPU = 3,
|
||||
};
|
||||
#define gp1_word_dma_direction(dir) (enc_gp0_cmd(gp1_cmd_DMADirection) | ((dir) & 0x3))
|
||||
|
||||
/* ---- Standard display ranges (NTSC + PAL pre-baked) ---- */
|
||||
/* Horizontal range values are in video clock units (8 units/pixel); vertical range values are scanline numbers. */
|
||||
enum {
|
||||
/* NTSC horizontal range: X1=608, X2=3168 */
|
||||
gp1_hrange_NTSC_x1 = 0x260,
|
||||
gp1_hrange_NTSC_x2 = 0xC60,
|
||||
/* PAL horizontal range (same as NTSC for most CRTs) */
|
||||
gp1_hrange_PAL_x1 = 0x260,
|
||||
gp1_hrange_PAL_x2 = 0xC60,
|
||||
|
||||
/* NTSC vertical range: Y1=24, Y2=264 */
|
||||
gp1_vrange_NTSC_y1 = 24,
|
||||
gp1_vrange_NTSC_y2 = 264,
|
||||
/* PAL vertical range: Y1=24, Y2=504 */
|
||||
gp1_vrange_PAL_y1 = 24,
|
||||
gp1_vrange_PAL_y2 = 504,
|
||||
};
|
||||
|
||||
#define gp1_word_horizontal_range_ntsc enc_gp1_hrange_word(gp1_hrange_NTSC_x1, gp1_hrange_NTSC_x2)
|
||||
#define gp1_word_horizontal_range_pal enc_gp1_hrange_word(gp1_hrange_PAL_x1, gp1_hrange_PAL_x2)
|
||||
#define gp1_word_vertical_range_ntsc enc_gp1_vrange_word(gp1_vrange_NTSC_y1, gp1_vrange_NTSC_y2)
|
||||
#define gp1_word_vertical_range_pal enc_gp1_vrange_word(gp1_vrange_PAL_y1, gp1_vrange_PAL_y2)
|
||||
|
||||
/* ---- Draw-mode setting (TPage / draw-area allowance) ---- */
|
||||
/* The "drawing enabled" word is the standard post-init state. */
|
||||
enum {
|
||||
gp0_DrawMode_DrawToDispBit = 10,
|
||||
};
|
||||
#define gp0_word_draw_mode_drawing_allowed (enc_gp0_cmd(gp0_cmd_DrawModeSetting) | (1 << gp0_DrawMode_DrawToDispBit))
|
||||
|
||||
/* ---- DrawArea pre-baked at origin (0,0) and full screen (320x240) ---- */
|
||||
#define gp0_word_draw_area_top_left_origin enc_gp0_draw_area_tl_word(0, 0)
|
||||
#define gp0_word_draw_area_bottom_right_320x240 enc_gp0_draw_area_br_word(320, 240)
|
||||
#define gp0_word_draw_area_bottom_right_640x480 enc_gp0_draw_area_br_word(640, 480)
|
||||
|
||||
#pragma endregion GPU Ports & Commands
|
||||
|
||||
#pragma region GPU Status
|
||||
/* ============================================================================
|
||||
* GPU status register bits
|
||||
* ============================================================================
|
||||
* Read from HW_GP1; the lower bits are DMA-block-size (variable-width).
|
||||
* ============================================================================ */
|
||||
enum {
|
||||
gp1_Status_BitReady = 31,
|
||||
gp1_Status_BitSendingDMA = 25,
|
||||
gp1_Status_DMABlockSizeShift = 0,
|
||||
};
|
||||
|
||||
#define gp1_status_is_ready() ((HW_GP1[0] >> gp1_Status_BitReady) & 1)
|
||||
#define gp1_status_is_sending_dma() ((HW_GP1[0] >> gp1_Status_BitSendingDMA) & 1)
|
||||
#pragma endregion GPU Status
|
||||
|
||||
#pragma region Primitives
|
||||
/* ============================================================================
|
||||
* Primitive structs (8 polygon variants + tag)
|
||||
* ============================================================================
|
||||
* Each struct follows the GPU-documented memory layout for the corresponding primitive command.
|
||||
* The PolyTag is the OT-link header; the rest of the struct is the primitive's body.
|
||||
*
|
||||
* The current working layouts match the existing demo
|
||||
* (floor_tri uses Poly_F3; cube_tri uses Poly_G4).
|
||||
* They are NOT necessarily byte-identical to the PSX-SPX reference layout.
|
||||
* The demo layout uses color+vertex interleaving that doesn't match the standard PSX SDK file format.
|
||||
* For PSX-SDK file compatibility, the textured variants (FT*, GT*) would need layout adjustments.
|
||||
* ============================================================================ */
|
||||
|
||||
/* ---------- RGB8 (3-byte packed color) ---------- */
|
||||
typedef Struct_(RGB8) { B1 r; B1 g; B1 b; };
|
||||
#define rgb8(r,g,b) ((RGB8){r,g,b})
|
||||
|
||||
/* ---------- PolyTag (the OT-link header; 1 word) ---------- */
|
||||
enum {
|
||||
PolyTag_len_bits = 8,
|
||||
PolyTag_addr_bits = 24,
|
||||
};
|
||||
typedef Struct_(PolyTag) {
|
||||
union {
|
||||
U4 code;
|
||||
struct {
|
||||
U4 addr: 24;
|
||||
U4 len: 8;
|
||||
};
|
||||
};
|
||||
};
|
||||
|
||||
/* DSL cast convention: every cast uses `C_()`, every pointer qualifier is `R_` (restrict) or `V_` (volatile).
|
||||
* No raw C-style casts. RHS values are assumed to be `U4` — caller passes a `U4` directly. */
|
||||
#define set_len(tag,v) (C_(PolyTag_R,tag)->len = u4_(v))
|
||||
#define set_addr(tag,v) (C_(PolyTag_R,tag)->addr = u4_(v))
|
||||
/* `set_code` is no longer in the new PolyTag design — the code byte lives in the primitive body
|
||||
* (e.g. `((Poly_F3*)(p))->code`), not in the tag.
|
||||
* Use the typed primitive structs (Poly_F3, Poly_G4, etc.) and the `set_poly_*` setters,
|
||||
* which set both the tag's length and the code. */
|
||||
#define get_len(tag) C_(U4,C_(PolyTag_R,tag)->len)
|
||||
#define get_addr(tag) C_(U4,C_(PolyTag_R,tag)->addr)
|
||||
|
||||
/* ---------- Poly_F3 (Flat Triangle; 5 words) ---------- */
|
||||
typedef Struct_(Poly_F3) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
union {
|
||||
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; };
|
||||
A3_V2_S2 points;
|
||||
};
|
||||
};
|
||||
|
||||
/* ---------- Poly_F4 (Flat Quad; 6 words) ---------- */
|
||||
typedef Struct_(Poly_F4) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
union {
|
||||
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; V2_S2 p3; };
|
||||
A4_V2_S2 points;
|
||||
};
|
||||
};
|
||||
|
||||
/* ---------- Poly_G3 (Gouraud Triangle; 7 words) ---------- */
|
||||
typedef Struct_(Poly_G3) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2;
|
||||
};
|
||||
|
||||
/* ---------- Poly_G4 (Gouraud Quad; 9 words) ---------- */
|
||||
typedef Struct_(Poly_G4) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2; RGB8 c3; B1 pad3;
|
||||
V2_S2 p3;
|
||||
};
|
||||
|
||||
/* ---------- Poly_FT3 (Flat Textured Triangle; placeholder layout) ---------- */
|
||||
/* TODO(Ed): verify the textured-variant layout against PSX-SPX when needed. */
|
||||
typedef Struct_(Poly_FT3) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 p0; U1 u0; U1 v0;
|
||||
V2_S2 p1; U1 u1; U1 v1;
|
||||
V2_S2 p2; U1 u2; U1 v2;
|
||||
};
|
||||
|
||||
/* ---------- Poly_FT4 (Flat Textured Quad) ---------- */
|
||||
typedef Struct_(Poly_FT4) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 p0; U1 u0; U1 v0;
|
||||
V2_S2 p1; U1 u1; U1 v1;
|
||||
V2_S2 p2; U1 u2; U1 v2;
|
||||
V2_S2 p3; U1 u3; U1 v3;
|
||||
};
|
||||
|
||||
/* ---------- Poly_GT3 (Gouraud Textured Triangle) ---------- */
|
||||
typedef Struct_(Poly_GT3) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 tp0; U1 u0; U1 v0;
|
||||
V2_S2 tp1; U1 u1; U1 v1;
|
||||
V2_S2 tp2; U1 u2; U1 v2;
|
||||
};
|
||||
|
||||
/* ---------- Poly_GT4 (Gouraud Textured Quad) ---------- */
|
||||
typedef Struct_(Poly_GT4) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2; RGB8 c3; B1 pad3;
|
||||
V2_S2 p3;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 tp0; U1 u0; U1 v0;
|
||||
V2_S2 tp1; U1 u1; U1 v1;
|
||||
V2_S2 tp2; U1 u2; U1 v2;
|
||||
V2_S2 tp3; U1 u3; U1 v3;
|
||||
};
|
||||
|
||||
/* ---------- Primitive setters (C-level) ----------
|
||||
* DSL cast convention: every cast via C_(), every pointer via R_/V_. */
|
||||
#define set_poly_f3(p) set_len(p, 4), C_(Poly_F3_R, p)->code = gp0_cmd_poly_f3
|
||||
#define set_poly_ft3(p) set_len(p, 7), C_(Poly_FT3_R,p)->code = gp0_cmd_poly_ft3
|
||||
#define set_poly_g3(p) set_len(p, 6), C_(Poly_G3_R, p)->code = gp0_cmd_poly_g3
|
||||
#define set_poly_gt3(p) set_len(p, 9), C_(Poly_GT3_R,p)->code = gp0_cmd_poly_gt3
|
||||
#define set_poly_f4(p) set_len(p, 5), C_(Poly_F4_R, p)->code = gp0_cmd_poly_f4
|
||||
#define set_poly_ft4(p) set_len(p, 9), C_(Poly_FT4_R,p)->code = gp0_cmd_poly_ft4
|
||||
#define set_poly_g4(p) set_len(p, 8), C_(Poly_G4_R, p)->code = gp0_cmd_poly_g4
|
||||
#define set_poly_gt4(p) set_len(p, 12), C_(Poly_GT4_R,p)->code = gp0_cmd_poly_gt4
|
||||
|
||||
/* ---------- Ordering table ops ---------- */
|
||||
#define orderingtbl_add_primitive(ot, p) set_addr(p, get_addr(ot)), set_addr(ot, p)
|
||||
#define orderingtbl_add_primitives(ot, p0, p1) set_addr(p1, get_addr(ot)), set_addr(ot, p0)
|
||||
|
||||
#pragma endregion Primitives
|
||||
|
||||
#pragma region TPage
|
||||
/* ============================================================================
|
||||
* Texture Page (TPage) bit layout
|
||||
* ============================================================================
|
||||
* The TPage data word sent via GP0(0x2X) has:
|
||||
* bits 0..3 = texture page X (4 bits, 64-px units, 0..16)
|
||||
* bit 4 = texture page Y (1 bit, 64-px units, 0/1)
|
||||
* bits 5..6 = semi-transparency (2 bits, 0..3)
|
||||
* bits 7..8 = texture page colors (2 bits, 4bpp/8bpp/16bpp/2bpp-mixed)
|
||||
* bit 9 = dither (1 bit, 0/1)
|
||||
* bit 10 = drawing to display area (1 bit)
|
||||
* bit 11 = texture disable (1 bit)
|
||||
* bits 12..31 = reserved (zero)
|
||||
* ============================================================================ */
|
||||
enum {
|
||||
/* ---- Layer 1: TPage bitfield shifts / widths / masks ---- */
|
||||
gp0_tpage_x_shift = 0, gp0_tpage_x_width = 4, gp0_tpage_x_mask = 0xF,
|
||||
gp0_tpage_y_shift = 4, gp0_tpage_y_width = 1, gp0_tpage_y_mask = 0x1,
|
||||
gp0_tpage_semi_trans_shift = 5, gp0_tpage_semi_trans_width = 2, gp0_tpage_semi_trans_mask = 0x3,
|
||||
gp0_tpage_color_depth_shift = 7, gp0_tpage_color_depth_width = 2, gp0_tpage_color_depth_mask = 0x3,
|
||||
gp0_tpage_dither_shift = 9, gp0_tpage_dither_width = 1, gp0_tpage_dither_mask = 0x1,
|
||||
gp0_tpage_draw_to_disp_shift = 10, gp0_tpage_draw_to_disp_width = 1, gp0_tpage_draw_to_disp_mask = 0x1,
|
||||
gp0_tpage_tex_disable_shift = 11, gp0_tpage_tex_disable_width = 1, gp0_tpage_tex_disable_mask = 0x1,
|
||||
|
||||
/* TPage color-depth payload values (NOT bit positions — these go in
|
||||
* the 2-bit field at gp0_tpage_color_depth_shift). */
|
||||
gp0_tpage_color_4bpp = 0x0,
|
||||
gp0_tpage_color_8bpp = 0x1,
|
||||
gp0_tpage_color_16bpp = 0x2,
|
||||
|
||||
/* TPage semi-transparency mode payload values (NOT bit positions). */
|
||||
gp0_tpage_semi_trans_none = 0x0,
|
||||
gp0_tpage_semi_trans_alpha = 0x1,
|
||||
gp0_tpage_semi_trans_add = 0x2,
|
||||
gp0_tpage_semi_trans_sub = 0x3,
|
||||
};
|
||||
|
||||
/* ---- Layer 1.5: TPage per-field encoders. Mirrors enc_gte_sf/mx/v in gte.h. ---- */
|
||||
#define enc_gp0_tpage_x(x) (((x) & gp0_tpage_x_mask) << gp0_tpage_x_shift)
|
||||
#define enc_gp0_tpage_y(y) (((y) & gp0_tpage_y_mask) << gp0_tpage_y_shift)
|
||||
#define enc_gp0_tpage_semi_trans(s) (((s) & gp0_tpage_semi_trans_mask) << gp0_tpage_semi_trans_shift)
|
||||
#define enc_gp0_tpage_color_depth(c) (((c) & gp0_tpage_color_depth_mask) << gp0_tpage_color_depth_shift)
|
||||
#define enc_gp0_tpage_dither(d) (((d) & gp0_tpage_dither_mask) << gp0_tpage_dither_shift)
|
||||
#define enc_gp0_tpage_draw_to_disp(d) (((d) & gp0_tpage_draw_to_disp_mask) << gp0_tpage_draw_to_disp_shift)
|
||||
#define enc_gp0_tpage_tex_disable(t) (((t) & gp0_tpage_tex_disable_mask) << gp0_tpage_tex_disable_shift)
|
||||
|
||||
/* ---- Layer 2: TPage composite encoder. Mirrors enc_gte_cmdw in gte.h ---- */
|
||||
#define enc_gp0_tpage_word(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
|
||||
(enc_gp0_tpage_x(x) \
|
||||
| enc_gp0_tpage_y(y) \
|
||||
| enc_gp0_tpage_semi_trans(semi_trans) \
|
||||
| enc_gp0_tpage_color_depth(color_depth) \
|
||||
| enc_gp0_tpage_dither(dither) \
|
||||
| enc_gp0_tpage_draw_to_disp(draw_to_disp) \
|
||||
| enc_gp0_tpage_tex_disable(tex_disable))
|
||||
|
||||
typedef Struct_(TexturePage) { U4 raw; };
|
||||
|
||||
/* ---- Layer 3: TPage semantic word builder ---- */
|
||||
#define gp0_word_tpage(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
|
||||
enc_gp0_tpage_word((x), (y), (semi_trans), (color_depth), (dither), (draw_to_disp), (tex_disable))
|
||||
#pragma endregion TPage
|
||||
|
||||
#pragma region CLUT
|
||||
/* ============================================================================
|
||||
* CLUT (Color Look-Up Table) semantics
|
||||
* ============================================================================
|
||||
* CLUT is loaded into VRAM by sending a GP0 command whose payload is:
|
||||
* bits 0..5 = Y in 16-px units (palette row)
|
||||
* bits 6..14 = X in 16-px units (palette column)
|
||||
* bits 15..23 = reserved (zero)
|
||||
* bits 24..31 = command byte — 0x20 (4bpp load) or 0x25 (8bpp load)
|
||||
* ============================================================================ */
|
||||
enum {
|
||||
/* ---- Layer 1: CLUT bitfield shifts / widths / masks ---- */
|
||||
gp0_clut_y_shift = 0, gp0_clut_y_width = 6, gp0_clut_y_mask = 0x3F,
|
||||
gp0_clut_x_shift = 6, gp0_clut_x_width = 9, gp0_clut_x_mask = 0x1FF,
|
||||
/* CLUT-load cmd-byte variants — the upper byte of the GP0 word. */
|
||||
gp0_clut_cmd_Load4bpp = 0x20,
|
||||
gp0_clut_cmd_Load8bpp = 0x25,
|
||||
};
|
||||
|
||||
/* ---- Layer 1.5: CLUT per-field encoders ---- */
|
||||
#define enc_gp0_clut_x(x) (((x) & gp0_clut_x_mask) << gp0_clut_x_shift)
|
||||
#define enc_gp0_clut_y(y) (((y) & gp0_clut_y_mask) << gp0_clut_y_shift)
|
||||
|
||||
/* ---- Layer 2: CLUT composite encoder ---- */
|
||||
#define enc_gp0_clut_word(cmd, x, y) (enc_gp0_cmd(cmd) | enc_gp0_clut_x(x) | enc_gp0_clut_y(y))
|
||||
|
||||
/* ---- Layer 3: CLUT semantic word builders — one per depth variant,
|
||||
* named cmd-byte (no opaque ternary). ---- */
|
||||
#define gp0_word_clut_load_4bpp(x, y) enc_gp0_clut_word(gp0_clut_cmd_Load4bpp, (x), (y))
|
||||
#define gp0_word_clut_load_8bpp(x, y) enc_gp0_clut_word(gp0_clut_cmd_Load8bpp, (x), (y))
|
||||
#pragma endregion CLUT
|
||||
|
||||
#pragma region TIM File Format
|
||||
/* ============================================================================
|
||||
* TIM file format constants and headers
|
||||
* ============================================================================
|
||||
* TIM (Sony .TIM texture image) file structure:
|
||||
* +0x00 U4 file_id (always 0x10 = TIM magic)
|
||||
* +0x04 U4 version (always 0x00 for v1)
|
||||
* +0x08 U4 flags (bits 0..2 = type, bit 3 = has_CLUT)
|
||||
* +0x0C ... CLUT section (if flags & 0x8)
|
||||
* +0x00 U4 clut_section_length
|
||||
* +0x04 U2 clut_org_x
|
||||
* +0x06 U2 clut_org_y
|
||||
* +0x08 U2 num_colors
|
||||
* +0x0A U2 depth_bpp
|
||||
* +0x0C ... palette data
|
||||
* ... ... Pixel section
|
||||
* +0x00 U4 px_section_length
|
||||
* +0x04 U2 px_width
|
||||
* +0x06 U2 px_height
|
||||
* +0x08 ... pixel data
|
||||
*
|
||||
* Future?: add `tim_load_to_vram(tim_ptr, vram_addr)` that emits the necessary GP0 commands.
|
||||
* Stoppped for now at the struct + enum level.
|
||||
* ============================================================================ */
|
||||
enum {
|
||||
tim_file_id_magic = 0x10,
|
||||
tim_type_4bpp = 0x00,
|
||||
tim_type_8bpp = 0x01,
|
||||
tim_type_16bpp = 0x02,
|
||||
tim_type_32bpp = 0x03,
|
||||
tim_type_mixed = 0x04,
|
||||
tim_flag_has_clut = 0x08,
|
||||
};
|
||||
|
||||
typedef Struct_(TIM_Header) {
|
||||
U4 file_id; /* always 0x10 = "TIM" magic */
|
||||
U4 version; /* ignored; always 0 */
|
||||
U4 flags; /* bits 0..2 = type, bit 3 = has_clut */
|
||||
};
|
||||
typedef Struct_(TIM_SectionHeader) {
|
||||
U4 section_length; /* bytes in this section including this header */
|
||||
U2 org_x; /* origin in VRAM */
|
||||
U2 org_y;
|
||||
U2 width; /* width in pixels */
|
||||
U2 height; /* height in pixels */
|
||||
};
|
||||
#pragma endregion TIM File Format
|
||||
|
||||
#pragma region Tape-Side Macros
|
||||
/* ============================================================================
|
||||
* Tape-side GPU operations (NOT in this header)
|
||||
* ============================================================================
|
||||
*
|
||||
* No `mac_gp0_send` or related macros live in gp.h.
|
||||
* Rationale: the Lottes tape model uses OT-DMA for primitive submission, so atom bodies write to main RAM (the OT/primitive buffer)
|
||||
* and to GTE state — never directly to the GPU ports at 0x1F801810 / 0x1F801814.
|
||||
* See `mac_format_f3_color`, `mac_insert_ot_tag`, `mac_gte_store_f3` in lottes_tape.h for the patterns atom bodies actually use.
|
||||
*
|
||||
* If a feature need arises requires tape-side GPU port writes
|
||||
* (e.g. DMA-kick to start GPU consumption of the OT, VBlank sync via GP1 status poll),
|
||||
* the right home is `lottes_tape.h` alongside the rest of the `mac_*` family:
|
||||
* 1. The caller pins a register to hold the IO base, e.g. register U4 r_io rgcc(R_T4) = IO_BASE_ADDR;
|
||||
* The compiler emits `lui R_T4, IO_BASE_ADDR_HI16` outside the atom body (in the C prologue before tape_run).
|
||||
* 2. The atom body uses `store_word(R_data, R_T4, GPIO_PORT0_OFFSET)` to write to GP0, and `store_word(R_data, R_T4, GPIO_PORT1_OFFSET)`
|
||||
* to write to GP1. Both are preprocessor-encodable because R_T4 is a fixed register and the GPIO_PORT*_OFFSET constants
|
||||
* fit in the `sw`'s 16-bit signed offset field. No placeholder-pun, no asm constraints, no hidden register choice.
|
||||
* Same pattern as the old graphics_hello/hello_gp_routines.s `reg_io_offset`/`gcmd_push` convention.
|
||||
*
|
||||
* This mirrors the existing tape-side wave-context discipline:
|
||||
* the caller binds the IO-base register via `rgcc()`, the macro assumes the binding is in effect,
|
||||
* and the encoding falls out at preprocessor time.
|
||||
* No additional GPU-domain macro layer required.
|
||||
* ============================================================================ */
|
||||
#pragma endregion Tape-Side Macros
|
||||
|
||||
@@ -0,0 +1,52 @@
|
||||
/* ============================================================================
|
||||
* duffle DSL — GPU Vendor Mnemonics (opt-in)
|
||||
* ============================================================================
|
||||
*
|
||||
* Provides the PSYQ-style CamelCase aliases for the duffle GPU primitive setters and OT operations.
|
||||
* The duffle snake_case names are primary; this header is for users who prefer the PSYQ SDK function names from the legacy C API.
|
||||
*
|
||||
* USAGE: #include "duffle/gp_vendor_sym.h" // after gp.h
|
||||
*
|
||||
* Mapping (vendor -> duffle):
|
||||
* Primitive setters (PSYQ SDK-style):
|
||||
* setPolyF3 -> set_poly_f3
|
||||
* setPolyF4 -> set_poly_f4
|
||||
* setPolyG3 -> set_poly_g3
|
||||
* setPolyG4 -> set_poly_g4
|
||||
* setPolyFT3 -> set_poly_ft3
|
||||
* setPolyFT4 -> set_poly_ft4
|
||||
* setPolyGT3 -> set_poly_gt3
|
||||
* setPolyGT4 -> set_poly_gt4
|
||||
*
|
||||
* OT operations:
|
||||
* AddPrim(ot, p) -> orderingtbl_add_primitive(ot, p)
|
||||
*
|
||||
* The gp0_cmd_* / gp1_cmd_* byte constants are already short and descriptive; no vendor alias is provided for them.
|
||||
*
|
||||
* The vendor mnemonics are NOT registered with the duffle word-count metadata (word_counts.metadata.h).
|
||||
* They expand to the duffle macros which DO have word-count entries
|
||||
* (the ones emitted by mac_format_f3_color / mac_gte_store_f3 / etc.). Verification: V13 (objdump byte-identical) holds.
|
||||
* ============================================================================ */
|
||||
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "gp.h"
|
||||
#endif
|
||||
|
||||
#ifndef DUFFLE_GP_VENDOR_SYM_H
|
||||
#define DUFFLE_GP_VENDOR_SYM_H
|
||||
|
||||
/* Primitive setters (PSYQ SDK-style) */
|
||||
#define setPolyF3(p) set_poly_f3(p)
|
||||
#define setPolyF4(p) set_poly_f4(p)
|
||||
#define setPolyG3(p) set_poly_g3(p)
|
||||
#define setPolyG4(p) set_poly_g4(p)
|
||||
#define setPolyFT3(p) set_poly_ft3(p)
|
||||
#define setPolyFT4(p) set_poly_ft4(p)
|
||||
#define setPolyGT3(p) set_poly_gt3(p)
|
||||
#define setPolyGT4(p) set_poly_gt4(p)
|
||||
|
||||
/* OT operations */
|
||||
#define AddPrim(ot, p) orderingtbl_add_primitive((ot), (p))
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,680 @@
|
||||
/* ============================================================================
|
||||
* duffle DSL Suffix Conventions
|
||||
* ============================================================================
|
||||
*
|
||||
* Every mnemonic in this header follows the same suffix grammar:
|
||||
*
|
||||
* Primitive commands: gp0_cmd_poly_f3 = 0x20 (byte opcode)
|
||||
* Packed 32-bit cmd: gp0_word_poly_f3(r, g, b) (32-bit, shifted)
|
||||
*
|
||||
* Type ordering: domain?_(direction)?_action_target_modifier_type?
|
||||
* Examples: add_ui (add + unsigned + immediate)
|
||||
* add_s (add + signed, R-type implicit)
|
||||
* shift_lleft (shift + logical + left)
|
||||
* shift_aright (shift + arithmetic + right)
|
||||
* call_reg(rs) (call + register, $ra implicit)
|
||||
* gte_mv_to_data_r (gte + mv + to + data + register)
|
||||
* gte_lw_v0_xy(base) (gte + lw + v0 + xy)
|
||||
* load_upper_i (load-upper + immediate, unique verb)
|
||||
*
|
||||
* Vendor mnemonics (gte_mtc2, gte_mfc2, gte_lwc2, gte_swc2, etc.) are
|
||||
* NOT in this header. They live in the opt-in `gte_vendor_sym.h` for
|
||||
* users who prefer the textbook MIPS assembly mnemonics.
|
||||
* ============================================================================ */
|
||||
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "dsl.h"
|
||||
# include "math.h"
|
||||
# include "mips.h"
|
||||
#endif
|
||||
|
||||
#pragma region ASM DSL
|
||||
/* ============================================================================
|
||||
* gte.h — Geometry Transformation Engine (COP2) for the PS1
|
||||
* ============================================================================
|
||||
*
|
||||
* Hand-rolled DSL for emitting GTE/MIPS instruction words as raw `.word`
|
||||
* constants from C. No GCC inline-assembly string syntax in the code body.
|
||||
*
|
||||
* STYLE NOTES
|
||||
* -----------
|
||||
* - Per-field encoders are named `enc_gte_<field>(value)` and each one
|
||||
* self-masks its argument before shifting. Mirrors the `enc_op / enc_rs
|
||||
* / enc_rt / ...` family in mips.h.
|
||||
* - The composite `enc_gte_cmdw(sf, mx, v, cv, lm, cmd)` is a flat OR of
|
||||
* the per-field encoders, plus the COP2/CO base.
|
||||
* - Pre-baked shortcuts (`gte_cmd_rtpt`, `gte_cmd_rtps`, …) are defined
|
||||
* for the common cases so call sites read like assembly source.
|
||||
* - All register/field values are enums (not `#define`s) so they show up
|
||||
* in debugger symbol tables and IDE autocomplete.
|
||||
*
|
||||
* SEE ALSO
|
||||
* --------
|
||||
* - mips.h: The MIPS encoder layer this builds on.
|
||||
*/
|
||||
|
||||
/* C2 data registers */
|
||||
|
||||
/* --- GTE Data Registers (Coprocessor 2) ---
|
||||
* Preprocessor-visible integer ids for the COP2 data register file.
|
||||
* Each enum value is bound to a parallel `_Code` `#define` so the
|
||||
* preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
||||
* Same pattern as the GPR `_Code` set in mips.h. */
|
||||
#define C2_VXY0_Code 0
|
||||
#define C2_VZ0_Code 1
|
||||
#define C2_VXY1_Code 2
|
||||
#define C2_VZ1_Code 3
|
||||
#define C2_VXY2_Code 4
|
||||
#define C2_VZ2_Code 5
|
||||
#define C2_RGB_Code 6
|
||||
#define C2_OTZ_Code 7
|
||||
#define C2_IR0_Code 8
|
||||
#define C2_IR1_Code 9
|
||||
#define C2_IR2_Code 10
|
||||
#define C2_IR3_Code 11
|
||||
#define C2_SXY0_Code 12
|
||||
#define C2_SXY1_Code 13
|
||||
#define C2_SXY2_Code 14
|
||||
#define C2_SXYP_Code 15
|
||||
#define C2_SZ0_Code 16
|
||||
#define C2_SZ1_Code 17
|
||||
#define C2_SZ2_Code 18
|
||||
#define C2_SZ3_Code 19
|
||||
#define C2_RGB0_Code 20
|
||||
#define C2_RGB1_Code 21
|
||||
#define C2_RGB2_Code 22
|
||||
#define C2_RES1_Code 23
|
||||
#define C2_MAC0_Code 24
|
||||
#define C2_MAC1_Code 25
|
||||
#define C2_MAC2_Code 26
|
||||
#define C2_MAC3_Code 27
|
||||
#define C2_IRGB_Code 28
|
||||
#define C2_ORGB_Code 29
|
||||
#define C2_LZCS_Code 30
|
||||
#define C2_LZCR_Code 31
|
||||
|
||||
enum {
|
||||
C2_VXY0 = C2_VXY0_Code, C2_VZ0 = C2_VZ0_Code, C2_VXY1 = C2_VXY1_Code, C2_VZ1 = C2_VZ1_Code,
|
||||
C2_VXY2 = C2_VXY2_Code, C2_VZ2 = C2_VZ2_Code, C2_RGB = C2_RGB_Code, C2_OTZ = C2_OTZ_Code,
|
||||
C2_IR0 = C2_IR0_Code, C2_IR1 = C2_IR1_Code, C2_IR2 = C2_IR2_Code, C2_IR3 = C2_IR3_Code,
|
||||
C2_SXY0 = C2_SXY0_Code, C2_SXY1 = C2_SXY1_Code, C2_SXY2 = C2_SXY2_Code, C2_SXYP = C2_SXYP_Code,
|
||||
C2_SZ0 = C2_SZ0_Code, C2_SZ1 = C2_SZ1_Code, C2_SZ2 = C2_SZ2_Code, C2_SZ3 = C2_SZ3_Code,
|
||||
C2_RGB0 = C2_RGB0_Code, C2_RGB1 = C2_RGB1_Code, C2_RGB2 = C2_RGB2_Code, C2_RES1 = C2_RES1_Code,
|
||||
C2_MAC0 = C2_MAC0_Code, C2_MAC1 = C2_MAC1_Code, C2_MAC2 = C2_MAC2_Code, C2_MAC3 = C2_MAC3_Code,
|
||||
C2_IRGB = C2_IRGB_Code, C2_ORGB = C2_ORGB_Code, C2_LZCS = C2_LZCS_Code, C2_LZCR = C2_LZCR_Code
|
||||
};
|
||||
|
||||
/* Semantic Aliases for GTE Data Registers */
|
||||
enum {
|
||||
gte_in_v0_xy = C2_VXY0, /* Input Vector 0 (X, Y) */
|
||||
gte_in_v0_z = C2_VZ0, /* Input Vector 0 (Z) */
|
||||
gte_in_v1_xy = C2_VXY1, /* Input Vector 1 (X, Y) */
|
||||
gte_in_v1_z = C2_VZ1, /* Input Vector 1 (Z) */
|
||||
gte_in_v2_xy = C2_VXY2, /* Input Vector 2 (X, Y) */
|
||||
gte_in_v2_z = C2_VZ2, /* Input Vector 2 (Z) */
|
||||
gte_in_rgb = C2_RGB, /* Input Color (R, G, B, MipsCode) */
|
||||
gte_out_scr_xy0 = C2_SXY0, /* Output Screen Coord 0 (X, Y) */
|
||||
gte_out_scr_xy1 = C2_SXY1, /* Output Screen Coord 1 (X, Y) */
|
||||
gte_out_scr_xy2 = C2_SXY2, /* Output Screen Coord 2 (X, Y) */
|
||||
gte_out_depth = C2_OTZ, /* Output Ordering Table Z (Depth) */
|
||||
gte_math_accum0 = C2_MAC0, /* Math Accumulator 0 */
|
||||
gte_math_accum1 = C2_MAC1, /* Math Accumulator 1 */
|
||||
gte_math_accum2 = C2_MAC2, /* Math Accumulator 2 */
|
||||
};
|
||||
|
||||
/* --- GTE Command Semantics (The Bitfield Meanings) ---
|
||||
* A GTE command is a single 32-bit word sent to COP2.
|
||||
* It is highly configurable via bitfields.
|
||||
*/
|
||||
|
||||
enum {
|
||||
/* Shift Fraction (Bit 19) - Determines fixed-point division */
|
||||
|
||||
gte_sf_fractional = 0, /* Divide result by 4096 (Standard 4.12 fixed point) */
|
||||
gte_sf_integer = 1, /* No division (Raw integer math) */
|
||||
|
||||
/* Matrix Select (Bits 18-17) - Which 3x3 matrix to multiply by */
|
||||
|
||||
gte_mx_rotation = 0, /* Rotation Matrix (RT) */
|
||||
gte_mx_light = 1, /* Light Matrix (LL) */
|
||||
gte_mx_color = 2, /* Color Matrix (LC) */
|
||||
gte_mx_none = 3, /* Reserved / Do not multiply */
|
||||
|
||||
/* Vector select (Bits 16-15) - Which input vector to use */
|
||||
|
||||
gte_v_v0 = 0, /* Use Vector 0 (VXY0, VZ0) */
|
||||
gte_v_v1 = 1, /* Use Vector 1 (VXY1, VZ1) */
|
||||
gte_v_v2 = 2, /* Use Vector 2 (VXY2, VZ2) */
|
||||
gte_v_ir_regs = 3, /* Use Intermediate Registers (IR1, IR2, IR3) */
|
||||
|
||||
/* Control Vector Select (Bits 14-13) - Which vector to ADD after multiplication */
|
||||
|
||||
gte_cv_translate = 0, /* Add Translation Vector (TRX, TRY, TRZ) */
|
||||
gte_cv_bg_color = 1, /* Add Background Color (RBK, GBK, BBK) */
|
||||
gte_cv_far_color = 2, /* Add Far Color (RFC, GFC, BFC) */
|
||||
gte_cv_none = 3, /* Add Zero (No addition) */
|
||||
|
||||
/* Limit/Clamp (Bit 10) - Prevents overflow artifacts */
|
||||
|
||||
gte_lm_normal = 0, /* Normal math (can overflow) */
|
||||
gte_lm_clamp = 1, /* Clamp results to valid hardware ranges (e.g., RGB 0-255) */
|
||||
|
||||
/* Core Command IDs (Bits 5-0) */
|
||||
|
||||
gte_cmd_rtps = 0x01, /* Rot/Trans Perspective Single (1 vertex) */
|
||||
gte_cmd_rtpt = 0x30, /* Rot/Trans Perspective Triple (3 vertices) */
|
||||
gte_cmd_nclip = 0x06, /* Normal Clipping (Backface culling) */
|
||||
gte_cmd_op = 0x0C, /* Outer Product */
|
||||
gte_cmd_mvmva = 0x12, /* Matrix Vector Multiply & Add (Custom math) */
|
||||
|
||||
/* --- GTE Command Bit-Field Layout ---
|
||||
* A GTE command word (sent to COP2 with RS=1) is laid out as:
|
||||
*
|
||||
* 31........25 24 23..19 18..17 16..15 14..13 12..11 10 9.......6 5.......0
|
||||
* +------------+--+-----+------+------+------+------+---+--------+----------+
|
||||
* | 0x3E (COP2)| 1| -- | sf | mx | v | cv | --| lm | -- | cmd |
|
||||
* +------------+--+-----+------+------+------+------+---+--------+----------+
|
||||
* \_____ GTE_PAYLOAD _____/ \__ GTE_CMD __/
|
||||
*
|
||||
* Shifts/masks below are the *bit positions* and *bit widths* of each
|
||||
* configurable field, used by the ENC_GTE_CMD encoder.
|
||||
* Mirrors the OPCODE_SHIFT / RS_SHIFT convention used in mips.h.
|
||||
*/
|
||||
|
||||
gte_shift_sf = 19, gte_width_sf = 1, gte_mask_sf = 0x1,
|
||||
gte_shift_mx = 17, gte_width_mx = 2, gte_mask_mx = 0x3,
|
||||
gte_shift_v = 15, gte_width_v = 2, gte_mask_v = 0x3,
|
||||
gte_shift_cv = 13, gte_width_cv = 2, gte_mask_cv = 0x3,
|
||||
gte_shift_lm = 10, gte_width_lm = 1, gte_mask_lm = 0x1,
|
||||
gte_shift_cmd = 0, gte_width_cmd = 6, gte_mask_cmd = 0x3F,
|
||||
};
|
||||
|
||||
/* --- GTE Control Register Indices (for ctc2/cfc2) ---
|
||||
* Preprocessor-visible integer ids for the COP2 control register file.
|
||||
* Each enum value is bound to a parallel `_Code` `#define` so the
|
||||
* preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
||||
* Same pattern as the GPR `_Code` set in mips.h. Note: indices 21-23
|
||||
* are reserved/unused on real hardware, so there's a gap. */
|
||||
#define gte_cr_RT11_Code 0
|
||||
#define gte_cr_RT12_Code 1 /* packed with RT13 in bits 16..31 */
|
||||
#define gte_cr_RT13_Code 2 /* packed with RT22 in bits 16..31 */
|
||||
#define gte_cr_RT21_Code 3 /* packed with RT31 in bits 16..31 */
|
||||
#define gte_cr_RT22_Code 4 /* RT33 alone (low 16 bits used) */
|
||||
// #define gte_cr_RT23_Code 5
|
||||
// #define gte_cr_RT31_Code 6
|
||||
// #define gte_cr_RT32_Code 7
|
||||
// #define gte_cr_RT33_Code 8
|
||||
#define gte_cr_TRX_Code 5 /* PSX SDK convention: C2 r5 = TRX (alone, 32-bit) */
|
||||
#define gte_cr_TRY_Code 6 /* PSX SDK convention: C2 r6 = TRY (alone, 32-bit) */
|
||||
#define gte_cr_TRZ_Code 7 /* PSX SDK convention: C2 r7 = TRZ (alone, 32-bit) */
|
||||
#define gte_cr_L11_Code 12
|
||||
#define gte_cr_L12_Code 13
|
||||
#define gte_cr_L13_Code 14
|
||||
#define gte_cr_L21_Code 15
|
||||
#define gte_cr_L22_Code 16
|
||||
#define gte_cr_L23_Code 17
|
||||
#define gte_cr_LR1_Code 18
|
||||
#define gte_cr_LR2_Code 19
|
||||
#define gte_cr_LR3_Code 20
|
||||
#define gte_cr_RBK_Code 24
|
||||
#define gte_cr_GBK_Code 25
|
||||
#define gte_cr_BBK_Code 26
|
||||
#define gte_cr_RFC_Code 27
|
||||
#define gte_cr_GFC_Code 28
|
||||
#define gte_cr_BFC_Code 29
|
||||
#define gte_cr_OFX_Code 30
|
||||
#define gte_cr_OFY_Code 31
|
||||
|
||||
enum {
|
||||
gte_cr_RT11 = gte_cr_RT11_Code, gte_cr_RT12 = gte_cr_RT12_Code, gte_cr_RT13 = gte_cr_RT13_Code,
|
||||
gte_cr_RT21 = gte_cr_RT21_Code, gte_cr_RT22 = gte_cr_RT22_Code, //gte_cr_RT23 = gte_cr_RT23_Code,
|
||||
// gte_cr_RT31 = gte_cr_RT31_Code, gte_cr_RT32 = gte_cr_RT32_Code, gte_cr_RT33 = gte_cr_RT33_Code,
|
||||
gte_cr_TRX = gte_cr_TRX_Code, gte_cr_TRY = gte_cr_TRY_Code, gte_cr_TRZ = gte_cr_TRZ_Code,
|
||||
gte_cr_L11 = gte_cr_L11_Code, gte_cr_L12 = gte_cr_L12_Code, gte_cr_L13 = gte_cr_L13_Code,
|
||||
gte_cr_L21 = gte_cr_L21_Code, gte_cr_L22 = gte_cr_L22_Code, gte_cr_L23 = gte_cr_L23_Code,
|
||||
gte_cr_LR1 = gte_cr_LR1_Code, gte_cr_LR2 = gte_cr_LR2_Code, gte_cr_LR3 = gte_cr_LR3_Code,
|
||||
gte_cr_RBK = gte_cr_RBK_Code, gte_cr_GBK = gte_cr_GBK_Code, gte_cr_BBK = gte_cr_BBK_Code,
|
||||
gte_cr_RFC = gte_cr_RFC_Code, gte_cr_GFC = gte_cr_GFC_Code, gte_cr_BFC = gte_cr_BFC_Code,
|
||||
gte_cr_OFX = gte_cr_OFX_Code, gte_cr_OFY = gte_cr_OFY_Code,
|
||||
};
|
||||
|
||||
enum { _C2_OPS_ = 0
|
||||
, op_lwc2 = 0x32 /* Load Word to Coprocessor 2 (GTE) */
|
||||
, op_swc2 = 0x3A /* Store Word from Coprocessor 2 (GTE) */
|
||||
};
|
||||
|
||||
/* COP2 transfer sub-opcodes (5-bit field in the `rs` slot of enc_gte_tx).
|
||||
*
|
||||
* Spans the 2x2 {From, To} × {Data, Control} register classes that the
|
||||
* GTE exposes:
|
||||
*
|
||||
* bit 1 (0x02): register class — 0 = data, 1 = control
|
||||
* bit 2 (0x04): direction — 0 = read, 1 = write
|
||||
*
|
||||
* The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit
|
||||
* numbers as the general MIPS `cop_mf` / `cop_mt` defined in mips.h
|
||||
* (which target the data register file on any coprocessor). They are
|
||||
* re-aliased here so the four-way table reads like the spec mnemonics
|
||||
* (MFC2 / CFC2 / MTC2 / CTC2) and so the encoding lives next to its
|
||||
* only consumer (this header).
|
||||
*
|
||||
* Vendor mnemonic aliases (gte_mfc2 / gte_mtc2 / gte_cfc2 / gte_ctc2)
|
||||
* live in gte_vendor_sym.h. */
|
||||
enum { _C2_TX_SUBS_ = 0
|
||||
, sub_mfc2 = 0x00 /* MFC2: Move From Coprocessor 2 data reg */
|
||||
, sub_cfc2 = 0x02 /* CFC2: Copy From Coprocessor 2 ctrl reg */
|
||||
, sub_mtc2 = 0x04 /* MTC2: Move To Coprocessor 2 data reg */
|
||||
, sub_ctc2 = 0x06 /* CTC2: Copy To Coprocessor 2 ctrl reg */
|
||||
};
|
||||
|
||||
/* COP2 (GTE) Transfer Format: mfc2 / cfc2 / mtc2 / ctc2 rt, rd
|
||||
* Layout: [op_cop2:6][sub:5][rt:5][rd:5][0:11]
|
||||
* - sub: one of sub_mfc2 / sub_cfc2 / sub_mtc2 / sub_ctc2
|
||||
* - rt: GPR source/dest
|
||||
* - rd: COP2 register index (0..31):
|
||||
* data class → C2_VXY0_Code..C2_LZCR_Code (gte_in_v0_xy..gte_math_accum2 aliases)
|
||||
* ctrl class → gte_cr_RT11_Code..gte_cr_OFY_Code */
|
||||
#define enc_gte_tx(sub, rt, rd) (enc_op(op_cop2) | enc_rs(sub) | enc_rt(rt) | enc_rd(rd))
|
||||
|
||||
|
||||
// #define gte_mv_to_data_r(rt, rd) enc_gte_tx(cop_mt, (rt), (rd)) /* Move GPR (rt) to GTE Control Register (rd) */
|
||||
// #define gte_mv_from_data_r(rt, rd) enc_gte_tx(cop_mf, (rt), (rd)) /* Move GTE Control Register (rd) to GPR (rt) */
|
||||
|
||||
/* GTE Data vs Control Register Transfers
|
||||
*
|
||||
* Each macro emits a single .word constant for one of MFC2/CFC2/MTC2/CTC2.
|
||||
*
|
||||
* `rd` is the C2 register index in the file the sub-opcode names:
|
||||
* gte_mv_from_data_r / gte_mv_to_data_r → C2 data register file
|
||||
* gte_mv_from_ctrl_r / gte_mv_to_ctrl_r → C2 ctrl register file
|
||||
*
|
||||
* Common pairs:
|
||||
* gte_mv_from_data_r(R_T0, C2_MAC0) — read MAC0 into a GPR
|
||||
* gte_mv_to_data_r (R_V0, C2_VXY0) — write GPR into VXY0
|
||||
* gte_mv_to_ctrl_r (R_T0, gte_cr_RT11) — write GPR into rotation matrix
|
||||
* gte_mv_from_ctrl_r(R_T0, gte_cr_OFX) — read screen-X offset */
|
||||
#define gte_mv_from_data_r(rt, rd) enc_gte_tx(sub_mfc2, (rt), (rd)) /* Move From data reg */
|
||||
#define gte_mv_from_ctrl_r(rt, rd) enc_gte_tx(sub_cfc2, (rt), (rd)) /* Copy From ctrl reg */
|
||||
#define gte_mv_to_data_r(rt, rd) enc_gte_tx(sub_mtc2, (rt), (rd)) /* Move To data reg */
|
||||
#define gte_mv_to_ctrl_r(rt, rd) enc_gte_tx(sub_ctc2, (rt), (rd)) /* Copy To ctrl reg */
|
||||
|
||||
/* COP2 Data Load (lwc2): `lwc2 rt, off(rs)`
|
||||
* Layout: [op_lwc2:6][rs:5][rt:5][imm:16]
|
||||
* - rs: GPR base address
|
||||
* - rt: COP2 data register index (0..31)
|
||||
* - imm: signed 16-bit offset
|
||||
* NOTE: When `rs` is a runtime register, the encoding cannot be pre-baked
|
||||
* into a .word — use the string-style `gte_load_v0` macro below instead. */
|
||||
#define enc_gte_lw(rt, base, off) enc_i(op_lwc2, (base), (rt), (off))
|
||||
/* Store Word */
|
||||
#define enc_gte_sw(rt, base, off) enc_i(op_swc2, (base), (rt), (off))
|
||||
|
||||
/* Semantic aliases for the COP2 data load/store. The `c2` in `lwc2`/
|
||||
* `swc2` is redundant when we're already inside the `gte_` namespace.
|
||||
* gte_lw rt, base, off → lwc2 rt, off(base)
|
||||
* gte_sw rt, base, off → swc2 rt, off(base)
|
||||
* For the typical user-facing vector-level load (xy + z as two
|
||||
* instructions), use the higher-level `gte_load_vN` macros below. */
|
||||
#define gte_lw(rt, base, off) enc_gte_lw(rt, base, off)
|
||||
#define gte_sw(rt, base, off) enc_gte_sw(rt, base, off)
|
||||
|
||||
/* GTE Command Format
|
||||
* Opcode is always MIPS_OP_COP2, RS is always 1 (CO).
|
||||
* The lower 25 bits are the GTE-specific command payload.
|
||||
*
|
||||
* The granular `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs`
|
||||
* pattern in mips.h: each one self-masks and shifts its own field, so a
|
||||
* caller can build up a GTE command piece by piece (handy for state-driven
|
||||
* MVMVA emitters that vary one field at a time).
|
||||
*
|
||||
* `ENC_GTE_CMD` is the all-in-one convenience for emitting a full command
|
||||
* word in one go. It just ORs the per-field encoders together. */
|
||||
#define gte_cmd_base (enc_op(op_cop2) | (1 << 25))
|
||||
|
||||
/* Per-field encoders. Each one does (value & mask) << shift on its own. */
|
||||
#define enc_gte_sf(sf) (((sf) & gte_mask_sf ) << gte_shift_sf )
|
||||
#define enc_gte_mx(mx) (((mx) & gte_mask_mx ) << gte_shift_mx )
|
||||
#define enc_gte_v(v) (((v) & gte_mask_v ) << gte_shift_v )
|
||||
#define enc_gte_cv(cv) (((cv) & gte_mask_cv ) << gte_shift_cv )
|
||||
#define enc_gte_lm(lm) (((lm) & gte_mask_lm ) << gte_shift_lm )
|
||||
#define enc_gte_cmd(cmd) (((cmd) & gte_mask_cmd) << gte_shift_cmd)
|
||||
|
||||
/* Composite: all six GTE fields + the COP2/CO base. */
|
||||
#define enc_gte_cmdw(sf, mx, v, cv, lm, cmd) ( \
|
||||
gte_cmd_base \
|
||||
| enc_gte_sf(sf) \
|
||||
| enc_gte_mx(mx) \
|
||||
| enc_gte_v(v) \
|
||||
| enc_gte_cv(cv) \
|
||||
| enc_gte_lm(lm) \
|
||||
| enc_gte_cmd(cmd) \
|
||||
)
|
||||
|
||||
/* GTE command words for the common cases.
|
||||
*
|
||||
* These are pure compile-time integer constants — the C compiler constant-folds them into `.word` directives in .rodata.
|
||||
* Use them inside `asm_inline(...)` blocks (see `gte_rtpt` below for the idiom).
|
||||
*
|
||||
* Decomposition (per the `enc_gte_<field>` definitions above):
|
||||
* gte_cmdw_<name> = gte_cmd_base | enc_gte_cmd(<cmd>)
|
||||
|
||||
* The SF/MX/V/CV/LM fields are all zero in the common cases
|
||||
* (standard rotation-matrix, no scaling factor, V0 vector, translation vector, no clamp),
|
||||
* so the only varying bits are the `cmd` field.
|
||||
*
|
||||
* Naming follows the file's convention: `gte_cmd_*` is the raw 6-bit `cmd` field id, `gte_cmdw_*`
|
||||
* is the fully-encoded 32-bit instruction word ready to drop into a `.word` directive.
|
||||
*
|
||||
* --------------------------------------------------------------------------
|
||||
* PsyQ-compatibility note (RTPS/RTPT):
|
||||
* The original Sony PsyQ `inline_n.h` ships RTPT as `cop2 0x0280030` and RTPS as `cop2 0x0180001`.
|
||||
* Both have `0x20` set in the upper-reserved region (bit 21) AND `sf=1` (bit 19) — i.e. the "no division" flag.
|
||||
* Per psx-spec these bits are reserved/must-be-zero,
|
||||
* but the real GTE hardware and PCSX-Redux's GTE model both IGNORE them on these two commands
|
||||
* (the perspective divide happens regardless of `sf`).
|
||||
*
|
||||
* If we emit a strictly-spec-compliant word (`sf=0`, reserved bits clear),
|
||||
* PCSX-Redux's GTE checks those bits more strictly than the silicon does and RTPT silently no-ops —
|
||||
* the floor's screen coordinates come out as raw projection-of-rotation (Z never divided),
|
||||
* `nclip` ends up wrong, and the triangle is culled.
|
||||
*
|
||||
* So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern everyone has shipped for 25 years.
|
||||
* NCLIP/OP/MVMVA stay spec-clean — their reserved bits really are zero in the original PsyQ source.
|
||||
* --------------------------------------------------------------------------
|
||||
*/
|
||||
#define gte_cmdw_psyq_compat (1u << 21 | enc_gte_sf(gte_sf_integer))
|
||||
|
||||
#define gte_cmdw_rtps (gte_cmd_base | enc_gte_cmd(gte_cmd_rtps ) | gte_cmdw_psyq_compat)
|
||||
#define gte_cmdw_rtpt (gte_cmd_base | enc_gte_cmd(gte_cmd_rtpt ) | gte_cmdw_psyq_compat)
|
||||
#define gte_cmdw_nclip (gte_cmd_base | enc_gte_cmd(gte_cmd_nclip))
|
||||
#define gte_cmdw_op (gte_cmd_base | enc_gte_cmd(gte_cmd_op ))
|
||||
#define gte_cmdw_outer_product gte_cmdw_op /* "outer product" -- NOCASH/Sdk terminology */
|
||||
#define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology */
|
||||
#define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva))
|
||||
|
||||
#define gte_cmdw_rotate_translate_perspective_single gte_cmdw_rtps
|
||||
#define gte_cmdw_rotate_translate_perspective_triple gte_cmdw_rtpt
|
||||
|
||||
/* PsyQ compatibility bits for AVSZ3 (Bits 20, 22, 24 must be set) */
|
||||
#define gte_cmdw_psyq_avsz3_compat (0x15 << 20)
|
||||
#define gte_cmd_avsz3 0x2D
|
||||
|
||||
#define gte_cmdw_avsz3 (gte_cmd_base | enc_gte_cmd(gte_cmd_avsz3) | gte_cmdw_psyq_avsz3_compat)
|
||||
|
||||
// Takes the three screen-space Z values of the triangle → averages them → writes the result into the OTZ register.
|
||||
#define gte_avg_sort_z3 gte_cmdw_avsz3
|
||||
|
||||
/* AVSZ4 — average Z of 4 vertices (for quads) */
|
||||
#define gte_cmd_avsz4 0x2E
|
||||
#define gte_cmdw_avsz4 (gte_cmd_base | enc_gte_cmd(gte_cmd_avsz4) | gte_cmdw_psyq_avsz3_compat)
|
||||
|
||||
#define gte_cmdw_avg_sort_z4 gte_cmdw_avsz4
|
||||
|
||||
/**
|
||||
* @brief Loads a single SVECTOR to GTE vector register V0
|
||||
*
|
||||
* @details Loads values from an SVECTOR struct to GTE data registers C2_VXY0
|
||||
* (XY at offset 0) and C2_VZ0 (Z at offset 4) using `lwc2`.
|
||||
*
|
||||
* Uses string-style GCC inline asm with `%0` substitution because the
|
||||
* base register `r0` is a runtime GPR chosen by the compiler.
|
||||
* It cannot be encoded into a static `.word` constant.
|
||||
*
|
||||
* Usage:
|
||||
* asm_gte_load_v0(svector_ptr);
|
||||
*/
|
||||
|
||||
/* lwc2 encoding helpers parameterized on the base GPR.
|
||||
*
|
||||
* gte_lw_v0_xy(base) → lwc2 $0, 0(base) ; C2_VXY0
|
||||
* gte_lw_v0_z(base) → lwc2 $1, 4(base) ; C2_VZ0
|
||||
* gte_lw_v1_xy(base) → lwc2 $2, 0(base) ; C2_VXY1
|
||||
* gte_lw_v1_z(base) → lwc2 $3, 4(base) ; C2_VZ1
|
||||
* gte_lw_v2_xy(base) → lwc2 $4, 0(base) ; C2_VXY2
|
||||
* gte_lw_v2_z(base) → lwc2 $5, 4(base) ; C2_VZ2
|
||||
*
|
||||
* `base` is the GPR number to bake into the .word constant's `rs` field.
|
||||
* These are pure compile-time integers; the C compiler constant-folds
|
||||
* them into .word directives. */
|
||||
|
||||
enum {
|
||||
GTE_Z_Offset = 4
|
||||
};
|
||||
|
||||
#define gte_lw_v0_xy(base) enc_gte_lw(gte_in_v0_xy, (base), 0)
|
||||
#define gte_lw_v0_z(base) enc_gte_lw(gte_in_v0_z, (base), GTE_Z_Offset)
|
||||
#define gte_lw_v1_xy(base) enc_gte_lw(gte_in_v1_xy, (base), 0)
|
||||
#define gte_lw_v1_z(base) enc_gte_lw(gte_in_v1_z, (base), GTE_Z_Offset)
|
||||
#define gte_lw_v2_xy(base) enc_gte_lw(gte_in_v2_xy, (base), 0)
|
||||
#define gte_lw_v2_z(base) enc_gte_lw(gte_in_v2_z, (base), GTE_Z_Offset)
|
||||
|
||||
/* gte_load_vN(r_ptr, base) — placeholder-punned lwc2 loaders
|
||||
*
|
||||
* Emits `.word` constants encoding `lwc2 $N, off(<base>)` for the chosen GTE vector register, where `<base>` is the GPR number you pass in
|
||||
* (typically one of R_T4..R_T9 for the standard "3-pointer" pattern).
|
||||
*
|
||||
* The caller MUST bind `r_ptr` to that same GPR via a register variable:
|
||||
* register V3_S2* p_in_12 __asm__("$12") = my_ptr;
|
||||
* gte_load_v0(p_in_12, R_T4); // R_T4 = 12, base is $12
|
||||
*
|
||||
* Then `"r"(r_ptr)` inside the asm binds to $12 (the only register `p_in_12` can live in),
|
||||
* which is exactly the register the .word constants expect. A `"$12"` clobber would conflict with the register-variable binding
|
||||
* ("asm specifier for variable conflicts with asm clobber list"), so we omit it.
|
||||
* The other ABI-clobbers ($2/$8/$9/$31) stay because the GTE instructions don't touch caller-saved GPRs but the kernel does treat them as volatile.
|
||||
*
|
||||
* WHICH REGISTER TO PICK
|
||||
* ----------------------
|
||||
* Any caller-saved GPR is safe. Recommended default for an RTPT-style 3-pointer pipeline:
|
||||
* gte_load_v0(p0, R_T4); // $12
|
||||
* gte_load_v1(p1, R_T5); // $13
|
||||
* gte_load_v2(p2, R_T6); // $14
|
||||
* Avoid $0 (zero), $1 (at), $26/$27 (k0/k1), $28-$31 (gp/sp/fp/ra).
|
||||
*
|
||||
* Shape of the generated `asm volatile (...)`:
|
||||
* code section : ".word %0, %1" (from asm_inline)
|
||||
* outputs section : (empty, the 2nd colon)
|
||||
* inputs section : "i"(w0), "i"(w1), "r"(r_ptr) — r_ptr bound to <base>
|
||||
* clobbers section : "$2", "$8", ..., "memory" (from asm_clobber)
|
||||
* 3 colons total, GCC-legal. No string-syntax mnemonics in the .word body.
|
||||
*
|
||||
* The `asm_clobber(...)` helper from gcc_asm.h prepends the colon that starts the clobbers section. */
|
||||
#define gte_load_v0(r_ptr, base) asm volatile( \
|
||||
asm_words( gte_lw_v0_xy(base), gte_lw_v0_z(base) ) \
|
||||
asm_rpins, r_use(r_ptr) \
|
||||
asm_clobber: rlit(R_V0), rlit(R_T0), rlit(R_T1), rlit(R_RA), clb_mem_drain \
|
||||
)
|
||||
|
||||
#define gte_load_v1(r_ptr, base) asm volatile( \
|
||||
asm_words( gte_lw_v1_xy(base), gte_lw_v1_z(base) ) \
|
||||
asm_rpins, r_use(r_ptr) \
|
||||
asm_clobber: rlit(R_V0), rlit(R_T0), rlit(R_T1), rlit(R_RA), clb_mem_drain \
|
||||
)
|
||||
|
||||
#define gte_load_v2(r_ptr, base) asm volatile( \
|
||||
asm_words( gte_lw_v2_xy(base), gte_lw_v2_z(base) ) \
|
||||
asm_rpins, r_use(r_ptr) \
|
||||
asm_clobber: rlit(R_V0), rlit(R_T0), rlit(R_T1), rlit(R_RA), clb_mem_drain \
|
||||
)
|
||||
|
||||
/* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — prelude to gte_cmd_rtpt.
|
||||
*
|
||||
* Loads all three GTE input vectors (6 words) from three separate pointers,
|
||||
* one per GTE vector register, each loaded from its own base GPR.
|
||||
* Caller must bind each `pN` to `bN` via a register variable.
|
||||
*
|
||||
* register V3_S2* p0 rgcc(R_T4) = verts[0].ptr; // → __asm__("$12")
|
||||
* register V3_S2* p1 rgcc(R_T5) = verts[1].ptr; // → __asm__("$13")
|
||||
* register V3_S2* p2 rgcc(R_T6) = verts[2].ptr; // → __asm__("$14")
|
||||
* gte_load_v0v1v2(p0, p1, p2, R_T4, R_T5, R_T6);
|
||||
* gte_rtpt();
|
||||
*/
|
||||
#define gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) asm volatile( \
|
||||
asm_words( \
|
||||
gte_lw_v0_xy(b0), gte_lw_v0_z(b0), \
|
||||
gte_lw_v1_xy(b1), gte_lw_v1_z(b1), \
|
||||
gte_lw_v2_xy(b2), gte_lw_v2_z(b2) ) \
|
||||
asm_rpins \
|
||||
, r_use(p0), r_use(p1), r_use(p2) \
|
||||
asm_clobber: rlit(R_V0), rlit(R_T0), rlit(R_T1), rlit(R_RA), clb_mem_drain \
|
||||
)
|
||||
|
||||
/**
|
||||
* @brief Rotate, Translate and Perspective Triple (23 cycles)
|
||||
*
|
||||
* @details Performs rotation, translation and perspective calculation of three
|
||||
* vertices at once. The equation performed is the same as gte_rtps() only
|
||||
* repeated three times for each vertex. The result of the first vertex is
|
||||
* stored in GTE data register C2_SXY0, the second vector in C2_SXY1 then
|
||||
* C2_SXY2.
|
||||
*
|
||||
* Encoder-style emission (no inline-asm strings in the code body):
|
||||
* 1. Two `nop` words fill the COP2 pipeline latency — the GTE
|
||||
* takes ~8 cycles per perspective divide, and the nops let any
|
||||
* preceding lwc2/swc2 retire before RTPT starts reading its
|
||||
* inputs from V0/V1/V2.
|
||||
* 2. The RTPT command word itself is `gte_cmdw_rtpt` (see the
|
||||
* pre-baked encoders above) — `0x0280030` decoded as
|
||||
* `op_cop2` | CO(1) | cmd=RTPT, with all SF/MX/V/CV/LM fields
|
||||
* zero (standard rotation, no scaling, V0 vector, translation
|
||||
* vector, no clamp).
|
||||
*
|
||||
* Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel
|
||||
* ABI) plus the standard "memory" barrier. Does not clobber any COP2
|
||||
* data/control register — those have to be saved by the caller if
|
||||
* they need to survive across the call (RTPT writes SXY0..2, SZ0..3,
|
||||
* OTZ, MAC0..3, IR0..3, etc.).
|
||||
*/
|
||||
#define gte_rtpt() \
|
||||
asm volatile( \
|
||||
asm_words( nop, nop, gte_cmdw_rtpt ) \
|
||||
asm_clobber: clbr_volatile_gprs \
|
||||
)
|
||||
|
||||
#define gte_rtpt_asm_str() \
|
||||
__asm__ volatile( \
|
||||
"nop;" \
|
||||
"nop;" \
|
||||
"cop2 0x0280030;")
|
||||
|
||||
/**
|
||||
* @brief Normal clipping (8 cycles)
|
||||
*
|
||||
* @details Computes the sign of three screen coordinates (C2_SXY0-2) used for
|
||||
* backface culling. If the value of C2_MAC0 is negative, the coordinates are
|
||||
* inverted and thus the triangle is back facing.
|
||||
*
|
||||
* The following equation is performed when executing this GTE command:
|
||||
*
|
||||
* MAC0 = SX0*SY1 + SX1*SY2 + SX2*SY0 - SX0*SY2 - SX1*SY0 - SX2*SY1
|
||||
*
|
||||
* Encoder-style emission (no inline-asm strings in the code body):
|
||||
* 1. Two `nop` words fill the COP2 pipeline latency - the GTE
|
||||
* pipeline takes a few cycles per op, and the nops let any
|
||||
* preceding lwc2/swc2/RTPT retire before NCLIP starts reading
|
||||
* its inputs from SXY0/SXY1/SXY2.
|
||||
* 2. The NCLIP command word itself is `gte_cmdw_nclip` (see the
|
||||
* pre-baked encoders above) - `0x01400006` decoded as
|
||||
* `op_cop2` | CO(1) | cmd=NCLIP, with all SF/MX/V/CV/LM fields
|
||||
* zero. NCLIP is spec-clean in the original PsyQ source
|
||||
* (unlike RTPS/RTPT which carry the `gte_cmdw_psyq_compat`
|
||||
* quirk), so `gte_cmdw_nclip` does NOT OR in any reserved bits.
|
||||
*
|
||||
* Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel
|
||||
* ABI) plus the standard "memory" barrier. Does not clobber any COP2
|
||||
* data/control register - those have to be saved by the caller if
|
||||
* they need to survive across the call (NCLIP writes MAC0 only; it
|
||||
* is purely a sign-of-double-product computation on SXY0..2).
|
||||
*/
|
||||
#define gte_nclip() \
|
||||
asm volatile( \
|
||||
asm_words( nop, nop, gte_cmdw_nclip ) \
|
||||
asm_clobber: clbr_volatile_gprs \
|
||||
)
|
||||
|
||||
#define gte_stotz(r0) __asm__ volatile("swc2 $7, 0( %0 )" : : "r"(r0) : "memory")
|
||||
|
||||
#define gte_stsxy3(r0, r1, r2) \
|
||||
__asm__ volatile( \
|
||||
"swc2 $12, 0( %0 );" \
|
||||
"swc2 $13, 0( %1 );" \
|
||||
"swc2 $14, 0( %2 )" \
|
||||
: \
|
||||
: "r"(r0), "r"(r1), "r"(r2) \
|
||||
: "memory")
|
||||
|
||||
#define gte_avsz3() \
|
||||
__asm__ volatile( \
|
||||
"nop;" \
|
||||
"nop;" \
|
||||
"cop2 0x0158002D;")
|
||||
|
||||
/* asm_gte_matrix_set_rotation(r0)
|
||||
*
|
||||
* Loads the 3x3 rotation matrix at `r0` into the GTE's rotation-matrix
|
||||
* control registers (RT11..RT22, indices 0..4) via ctc2.
|
||||
*
|
||||
* Memory layout at r0: five contiguous 32-bit words (offsets 0..16),
|
||||
* each holding two packed 16-bit matrix elements. The first 1.5 rows
|
||||
* of a standard PSX SDK MATRIX struct (where each row is laid out as
|
||||
* [RT_xx, RT_xy] | [RT_xz, pad] | ...).
|
||||
*
|
||||
* Generated MIPS (mirrors the source macro):
|
||||
* lw $12, 0( %0 ) ; word 0
|
||||
* lw $13, 4( %0 ) ; word 1
|
||||
* ctc2 $12, $0 ; → C2_RT11
|
||||
* ctc2 $13, $1 ; → C2_RT12
|
||||
* lw $12, 8( %0 ) ; word 2
|
||||
* lw $13, 12( %0 ) ; word 3
|
||||
* lw $14, 16( %0 ) ; word 4
|
||||
* ctc2 $12, $2 ; → C2_RT13
|
||||
* ctc2 $13, $3 ; → C2_RT21
|
||||
* ctc2 $14, $4 ; → C2_RT22
|
||||
*
|
||||
* Same contract as gte_load_v0: caller MUST bind `r0` to $12 via a
|
||||
* register variable (`rgcc(R_T4)`) for the `lw $12, off(...)`
|
||||
* instructions to read from the right base. The `"r"(r0)` constraint
|
||||
* alone doesn't force a specific GPR — it just lets GCC pick one.
|
||||
* The .word constants here bake R_T4/R_T5/R_T6 into the `rs` field
|
||||
* of each lw, so the lw instructions will only do the right thing
|
||||
* if $12/$13/$14 hold the matrix base at runtime.
|
||||
*
|
||||
* M3_S2* m = ...;
|
||||
* register M3_S2* m_in_12 rgcc(R_T4) = m;
|
||||
* asm_gte_matrix_set_rotation(m_in_12);
|
||||
*
|
||||
* We clobber $12/$13/$14 (the ones we use as scratch inside the
|
||||
* inline asm) plus the system clobbers; we don't clobber `r0` because
|
||||
* the `rgcc` binding already says "this variable lives in $12".
|
||||
*
|
||||
* WARNING: Incomplete by design. The source macro only writes RT11..RT22
|
||||
* (5 of 9 rotation elements); RT23 and the entire RT3x row are left
|
||||
* untouched. Real libpsn00b SetRotMatrix writes all 9. Use only when the
|
||||
* GTE's remaining rotation entries are already correct, or you will
|
||||
* get stale-RT2x/RT3x artifacts in RTPS/RTPT/MVMVA output.
|
||||
*/
|
||||
#define asm_gte_matrix_set_rotation(r0) \
|
||||
asm volatile( \
|
||||
asm_words( \
|
||||
load_word(R_T5, R_T4, 0) \
|
||||
, load_word(R_T6, R_T4, 4) \
|
||||
, gte_mv_to_data_r( R_T5, 0) \
|
||||
, gte_mv_to_data_r( R_T6, 1) \
|
||||
, load_word(R_T5, R_T4, 8) \
|
||||
, load_word(R_T6, R_T4, 12) \
|
||||
, load_word(R_T4, R_T4, 16) \
|
||||
, gte_mv_to_data_r( R_T5, 2) \
|
||||
, gte_mv_to_data_r( R_T6, 3) \
|
||||
, gte_mv_to_data_r( R_T4, 4) \
|
||||
) \
|
||||
, r_use(r0) \
|
||||
asm_clobber: clbr_volatile_gprs, rlit(R_T4), rlit(R_T5), rlit(R_T6) \
|
||||
)
|
||||
|
||||
#pragma endregion ASM DSL
|
||||
|
||||
#pragma region Reserved
|
||||
|
||||
|
||||
|
||||
#pragma endregion Reserved
|
||||
@@ -0,0 +1,42 @@
|
||||
/* ============================================================================
|
||||
* duffle DSL — GTE Vendor Mnemonics (opt-in)
|
||||
* ============================================================================
|
||||
*
|
||||
* Provides the textbook MIPS assembly mnemonics for the GTE/COP2 instructions as thin aliases to the duffle macros in gte.h.
|
||||
* The duffle names are primary; this header is for users who prefer the textbook mnemonics.
|
||||
*
|
||||
* USAGE: #include "duffle/gte_vendor_sym.h" // after gte.h
|
||||
*
|
||||
* Mapping (vendor -> duffle):
|
||||
* Transfers (move GPR <-> GTE control/data register):
|
||||
* gte_mfc2 -> gte_mv_from_data_r (move from coprocessor 2 data reg)
|
||||
* gte_mtc2 -> gte_mv_to_data_r (move to coprocessor 2 data reg)
|
||||
* gte_cfc2 -> gte_mv_from_ctrl_r (move from coprocessor 2 control reg)
|
||||
* gte_ctc2 -> gte_mv_to_ctrl_r (move to coprocessor 2 control reg)
|
||||
*
|
||||
* Data load/store (load/store word to coprocessor 2 data register):
|
||||
* gte_lwc2(rt, base, off) -> gte_lw(rt, base, off)
|
||||
* gte_swc2(rt, base, off) -> gte_sw(rt, base, off)
|
||||
* (the lower-level vector variants gte_lw_v0_xy etc. don't have
|
||||
* vendor mnemonics; they're already gte_-prefixed and short)
|
||||
* ============================================================================ */
|
||||
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "gte.h"
|
||||
#endif
|
||||
|
||||
#ifndef DUFFLE_GTE_VENDOR_SYM_H
|
||||
#define DUFFLE_GTE_VENDOR_SYM_H
|
||||
|
||||
/* Transfers (move GPR <-> GTE control/data register) */
|
||||
#define gte_mfc2(rt, rd) gte_mv_from_data_r((rt), (rd))
|
||||
#define gte_mtc2(rt, rd) gte_mv_to_data_r((rt), (rd))
|
||||
#define gte_cfc2(rt, rd) gte_mv_from_ctrl_r((rt), (rd))
|
||||
#define gte_ctc2(rt, rd) gte_mv_to_ctrl_r((rt), (rd))
|
||||
|
||||
/* Data load/store (load/store word to coprocessor 2 data register) */
|
||||
#define gte_lwc2(rt, base, off) gte_lw((rt), (base), (off))
|
||||
#define gte_swc2(rt, base, off) gte_sw((rt), (base), (off))
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,356 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "dsl.h"
|
||||
# include "gcc_asm.h"
|
||||
# include "mips.h"
|
||||
# include "gte.h"
|
||||
# include "memory.h"
|
||||
# include "atom_dsl.h"
|
||||
# include "gen/duffle.macs.h"
|
||||
# include "gen/duffle.offsets.h"
|
||||
#endif
|
||||
|
||||
typedef U4 const MipsCode;
|
||||
typedef Slice_(MipsCode);
|
||||
typedef Slice_MipsCode MipsAtom;
|
||||
|
||||
#define MipsAtom_(sym) MipsCode sym [] align_(4) =
|
||||
|
||||
// Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names).
|
||||
// MipsAtomComp_(ac_X) { body }
|
||||
// expands to:
|
||||
// MipsCode ac_X[] align_(4) = { body };
|
||||
#define MipsAtomComp_(sym) MipsCode sym [] align_(4) =
|
||||
|
||||
// Used for components with value-args (e.g., ac_format_f3_color).
|
||||
// FI_ MipsAtom ac_X(args) MipsAtomComp_Proc_(ac_X, { body })
|
||||
// expands to:
|
||||
// FI_ MipsAtom ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return slice_from_array(MipsCode, ac_X); }
|
||||
#define MipsAtomComp_Proc_(sym, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; return slice_from_array(MipsCode, sym); }
|
||||
|
||||
// Auto-generated component macros (<module>/gen/<dir>/<dir>.macs.h) are included manually by the unity build.
|
||||
|
||||
/* Register aliases */
|
||||
enum {
|
||||
R_AtomJmp = R_T9 atom_reg, /* debug-visible; tape yield handshake scratch */
|
||||
R_TapePtr = R_T8 atom_reg, /* The Instruction Stream Pointer */
|
||||
R_InCursor = R_T4,
|
||||
|
||||
R_PrimCursor = R_T7 atom_reg atom_type(U4 *), /* VRAM output cursor (primitive buffer) */
|
||||
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2 *), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
|
||||
R_VertBase = R_T5 atom_reg atom_type(V3_S2 *), /* Base address of the vertex array */
|
||||
R_OtBase = R_T6 atom_reg atom_type(U4 *), /* Base address of the Ordering Table */
|
||||
|
||||
/* Stringification codes for the GCC inline assembler clobber lists. */
|
||||
#define R_TapePtr_Code R_T8_Code
|
||||
#define R_InCursor_Code R_T4_Code
|
||||
|
||||
#define R_PrimCursor_Code R_T7_Code
|
||||
#define R_FaceCursor_Code R_T4_Code
|
||||
#define R_VertBase_Code R_T5_Code
|
||||
#define R_OtBase_Code R_T6_Code
|
||||
};
|
||||
|
||||
#pragma region Tape Drive
|
||||
/* ---------------------------------------------------------------------------
|
||||
* TAPE DRIVE ABI & REGISTER ALIASES (the enum moved earlier; see below)
|
||||
* ---------------------------------------------------------------------------*/
|
||||
|
||||
/* The 'Exit' Atom */
|
||||
atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(rret_addr), nop };
|
||||
|
||||
/* Generalized Tape Engine Runner */
|
||||
NI_ void tape_run(Slice_MipsCode tape) { register U4* tp rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile(
|
||||
asm_words(
|
||||
add_ui( R_SP, R_SP, -MipsStackAlignment) /* Allocate stack space */
|
||||
, store_word( R_RA, R_SP, 0) /* Safely backup $ra to the stack */
|
||||
, load_word( R_AtomJmp, R_TapePtr, 0) /* Bootstrap the first jump */
|
||||
, add_ui_self(R_TapePtr, S_(MipsCode)) /* Advance tape */
|
||||
, call_reg( R_AtomJmp) /* jalr $t9 */
|
||||
, nop /* Branch delay slot */
|
||||
, load_word( R_RA, R_SP, 0) /* Restore $ra from stack */
|
||||
, add_ui_self(R_SP, MipsStackAlignment) /* Deallocate stack space */
|
||||
)
|
||||
asm_rpins, r_use(tp)
|
||||
asm_clobber:
|
||||
rlit(R_AT)
|
||||
, rlit(R_V0), rlit(R_V1)
|
||||
, rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3)
|
||||
/* Tell GCC the tape engine owns and destroys the workspace registers */
|
||||
, rlit(R_PrimCursor), rlit(R_FaceCursor), rlit(R_VertBase), rlit(R_OtBase)
|
||||
, rlit(R_T9)
|
||||
, clb_mem_drain
|
||||
); }
|
||||
|
||||
typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; };
|
||||
FI_ void tb_init(TapeBuilder* tb, FArena* arena) { tb->ptr = arena->start; tb->used = 0; }
|
||||
FI_ TapeBuilder tb_make_old( FArena* arena) { return (TapeBuilder){ arena->start, 0 }; }
|
||||
FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ mem.ptr, mem.len, 0 }; }
|
||||
|
||||
#define tb_emit_(tb, atom) tb_emit(tb, atom)
|
||||
FI_ void tb_emit(TapeBuilder* tb, MipsCode* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
|
||||
FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; }
|
||||
|
||||
FI_ Slice_MipsCode tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Slice_MipsCode){ C_(U4*,tb->ptr), tb->used }; }
|
||||
FI_ Slice_MipsCode tb_slice(TapeBuilder tb) { return (Slice_MipsCode){ C_(U4*,tb.ptr), tb.used }; }
|
||||
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit))
|
||||
|
||||
#pragma endregion Tape Drive
|
||||
|
||||
#pragma region Macro Mips Atom Components
|
||||
/* ---------------------------------------------------------------------------
|
||||
* MACRO ATOM Components (Reusable Assembly Components)
|
||||
* These do NOT yield. They are expanded inline inside Tape Atoms.
|
||||
* ---------------------------------------------------------------------------*/
|
||||
|
||||
// The 'Yield' sequence for Tape Atoms (mac_yield).
|
||||
atom_dbg_skip MipsAtomComp_(ac_yield) {
|
||||
load_word(R_AtomJmp, R_TapePtr, 0),
|
||||
add_ui_self( R_TapePtr, S_(MipsCode)),
|
||||
jump_reg( R_AtomJmp), nop,
|
||||
};
|
||||
|
||||
/* Words: 3; Loads 3 S2 indices from the face array */
|
||||
atom_dbg_skip MipsAtomComp_(ac_load_tri_indices) {
|
||||
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
||||
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
||||
};
|
||||
|
||||
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
|
||||
atom_dbg_skip MipsAtomComp_(ac_gte_load_tri_verts) {
|
||||
shift_lleft(R_AT, R_T0, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
shift_lleft(R_AT, R_T1, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
||||
shift_lleft(R_AT, R_T2, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
||||
};
|
||||
|
||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
|
||||
* Hardcoded for Poly_F3 (5 words). For Poly_G4, use ac_insert_ot_tag_g4. */
|
||||
MipsAtomComp_(ac_insert_ot_tag_f3) {
|
||||
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
|
||||
add_u_self( R_T1, R_OtBase), // T1 = & OrderingTable[OTZ]
|
||||
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
|
||||
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits), // V0 = (5 - 1) << 24 = 4 << 24
|
||||
mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
|
||||
or_u( R_AT, R_AT, R_V0), // Merge length
|
||||
store_word( R_AT, R_PrimCursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr)
|
||||
shift_lleft( R_AT, R_PrimCursor, S_(PolyTag_len_bits)), // AT = (prim_length << 24) | old_addr
|
||||
shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
|
||||
store_word( R_AT, R_T1, O_(PolyTag,code)), // OrderingTable[OTZ] = PrimCursor
|
||||
};
|
||||
|
||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
|
||||
* Hardcoded for Poly_G4 (9 words). For Poly_F3, use ac_insert_ot_tag_f3. */
|
||||
MipsAtomComp_(ac_insert_ot_tag_g4) {
|
||||
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
|
||||
add_u_self( R_T1, R_OtBase), // T1 = & OrderingTable[OTZ]
|
||||
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
|
||||
load_upper_i(R_V0, (S_(Poly_G4)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits), // V0 = (9 - 1) << 24 = 8 << 24
|
||||
mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
|
||||
or_u( R_AT, R_AT, R_V0), // Merge length
|
||||
store_word( R_AT, R_PrimCursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr)
|
||||
shift_lleft( R_AT, R_PrimCursor, S_(PolyTag_len_bits)), // AT = (prim_length << 24) | old_addr
|
||||
shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
|
||||
store_word( R_AT, R_T1, O_(PolyTag,code)), // OrderingTable[OTZ] = PrimCursor
|
||||
};
|
||||
|
||||
/* Words: 3; Emits one (cmd|color) word to R_PrimCursor at the given
|
||||
* byte offset. Internal helper used by the *_format_*_color macros. */
|
||||
FI_ MipsAtom ac_pack_color_word(U4 off, U4 cmd, U1 r, U1 g, U1 b)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, {
|
||||
load_upper_i(R_AT, (cmd) << 8 | (b)),
|
||||
or_i_self( R_AT, ((g) << 8) | (r)),
|
||||
store_word( R_AT, R_PrimCursor, (off)),
|
||||
})
|
||||
|
||||
/* Words: 3; Emits the F3 command+color word (cmd byte | BLUE | GREEN | RED)
|
||||
* Args: _r, _g, _b are 8-bit RGB byte values (not raw 16-bit fields). */
|
||||
FI_ MipsAtom ac_format_f3_color(U1 r, U1 g, U1 b)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, { mac_pack_color_word(O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
|
||||
|
||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
|
||||
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
|
||||
atom_dbg_skip MipsAtomComp_(ac_gte_store_f3_post_rtpt) {
|
||||
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_F3,p0)),
|
||||
gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_F3,p1)),
|
||||
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_F3,p2)),
|
||||
};
|
||||
|
||||
/* Words: 12; Emits the four (code|color) words of a Poly_G4.
|
||||
* Args: rN,gN,bN are 8-bit RGB byte values for each of the 4 vertices. */
|
||||
FI_ MipsAtom ac_format_g4_color(
|
||||
U1 r0, U1 g0, U1 b0,
|
||||
U1 r1, U1 g1, U1 b1,
|
||||
U1 r2, U1 g2, U1 b2,
|
||||
U1 r3, U1 g3, U1 b3)
|
||||
MipsAtomComp_Proc_(ac_format_g4_color, {
|
||||
mac_pack_color_word(O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0),
|
||||
mac_pack_color_word(O_(Poly_G4,c1), 0, r1,g1,b1),
|
||||
mac_pack_color_word(O_(Poly_G4,c2), 0, r2,g2,b2),
|
||||
mac_pack_color_word(O_(Poly_G4,c3), 0, r3,g3,b3),
|
||||
})
|
||||
|
||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
|
||||
* G4 triangle portion to p0/p1/p2.
|
||||
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
|
||||
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2
|
||||
* get overwritten with v3 (RTPS writes only to SXY2, but to keep the
|
||||
* three registers aligned with v0/v1/v2 you must store before RTPS).
|
||||
* The macro name declares the pipeline position; check #6 (GTE state-
|
||||
* machine validation) verifies the call site matches the declaration. */
|
||||
atom_dbg_skip MipsAtomComp_(ac_gte_store_g4_p012_post_rtpt_pre_rtps) {
|
||||
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_G4,p0)),
|
||||
gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_G4,p1)),
|
||||
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2)),
|
||||
};
|
||||
|
||||
/* Words: 1; Stores the V3 screen coord to the G4's p3 slot.
|
||||
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its
|
||||
* single-vertex result to SXY2; SXY0 still holds v0.screen from the
|
||||
* earlier RTPT — DO NOT read SXY0 here, that's the bug this name
|
||||
* prevents).
|
||||
*/
|
||||
atom_dbg_skip MipsAtomComp_(ac_gte_store_g4_p3_post_rtps) { gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p3)) };
|
||||
|
||||
#pragma endregion Macro Atom Components
|
||||
|
||||
#pragma region Mips Atom Builder
|
||||
// This allows for runtime procedural authoring of mips atoms.
|
||||
|
||||
typedef Struct_(FMipsAtom512) { U4 data[512]; U4 used; };
|
||||
|
||||
// FArena Related
|
||||
typedef Relative_(FArena) Struct_(MipsAtomBuilder) { U4 start; U4 capacity; U4 used; };
|
||||
// Whatever the builder is writting to should most likely coresspond
|
||||
// to something that can fit within instruction cache?
|
||||
|
||||
FI_ void atombuilder_unroll(MipsAtomBuilder_R ab, Slice_MipsCode_R code) {
|
||||
assert(ab->capacity - ab->used - code->len);
|
||||
mem_copy(ab->start, u4_(code->ptr), code->len);
|
||||
mem_bump(ab->start, ab->capacity, & ab->used, code->len);
|
||||
}
|
||||
#define atombuilder_unroll_mac(ab, mac) atombuilder_unroll(ab, slice_arg_from_array(Slice_MipsCode, mac))
|
||||
|
||||
// When done authoring, utilize this to cap-off the atom
|
||||
FI_ void atombuilder_end(MipsAtomBuilder_R ab) {
|
||||
mem_copy(ab->start, u4_(ac_yield), S_(ac_yield));
|
||||
mem_bump(ab->start, ab->capacity, & ab->used, S_(ac_yield));
|
||||
}
|
||||
|
||||
#define mipsatom_from_builder(ab) (MipsAtom){ab.start, ab.used}
|
||||
|
||||
#pragma endregion Mips Atom Builder
|
||||
|
||||
#pragma region Baked Mips Atoms
|
||||
// These atoms are resolved at compile time and are (usually) statically linked readonly data.
|
||||
|
||||
enum {
|
||||
bios_flushcache = 0x44,
|
||||
bios_table_addr = 0xA0,
|
||||
};
|
||||
|
||||
/* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0).
|
||||
* Sequence (per MIPS ABI; arguments in arg registers, RA pushed to stack):
|
||||
* 1. sp -= 8; sw $ra, 4($sp) ; save RA
|
||||
* 2. $a0 = bios_flushcache (arg0)
|
||||
* 3. $t0 = bios_table_addr ; t0 = &BIOS A-function table
|
||||
* 4. jalr $t0, $ra ; call BIOS(flushcache)
|
||||
* nop ; branch delay slot
|
||||
* 5. lw $ra, 4($sp); jr $ra ; restore & return
|
||||
* 6. sp += 8
|
||||
*/
|
||||
internal MipsAtom_(mips_flush_icache) {
|
||||
add_ui(rstack_ptr, rstack_ptr, -MipsStackAlignment), // sp -= 8
|
||||
store_word(rret_addr, rstack_ptr, S_(U4)), // sw $ra, 4($sp)
|
||||
add_ui(rret_0, rdiscard, bios_flushcache), // addiu $a0, $0, 0x44
|
||||
add_ui(rtmp_0, rdiscard, bios_table_addr), // addiu $t0, $0, 0xA0
|
||||
jump_link(rtmp_0, rret_addr), nop, // jalr $t0, $ra, BD slot
|
||||
load_word(rret_addr, rstack_ptr, S_(U4)), // lw $ra, 4($sp)
|
||||
jump_reg(rret_addr), // jr $ra
|
||||
add_ui(rstack_ptr, rstack_ptr, MipsStackAlignment), // sp += 8 (BD)
|
||||
mac_yield(),
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_SetGteWorld) {
|
||||
M3_S2* transform;
|
||||
};
|
||||
internal MipsAtom_(set_gte_world) atom_info(
|
||||
atom_bind(Binds_SetGteWorld)
|
||||
, atom_reads(R_TapePtr)
|
||||
){
|
||||
/* Pop matrix address from tape into R_T3 ($11) */
|
||||
load_word(R_T3, R_TapePtr, O_(Binds_SetGteWorld,transform)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_SetGteWorld)),
|
||||
/* Load 3x3 Rotation + 3x1 Translation from R_T3 into GTE CONTROL Regs (ctc2) */
|
||||
load_word(R_T0, R_T3, 0), load_word(R_T1, R_T3, 4),
|
||||
gte_mv_to_ctrl_r(R_T0, gte_cr_RT11), gte_mv_to_ctrl_r(R_T1, gte_cr_RT12),
|
||||
load_word(R_T0, R_T3, 8), load_word(R_T1, R_T3, 12), load_word(R_T2, R_T3, 16),
|
||||
gte_mv_to_ctrl_r(R_T0, gte_cr_RT13), gte_mv_to_ctrl_r(R_T1, gte_cr_RT21), gte_mv_to_ctrl_r(R_T2, gte_cr_RT22),
|
||||
load_word(R_T0, R_T3, 20), load_word(R_T1, R_T3, 24), load_word(R_T2, R_T3, 28),
|
||||
gte_mv_to_ctrl_r(R_T0, gte_cr_TRX), gte_mv_to_ctrl_r(R_T1, gte_cr_TRY), gte_mv_to_ctrl_r(R_T2, gte_cr_TRZ),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
/* DIAGNOSTIC 1: Pure tape loop test */
|
||||
internal MipsAtom_(diag_yield) { mac_yield() };
|
||||
|
||||
/* DIAGNOSTIC 2: Pure memory test (No GTE). Draws a fixed cyan triangle. */
|
||||
internal MipsAtom_(diag_color) {
|
||||
store_word( R_0, R_T7, 0),
|
||||
load_upper_i(R_AT, gp0_cmd_poly_f3 << 8 | 0xFF), /* High: MipsCode Poly_F3(0x20) + Color B:FF */
|
||||
or_i_self( R_AT, 0xFF00), /* Low: Color G:FF, R:00 (Cyan) */
|
||||
store_word( R_AT, R_T7, 4),
|
||||
|
||||
/* Fake coordinates - Swapped winding order to prevent GPU culling! */
|
||||
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 8), /* (16, 16) */
|
||||
load_upper_i(R_AT, 0x0050), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 12), /* (80, 16) */
|
||||
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0050), store_word(R_AT, R_T7, 16), /* (16, 80) */
|
||||
|
||||
add_ui( R_T1, R_0, 10),
|
||||
shift_lleft_self(R_T1, S_(U4)/2),
|
||||
add_u_self( R_T1, R_T6),
|
||||
|
||||
load_word( R_AT, R_T1, 0),
|
||||
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits),
|
||||
store_word( R_AT, R_T7, 0),
|
||||
shift_lleft(R_AT, R_T7, S_(PolyTag_len_bits)), shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
|
||||
or_u_self( R_AT, R_V0),
|
||||
store_word( R_AT, R_T1, 0),
|
||||
|
||||
add_ui(R_T7, R_T7, 20),
|
||||
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
/* DIAGNOSTIC 3: Pure GTE test (No Memory Writes) */
|
||||
internal MipsAtom_(diag_gte) {
|
||||
/* Load 3 indices */
|
||||
load_half_u(R_T0, R_T4, 0),
|
||||
load_half_u(R_T1, R_T4, 2),
|
||||
load_half_u(R_T2, R_T4, 4),
|
||||
|
||||
/* Load Vertices into GTE */
|
||||
shift_lleft( R_AT, R_T0, 3), add_u( R_AT, R_AT, R_T5),
|
||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
|
||||
shift_lleft( R_AT, R_T1, 3), add_u(R_AT, R_AT, R_T5),
|
||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
||||
|
||||
shift_lleft(R_AT, R_T2, 3), add_u(R_AT, R_AT, R_T5),
|
||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
||||
|
||||
/* Run Math */
|
||||
nop2, gte_cmdw_rtpt,
|
||||
nop2, gte_cmdw_nclip,
|
||||
nop2,
|
||||
|
||||
/* Advance Face Cursor and Yield */
|
||||
add_ui(R_T4, R_T4, 8),
|
||||
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
#pragma endregion Baked Mips Atoms
|
||||
+26
-22
@@ -7,34 +7,38 @@
|
||||
#define max(A, B) (((A) > (B)) ? (A) : (B))
|
||||
#define clamp_bot(X, B) max(X, B)
|
||||
|
||||
typedef def_farray(U4, 2);
|
||||
typedef def_farray(S2, 2);
|
||||
typedef def_farray(S2, 3);
|
||||
typedef def_farray(S4, 2);
|
||||
typedef def_farray(S4, 3);
|
||||
typedef def_farray(S4, 4);
|
||||
typedef S2 A3A3_S2[3][3];
|
||||
enum {
|
||||
v3s2_byteoff = 3, // log2(8), used with shift_left_logical op for index via byte offset.
|
||||
};
|
||||
|
||||
typedef def_struct(Extent2_S2) { S2 width; S2 height; };
|
||||
typedef def_struct(Extent2_S4) { S4 width; S4 height; };
|
||||
typedef Array_(U4, 2);
|
||||
typedef Array_(S2, 2);
|
||||
typedef Array_(S2, 3);
|
||||
typedef Array_(S4, 2);
|
||||
typedef Array_(S4, 3);
|
||||
typedef Array_(S4, 4);
|
||||
typedef S2 A3x3_S2[3][3];
|
||||
|
||||
typedef def_struct(V2_S2) { S2 x; S2 y; };
|
||||
typedef def_struct(V2_S4) { S4 x; S4 y; };
|
||||
typedef def_struct(V3_S2) { S2 x; S2 y; S2 z; S2 pad; };
|
||||
typedef def_struct(V3_S4) { S4 x; S4 y; S4 z; S4 pad; };
|
||||
typedef def_struct(V4_S2) { S2 x; S2 y; S2 z; S2 w; };
|
||||
typedef def_struct(V4_S4) { S4 x; S4 y; S4 z; S4 w; };
|
||||
typedef Struct_(Extent2_S2) { S2 width; S2 height; };
|
||||
typedef Struct_(Extent2_S4) { S4 width; S4 height; };
|
||||
|
||||
typedef def_struct(R2_S2) { V2_S2 p0; V2_S2 p1; };
|
||||
typedef def_struct(R2_S4) { V2_S4 p0; V2_S4 p1; };
|
||||
typedef Struct_(V2_S2) { S2 x; S2 y; };
|
||||
typedef Struct_(V2_S4) { S4 x; S4 y; };
|
||||
typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; };
|
||||
typedef Struct_(V3_S4) { S4 x; S4 y; S4 z; S4 pad; };
|
||||
typedef Struct_(V4_S2) { S2 x; S2 y; S2 z; S2 w; };
|
||||
typedef Struct_(V4_S4) { S4 x; S4 y; S4 z; S4 w; };
|
||||
|
||||
typedef def_struct(Rect_S2) { S2 x; S2 y; S2 width; S2 height; };
|
||||
typedef def_struct(Rect_S4) { S4 x; S4 y; S4 width; S4 height; };
|
||||
typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; };
|
||||
typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; };
|
||||
|
||||
typedef def_struct(M3_S2) { A3A3_S2 m; A3_S4 t; };
|
||||
typedef Struct_(Rect_S2) { S2 x; S2 y; S2 width; S2 height; };
|
||||
typedef Struct_(Rect_S4) { S4 x; S4 y; S4 width; S4 height; };
|
||||
|
||||
typedef def_farray(V2_S2, 3);
|
||||
typedef def_farray(V2_S2, 4);
|
||||
typedef Struct_(M3_S2) { A3x3_S2 m; A3_S4 t; };
|
||||
|
||||
typedef Array_(V2_S2, 3);
|
||||
typedef Array_(V2_S2, 4);
|
||||
|
||||
#define v2s2(x,y) (V2_S2){x,y}
|
||||
#define v3s2(x,y,z) (V3_S2){x,y,z,0}
|
||||
|
||||
+84
-102
@@ -3,7 +3,14 @@
|
||||
# include "dsl.h"
|
||||
#endif
|
||||
|
||||
inline U4 align_pow2(U4 x, U4 b) {
|
||||
#define MEM_ALIGNMENT_DEFAULT 4
|
||||
|
||||
#define assert_bounds(point, start, end) for(;0;){ \
|
||||
assert((start) <= (point)); \
|
||||
assert((point) <= (end)); \
|
||||
} while(0)
|
||||
|
||||
I_ U4 align_pow2(U4 x, U4 b) {
|
||||
assert(b != 0);
|
||||
assert((b & (b - 1)) == 0); // Check power of 2
|
||||
return ((x + b - 1) & (~(b - 1)));
|
||||
@@ -11,17 +18,17 @@ inline U4 align_pow2(U4 x, U4 b) {
|
||||
|
||||
#define align_struct(type_width) ((U4)(((type_width) + 3) & ~3))
|
||||
|
||||
#define assert_bounds(point, start, end) do { \
|
||||
U4 pos_point = cast(U4, point); \
|
||||
U4 pos_start = cast(U4, start); \
|
||||
U4 pos_end = cast(U4, end); \
|
||||
assert(pos_start <= pos_point); \
|
||||
assert(pos_point <= pos_end); \
|
||||
} while(0)
|
||||
FI_ void mem_bump(U4 start, U4 cap, U4*R_ used, U4 amount) {
|
||||
assert(amount <= (cap - used[0]));
|
||||
used[0] += amount;
|
||||
}
|
||||
|
||||
void* memory_copy (void* restrict dest, void const* restrict src, U4 length) __asm__("memcpy");
|
||||
void* memory_copy_overlapping(void* restrict dest, void const* restrict src, U4 length);
|
||||
B4 memory_zero (void* dest, U4 length);
|
||||
FI_ U4 mem_copy (U4 dest, U4 src, U4 len) { return (U4)(__builtin_memcpy ((void*)dest, (void const*)src, len)); }
|
||||
FI_ U4 mem_copy_overlapping(U4 dest, U4 src, U4 len) { return (U4)(__builtin_memmove((void*)dest, (void const*)src, len)); }
|
||||
FI_ U4 mem_fill (U4 dest, U4 value, U4 len) { return (U4)(__builtin_memset ((void*)dest, (int) value, len)); }
|
||||
FI_ B4 mem_zero (U4 dest, U4 len) { if(dest == 0){return false;} mem_fill(dest, 0, len); return true; }
|
||||
|
||||
#pragma region DAG
|
||||
|
||||
#define check_nil(nil, p) ((p) == 0 || (p) == nil)
|
||||
#define set_nil(nil, p) ((p) = nil)
|
||||
@@ -42,101 +49,76 @@ B4 memory_zero (void* dest, U4 length);
|
||||
)
|
||||
#define sll_queue_push_n(f, l, n, next) sll_queue_push_nz(0, f, l, n, next)
|
||||
|
||||
#pragma region Allocator Interface
|
||||
#if 0
|
||||
typedef def_enum(U4, AllocatorOp) {
|
||||
AllocatorOp_Alloc_NoZero = 0, // If Alloc exist, so must No_Zero
|
||||
AllocatorOp_Alloc,
|
||||
AllocatorOp_Free,
|
||||
AllocatorOp_Reset,
|
||||
AllocatorOp_Grow_NoZero,
|
||||
AllocatorOp_Grow,
|
||||
AllocatorOp_Shrink,
|
||||
AllocatorOp_Rewind,
|
||||
AllocatorOp_SavePoint,
|
||||
AllocatorOp_Query, // Must always be implemented
|
||||
};
|
||||
typedef def_enum(U4, AllocatorQueryFlags) {
|
||||
AllocatorQuery_Alloc = (1 << 0),
|
||||
AllocatorQuery_Free = (1 << 1),
|
||||
// Wipe the allocator's state
|
||||
AllocatorQuery_Reset = (1 << 2),
|
||||
// Supports both grow and shrink
|
||||
AllocatorQuery_Shrink = (1 << 4),
|
||||
AllocatorQuery_Grow = (1 << 5),
|
||||
AllocatorQuery_Resize = AllocatorQuery_Grow | AllocatorQuery_Shrink,
|
||||
// Ability to rewind to a save point (ex: arenas, stack), must also be able to save such a point
|
||||
AllocatorQuery_Rewind = (1 << 6),
|
||||
};
|
||||
typedef struct AllocatorProc_In AllocatorProc_In;
|
||||
typedef struct AllocatorProc_Out AllocatorProc_Out;
|
||||
typedef void def_proc(AllocatorProc) (AllocatorProc_In In, AllocatorProc_Out* Out);
|
||||
typedef def_struct(AllocatorSP) {
|
||||
AllocatorProc* type_sig;
|
||||
U4 slot;
|
||||
};
|
||||
struct AllocatorProc_In {
|
||||
void* data;
|
||||
U4 requested_size;
|
||||
U4 alignment;
|
||||
union {
|
||||
Slice_B1 old_allocation;
|
||||
AllocatorSP save_point;
|
||||
};
|
||||
AllocatorOp op;
|
||||
byte_pad(4);
|
||||
};
|
||||
struct AllocatorProc_Out {
|
||||
union {
|
||||
Slice_B1 allocation;
|
||||
AllocatorSP save_point;
|
||||
};
|
||||
AllocatorQueryFlags features;
|
||||
U4 left; // Contiguous memory left
|
||||
U4 max_alloc;
|
||||
U4 min_alloc;
|
||||
// byte_pad(8);
|
||||
};
|
||||
typedef def_struct(AllocatorInfo) {
|
||||
AllocatorProc* proc;
|
||||
void* data;
|
||||
};
|
||||
static_assert(size_of(AllocatorSP) <= size_of(Slice_B1));
|
||||
typedef def_struct(AllocatorQueryInfo) {
|
||||
AllocatorSP save_point;
|
||||
AllocatorQueryFlags features;
|
||||
U4 left; // Contiguous memory left
|
||||
U4 max_alloc;
|
||||
U4 min_alloc;
|
||||
// byte_pad(4);
|
||||
};
|
||||
static_assert(size_of(AllocatorProc_Out) == size_of(AllocatorQueryInfo));
|
||||
#pragma endregion DAG
|
||||
|
||||
#define MEMORY_ALIGNMENT_DEFAULT (2 * size_of(void*))
|
||||
#pragma region Slice
|
||||
|
||||
AllocatorQueryInfo allocator_query(AllocatorInfo ainfo);
|
||||
typedef unsigned char UTF8;
|
||||
typedef Struct_(Str8) { UTF8* ptr; U4 len; };
|
||||
typedef Struct_(Slice_Str8) { Str8* ptr; U4 len; };
|
||||
#define slit(string_literal) (Str8){ (UTF8*) string_literal, S_(string_literal) - 1 }
|
||||
|
||||
void mem_free (AllocatorInfo ainfo, Slice_B1 mem);
|
||||
void mem_reset (AllocatorInfo ainfo);
|
||||
void mem_rewind (AllocatorInfo ainfo, AllocatorSP save_point);
|
||||
AllocatorSP mem_save_point(AllocatorInfo ainfo);
|
||||
typedef Struct_(Slice) { U4 ptr, len; }; // Untyped Slice
|
||||
FI_ Slice slice_ut_(U4 ptr, U4 len) { return (Slice){ptr, len}; }
|
||||
|
||||
typedef def_struct(Opts_mem_alloc) { U4 alignment; B4 no_zero; byte_pad(4); };
|
||||
typedef def_struct(Opts_mem_grow) { U4 alignment; B4 no_zero; byte_pad(4); };
|
||||
typedef def_struct(Opts_mem_shrink) { U4 alignment; };
|
||||
typedef def_struct(Opts_mem_resize) { U4 alignment; B4 no_zero; byte_pad(4); };
|
||||
#define Slice_(type) Struct_(tmpl(Slice,type)) { type* ptr; U4 len; }
|
||||
typedef Slice_(B1);
|
||||
#define slice_assert(s) do { assert((s).ptr != 0); assert((s).len > 0); } while(0)
|
||||
#define slice_end(slice) ((slice).ptr + (slice).len)
|
||||
#define S_slice(s) ((s).len * S_((s).ptr[0]))
|
||||
|
||||
Slice_B1 mem__alloc (AllocatorInfo ainfo, U4 size, Opts_mem_alloc* opts);
|
||||
Slice_B1 mem__grow (AllocatorInfo ainfo, Slice_B1 mem, U4 size, Opts_mem_grow* opts);
|
||||
Slice_B1 mem__resize(AllocatorInfo ainfo, Slice_B1 mem, U4 size, Opts_mem_resize* opts);
|
||||
Slice_B1 mem__shrink(AllocatorInfo ainfo, Slice_B1 mem, U4 size, Opts_mem_shrink* opts);
|
||||
#define slice_ut(ptr,len) slice_ut_(u4_(ptr), u4_(len))
|
||||
#define slice_ut_arr(a) slice_ut_(u4_(a), S_(a))
|
||||
#define slice_to_ut(s) slice_ut_(u4_((s).ptr), S_slice(s))
|
||||
|
||||
#define mem_alloc(ainfo, size, ...) mem__alloc (ainfo, size, opt_args(Opts_mem_alloc, __VA_ARGS__))
|
||||
#define mem_grow(ainfo, mem, size, ...) mem__grow (ainfo, mem, size, opt_args(Opts_mem_grow, __VA_ARGS__))
|
||||
#define mem_resize(ainfo, mem, size, ...) mem__resize(ainfo, mem, size, opt_args(Opts_mem_resize, __VA_ARGS__))
|
||||
#define mem_shrink(ainfo, mem, size, ...) mem__shrink(ainfo, mem, size, opt_args(Opts_mem_shrink, __VA_ARGS__))
|
||||
#define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter)
|
||||
#define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = array_decl(type,__VA_ARGS__), .len = array_len( array_decl(type,__VA_ARGS__)) }
|
||||
#define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = S_(array) }
|
||||
|
||||
#define alloc_type(ainfo, type, ...) (type*) mem__alloc(ainfo, size_of(type), opt_args(Opts_mem_alloc, __VA_ARGS__)).ptr
|
||||
#define alloc_slice(ainfo, type, num, ...) (tmpl(Slice,type)){ mem__alloc(ainfo, size_of(type) * num, opt_args(Opts_mem_alloc, __VA_ARGS__)).ptr, num }
|
||||
#endif
|
||||
#pragma endregion Allocator Interface
|
||||
FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(s.ptr, s.len); }
|
||||
#define slice_zero(s) slice_zero_(slice_to_ut(s))
|
||||
|
||||
FI_ void slice_copy_(Slice dest, Slice src) {
|
||||
assert(dest.len >= src.len);
|
||||
slice_assert(dest);
|
||||
slice_assert(src);
|
||||
mem_copy(dest.ptr, src.ptr, src.len);
|
||||
}
|
||||
#define slice_copy(dest, src) do { \
|
||||
static_assert(T_same(dest, src)); \
|
||||
slice_copy_(slice_to_ut(dest), slice_to_ut(src)); \
|
||||
} while(0)
|
||||
|
||||
typedef Slice_(U4);
|
||||
|
||||
#pragma endregion Slice
|
||||
|
||||
#pragma region FArena
|
||||
|
||||
typedef Opt_(farena) { U4 alignment, type_width; };
|
||||
typedef Struct_(FArena) { U4 start, capacity, used; };
|
||||
FI_ void farena_init(FArena_R arena, Slice mem) { assert(arena != nullptr);
|
||||
arena->start = mem.ptr;
|
||||
arena->capacity = mem.len;
|
||||
arena->used = 0;
|
||||
}
|
||||
FI_ FArena farena_make(Slice mem) { FArena a; farena_init(& a, mem); return a; }
|
||||
I_ Slice farena_push(FArena_R arena, U4 amount, Opt_farena o) {
|
||||
if (amount == 0) { return (Slice){}; }
|
||||
U4 desired = amount * (o.type_width == 0 ? 1 : o.type_width);
|
||||
U4 to_commit = align_pow2(desired, o.alignment ? o.alignment : MEM_ALIGNMENT_DEFAULT);
|
||||
U4 ptr = arena->start + arena->used;
|
||||
mem_bump(arena->start, arena->capacity, & arena->used, to_commit);
|
||||
return (Slice){ ptr, to_commit };
|
||||
}
|
||||
FI_ void farena_reset (FArena_R arena) { arena->used = 0; }
|
||||
FI_ void farena_rewind(FArena_R arena, U4 save_point) {
|
||||
U4 end = arena->start + arena->used; assert_bounds(save_point, arena->start, end);
|
||||
arena->used -= save_point - arena->start;
|
||||
}
|
||||
FI_ U4 farena_save(FArena arena) { return arena.used; }
|
||||
#define farena_push_(arena, amount, ...) farena_push((arena), (amount), opt_(farena, __VA_ARGS__))
|
||||
#define farena_push_type(arena, type, ...) C_(type*, farena_push((arena), 1, opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr)
|
||||
#define farena_push_array(arena, type, amount, ...) (tmpl(Slice,type)){ C_(type*, farena_push((arena), (amount), opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr), (amount) }
|
||||
|
||||
#pragma endregion FArena
|
||||
|
||||
@@ -0,0 +1,567 @@
|
||||
/* ============================================================================
|
||||
* duffle DSL Suffix Conventions
|
||||
* ============================================================================
|
||||
* Every mnemonic in this header follows the same suffix grammar:
|
||||
* _i: Immediate value (16-bit constant operand).
|
||||
* Combine with _u or _s (single-letter modifier + type combined): add_ui, add_si.
|
||||
* Examples: add_ui, add_si, and_i, or_i, xor_i, load_upper_i. and_i is sign-agnostic (andi zero-extends).
|
||||
* load_upper_i is a unique verb; _i is the immediate marker, not a modifier+type combination.
|
||||
* _u: Unsigned (no-overflow, no-sign-extension).
|
||||
* R-type arithmetic examples: add_u, sub_u, mult_u, div_u. I-type (combined with _i): add_ui.
|
||||
* _s: Signed (overflow-traps, sign-extends).
|
||||
* R-type: add_s, sub_s, mult_s, div_s, set_lt_s. I-type (combined with _i): add_si.
|
||||
*
|
||||
* --- Shift family (R-type): verb-modifier-direction ---
|
||||
* The shift macros use `shift_<modifier><direction>`.
|
||||
* Modifier is the single letter `l` (logical) or `a` (arithmetic).
|
||||
* Direction is the word `left` or `right`. Combined: `_lleft`, `_lright`, `_aright`.
|
||||
* Examples: shift_lleft( rd, rt, shamt) (= sll)
|
||||
* shift_lright(rd, rt, shamt) (= srl)
|
||||
* shift_aright(rd, rt, shamt) (= sra)
|
||||
* (no `_aleft`; MIPS has no `sla` — arithmetic-left is bit-identical to logical-left, so use shift_lleft for that case)
|
||||
*
|
||||
* --- Jump/Call family ---
|
||||
* Simple jumps keep the original short names: jump (j), jump_reg (jr), jump_link (jalr rs, rd).
|
||||
* The jump-and-link-to variants (jal, jalr rs with default $ra) get the `call_` verb instead:
|
||||
* call_addr (jal), call_reg (jalr rs, default $ra).
|
||||
* Examples: jump(off) (= j)
|
||||
* jump_reg(rs) (= jr)
|
||||
* jump_link(rs, rd) (= jalr rs, rd)
|
||||
* call_reg(rs) (= jalr rs, default $ra)
|
||||
* call_addr(off) (= jal)
|
||||
*
|
||||
* _r: Register marker — used only when the register type needs disambiguation (e.g., GTE data register vs control register).
|
||||
* NOT used in plain R-type arithmetic (the R-type is implicit). Examples: gte_mv_to_data_r, gte_mv_to_ctrl_r.
|
||||
* _self: Destination equals one source operand.
|
||||
* Examples: add_ui_self (I-type, to self), add_u_self (R-type, to self).
|
||||
* _mv_to_: Direction: data flows into X.
|
||||
* Example: gte_mv_to_data_r, gte_mv_to_ctrl_r.
|
||||
* _mv_from_: Direction: data flows out of X.
|
||||
* Example: gte_mv_from_data_r, gte_mv_from_ctrl_r.
|
||||
* _str: String-form — emits inline-asm string instead of `.word`.
|
||||
* Example: gte_rtpt_asm_str.
|
||||
* _2w / _1w: Word count of the emitted sequence.
|
||||
* Example: load_imm_2w.
|
||||
*
|
||||
* _cop2: RESERVED — DO NOT USE in macro names. The `gte_` namespace prefix already implies coprocessor 2. Use `c2` only in:
|
||||
* (a) integer opcode enums (op_lwc2 = 0x32, op_swc2 = 0x3A)
|
||||
* (b) vendor-mnemonic macro aliases (gte_mtc2, gte_mfc2)
|
||||
*
|
||||
* Primitive commands: gp0_cmd_poly_f3 = 0x20 (byte opcode)
|
||||
* Packed 32-bit cmd: gp0_word_poly_f3(r, g, b) (32-bit, shifted)
|
||||
*
|
||||
* Type ordering: domain?_(direction)?_action_target_modifier_type?
|
||||
* Examples: add_ui (add + unsigned + immediate)
|
||||
* add_s (add + signed, R-type implicit)
|
||||
* shift_lleft (shift + logical + left)
|
||||
* shift_aright (shift + arithmetic + right)
|
||||
* call_reg(rs) (call + register, $ra implicit)
|
||||
* gte_mv_to_data_r (gte + mv + to + data + register)
|
||||
* gte_lw_v0_xy(base) (gte + lw + v0 + xy)
|
||||
* load_upper_i (load-upper + immediate, unique verb)
|
||||
*
|
||||
* Vendor mnemonics (sll, srl, sra, jr, j, jal, jalr) are NOT in this header.
|
||||
* They live in the opt-in `mips_vendor_sym.h` for users who prefer the textbook MIPS assembly mnemonics.
|
||||
* ============================================================================ */
|
||||
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "dsl.h"
|
||||
# include "gcc_asm.h"
|
||||
#endif
|
||||
|
||||
enum {
|
||||
MipsStackAlignment = 8,
|
||||
};
|
||||
|
||||
/* ============================================================================
|
||||
* REGISTER INTEGER IDS (preprocessor-visible)
|
||||
* ============================================================================
|
||||
* Every R_* enum below has a parallel R_*_Code `#define` so that the preprocessor can stringify the integer
|
||||
* (e.g. for asm clobber lists and register-variable declarations via `rgcc(R_X)`).
|
||||
* The enum value is bound to the `#define` so the two forms cannot drift apart.
|
||||
*
|
||||
* Only registers that get stringified need a `_Code` form; the rest are plain enum values.
|
||||
* If you need to add a new one, follow the pattern:
|
||||
* #define R_T7_Code 15
|
||||
* R_T7 = R_T7_Code, // in the enum
|
||||
*
|
||||
* User code should always reference the enum form (`R_T4`) at arithmetic sites and let
|
||||
* `rlit(R_T4_Code)` / `rgcc(R_T4)` handle the stringify cases — never write the bare number `12`.
|
||||
* ============================================================================ */
|
||||
#define R_0_Code 0
|
||||
#define R_AT_Code 1
|
||||
#define R_V0_Code 2
|
||||
#define R_V1_Code 3
|
||||
#define R_A0_Code 4
|
||||
#define R_A1_Code 5
|
||||
#define R_A2_Code 6
|
||||
#define R_A3_Code 7
|
||||
#define R_T0_Code 8
|
||||
#define R_T1_Code 9
|
||||
#define R_T2_Code 10
|
||||
#define R_T3_Code 11
|
||||
#define R_T4_Code 12
|
||||
#define R_T5_Code 13
|
||||
#define R_T6_Code 14
|
||||
#define R_T7_Code 15
|
||||
#define R_S0_Code 16
|
||||
#define R_S1_Code 17
|
||||
#define R_S2_Code 18
|
||||
#define R_S3_Code 19
|
||||
#define R_S4_Code 20
|
||||
#define R_S5_Code 21
|
||||
#define R_S6_Code 22
|
||||
#define R_S7_Code 23
|
||||
#define R_T8_Code 24
|
||||
#define R_T9_Code 25
|
||||
#define R_K0_Code 26
|
||||
#define R_K1_Code 27
|
||||
#define R_GP_Code 28
|
||||
#define R_SP_Code 29
|
||||
#define R_FP_Code 30
|
||||
#define R_RA_Code 31
|
||||
|
||||
enum {
|
||||
/* --- MIPS CPU Registers --- */
|
||||
|
||||
R_0 = R_0_Code, R_AT = R_AT_Code, R_V0 = R_V0_Code, R_V1 = R_V1_Code,
|
||||
R_A0 = R_A0_Code, R_A1 = R_A1_Code, R_A2 = R_A2_Code, R_A3 = R_A3_Code,
|
||||
R_T0 = R_T0_Code, R_T1 = R_T1_Code, R_T2 = R_T2_Code, R_T3 = R_T3_Code,
|
||||
R_T4 = R_T4_Code, R_T5 = R_T5_Code, R_T6 = R_T6_Code, R_T7 = R_T7_Code,
|
||||
R_S0 = R_S0_Code, R_S1 = R_S1_Code, R_S2 = R_S2_Code, R_S3 = R_S3_Code,
|
||||
R_S4 = R_S4_Code, R_S5 = R_S5_Code, R_S6 = R_S6_Code, R_S7 = R_S7_Code,
|
||||
R_T8 = R_T8_Code, R_T9 = R_T9_Code, R_K0 = R_K0_Code, R_K1 = R_K1_Code,
|
||||
R_GP = R_GP_Code, R_SP = R_SP_Code, R_FP = R_FP_Code, R_RA = R_RA_Code
|
||||
|
||||
/* Semantic Aliases for MIPS Registers (O32 ABI) */
|
||||
|
||||
, rdiscard = R_0 /* Hardwired to 0 */
|
||||
, rasm_tmp = R_AT /* Assembler temporary (destroyed by some assembler pseudoinstructions!) */
|
||||
, rret_0 = R_V0 /* Function return value */
|
||||
, rret_1 = R_V1 /* Second return value (e.g., 64-bit) */
|
||||
, rarg_0 = R_A0 /* First function argument */
|
||||
, rarg_1 = R_A1 /* Second function argument */
|
||||
, rarg_2 = R_A2 /* Third function argument */
|
||||
, rarg_3 = R_A3 /* Fourth function argument */
|
||||
, rtmp_0 = R_T0 /* Temporary (Caller saved) */
|
||||
, rtmp_1 = R_T1 /* Temporary (Caller saved) */
|
||||
, rtmp_2 = R_T2 /* Temporary (Caller saved) */
|
||||
, rtmp_3 = R_T3 /* Temporary (Caller saved) */
|
||||
, rtmp_4 = R_T4 /* Temporary (Caller saved) — common GTE base pointer */
|
||||
, rtmp_9 = R_T9 /* Temporary (Caller saved) — common GTE base pointer */
|
||||
, rstatic_0 = R_S0 /* Static (Callee saved, preserved across calls) */
|
||||
, rstatic_1 = R_S1
|
||||
, rstatic_2 = R_S2
|
||||
, rstatic_3 = R_S3
|
||||
, rstatic_4 = R_S4
|
||||
, rstatic_5 = R_S5
|
||||
, rstatic_6 = R_S6
|
||||
, rstatic_7 = R_S7
|
||||
, rsaved_0 = R_S0 /* Alias for rstatic_0 (alternate vocabulary) */
|
||||
, rstack_ptr = R_SP /* Stack Pointer */
|
||||
, rret_addr = R_RA /* Return Address (populated by JAL) */
|
||||
|
||||
/* --- MIPS CPU Opcodes (Bits 31-26) --- */
|
||||
|
||||
, op_special = 0x00 /* R-Type instructions (uses FUNCT field) */
|
||||
, op_bcond = 0x01 /* Branch on condition */
|
||||
, op_j = 0x02 /* Jump */
|
||||
, op_jal = 0x03 /* Jump and Link */
|
||||
, op_beq = 0x04 /* Branch on Equal */
|
||||
, op_bne = 0x05 /* Branch on Not Equal */
|
||||
, op_blez = 0x06 /* Branch on Less Than or Equal to Zero */
|
||||
, op_bgtz = 0x07 /* Branch on Greater Than Zero */
|
||||
, op_addi = 0x08 /* Add Immediate */
|
||||
, op_addiu = 0x09 /* Add Immediate Unsigned */
|
||||
, op_slti = 0x0A /* Set on Less Than Immediate */
|
||||
, op_sltiu = 0x0B /* Set on Less Than Immediate Unsigned */
|
||||
, op_andi = 0x0C /* AND Immediate */
|
||||
, op_ori = 0x0D /* OR Immediate */
|
||||
, op_xori = 0x0E /* XOR Immediate */
|
||||
, op_lui = 0x0F /* Load Upper Immediate */
|
||||
, op_cop0 = 0x10 /* Coprocessor 0 (System) */
|
||||
, op_cop1 = 0x11 /* Coprocessor 1 (Reserved, FP Unit, Omitted by Sony) */
|
||||
, op_cop2 = 0x12 /* Coprocessor 2 (GTE) */
|
||||
, op_cop3 = 0x13 /* Coprocessor 3 (Reserved, Unused)*/
|
||||
/* 14-1F: N/A */
|
||||
, op_lb = 0x20 /* Load Byte */
|
||||
, op_lh = 0x21 /* Load Halfword */
|
||||
, op_lwl = 0x22 /* Load Word (Left Bits) */
|
||||
, op_lw = 0x23 /* Load Word */
|
||||
, op_lbu = 0x24 /* Load Byte Unsigned */
|
||||
, op_lhu = 0x25 /* Load Halfword Unsigned */
|
||||
, op_lwr = 0x26 /* Load Word (Right Bits) */
|
||||
/* 27: N/A */
|
||||
, op_sb = 0x28 /* Store Byte */
|
||||
, op_sh = 0x29 /* Store Halfword */
|
||||
, op_swl = 0x2A /* Store Word (Left Bits) */
|
||||
, op_sw = 0x2B /* Store Word */
|
||||
/* 2C-2D: N/A */
|
||||
, op_swr = 0x2E /* Store Word (Right Bits) */
|
||||
/* 2F: N/A */
|
||||
// , op_lwc0
|
||||
|
||||
// , op_load_addr = op_la
|
||||
// , op_load_imm = op_li
|
||||
, op_jump = op_j
|
||||
, op_jump_nlink = op_jal
|
||||
|
||||
/* --- MIPS CPU Function Codes (Bits 5-0, used when OP == MIPS_OP_SPECIAL) --- */
|
||||
|
||||
, fc_sll = 0x00 /* Shift Word Left Logical */
|
||||
, fc_srl = 0x02 /* Shift Word Right Logical */
|
||||
, fc_sra = 0x03 /* Shift Word Right Arithmetic */
|
||||
, fc_sllv = 0x04 /* Shift Word Left Logical Variable */
|
||||
, fc_srlv = 0x06 /* Shift Word Right Logical Variable */
|
||||
, fc_srav = 0x07 /* Shift Word Right Arithmetic Variable */
|
||||
, fc_jr = 0x08 /* Jump Register */
|
||||
, fc_jalr = 0x09 /* Jump and Link Register */
|
||||
, fc_syscall = 0x0C /* System Call */
|
||||
, fc_break = 0x0D /* Breakpoint */
|
||||
, fc_mfhi = 0x10 /* Move From HI */
|
||||
, fc_mthi = 0x11 /* Move To HI */
|
||||
, fc_mflo = 0x12 /* Move From LO */
|
||||
, fc_mtlo = 0x13 /* Move To LO */
|
||||
, fc_mult = 0x18 /* Multiply Word */
|
||||
, fc_multu = 0x19 /* Multiply Unsigned Word */
|
||||
, fc_div = 0x1A /* Divide Word */
|
||||
, fc_divu = 0x1B /* Divide Unsigned Word */
|
||||
, fc_add = 0x20 /* Add Word */
|
||||
, fc_addu = 0x21 /* Add Unsigned Word */
|
||||
, fc_sub = 0x22 /* Subtract Word */
|
||||
, fc_subu = 0x23 /* Subtract Unsigned Word */
|
||||
, fc_and = 0x24 /* AND */
|
||||
, fc_or = 0x25 /* OR */
|
||||
, fc_xor = 0x26 /* XOR */
|
||||
, fc_nor = 0x27 /* NOR */
|
||||
, fc_slt = 0x2A /* Set on Less Than */
|
||||
, fc_sltu = 0x2B /* Set on Less Than Unsigned */
|
||||
|
||||
, fc_jump_reg = fc_jr
|
||||
|
||||
/* --- Coprocessor 0 (System Control & Exceptions) --- */
|
||||
|
||||
, cop_mf = 0x00 /* Move From Coprocessor */
|
||||
, cop_mt = 0x04 /* Move To Coprocessor */
|
||||
};
|
||||
|
||||
|
||||
// Bitfield Packets (Encoders)
|
||||
|
||||
enum { _BitOffsets = 0
|
||||
/* Bit Offsets for MIPS Instruction Fields */
|
||||
|
||||
, OPCODE_SHIFT = 26
|
||||
, RS_SHIFT = 21
|
||||
, RT_SHIFT = 16
|
||||
, RD_SHIFT = 11
|
||||
, SHAMT_SHIFT = 6 /* Shift Amount */
|
||||
, FC_SHIFT = 0
|
||||
|
||||
/* Bit Masks to prevent overflow into adjacent fields */
|
||||
|
||||
, OPCODE_MASK = 0x3F
|
||||
, REG_MASK = 0x1F
|
||||
, SHAMT_MASK = 0x1F /* Shift Amount */
|
||||
, FC_MASK = 0x3F
|
||||
, IMM_MASK = 0xFFFF
|
||||
};
|
||||
|
||||
#define enc_op(op) (((op) & OPCODE_MASK) << OPCODE_SHIFT)
|
||||
#define enc_rs(rs) (((rs) & REG_MASK) << RS_SHIFT)
|
||||
#define enc_rt(rt) (((rt) & REG_MASK) << RT_SHIFT)
|
||||
#define enc_rd(rd) (((rd) & REG_MASK) << RD_SHIFT)
|
||||
#define enc_shamt(shamt) (((shamt) & SHAMT_MASK) << SHAMT_SHIFT)
|
||||
#define enc_fc(fc) (((fc) & FC_MASK) << FC_SHIFT)
|
||||
#define enc_imm(imm) (((imm) & IMM_MASK))
|
||||
|
||||
/* MIPS R-Type Instruction Format (Register-to-Register) */
|
||||
#define enc_r(op, rs, rt, rd, shamt, fc) (enc_op(op) | enc_rs(rs) | enc_rt(rt) | enc_rd(rd) | enc_shamt(shamt) | enc_fc(fc))
|
||||
/* MIPS I-Type Instruction Format (Immediate/Constant) */
|
||||
#define enc_i(op, rs, rt, imm) (enc_op(op) | enc_rs(rs) | enc_rt(rt) | enc_imm(imm))
|
||||
|
||||
/* COP0 (System) Transfer Format: mtc0 rt, rd or mfc0 rt, rd
|
||||
* `sub` is the COP0 sub-opcode (cop_mf=0 or cop_mt=4), placed in rs slot.
|
||||
* `rt` is the GPR operand (in rt slot).
|
||||
* `rd` is the COP0 register index (in rd slot at bits 15..11). */
|
||||
#define enc_cop0_tx(sub, rt, rd) enc_i(op_cop0, (sub), (rt), ((rd) << 11))
|
||||
|
||||
/* Semantic aliases for COP0 transfer. `sys_` is the namespace marker
|
||||
* for system-control instructions (analogous to `gte_` for COP2).
|
||||
* sys_mov_to_cop0 rt, rd → mtc0 rt, rd
|
||||
* sys_mov_from_cop0 rt, rd → mfc0 rt, rd
|
||||
* sys_rfe → rfe (return from exception) */
|
||||
#define sys_mov_to_cop0(rt, rd) enc_cop0_tx(cop_mt, (rt), (rd))
|
||||
#define sys_mov_from_cop0(rt, rd) enc_cop0_tx(cop_mf, (rt), (rd))
|
||||
#define sys_rfe() enc_rfe()
|
||||
|
||||
/* COP0 Return From Exception (rfe) */
|
||||
#define enc_rfe() 0x42000010
|
||||
|
||||
/* --- Semantic Encoders (MIPS mnemonics) ---
|
||||
* Argument order matches the MIPS assembly syntax:
|
||||
* dest-first, then source operands, then immediate last.
|
||||
*
|
||||
* load_word(rt, base, off) → lw rt, off(base)
|
||||
* store_word(rt, base, off) → sw rt, off(base)
|
||||
* add_ui(rt, rs, imm) → addiu rt, rs, imm
|
||||
* shift_lleft(rd, rt, shamt) → sll rd, rt, shamt
|
||||
* shift_lright(rd, rt, shamt) → srl rd, rt, shamt
|
||||
* shift_aright(rd, rt, shamt) → sra rd, rt, shamt
|
||||
* jump_reg(rs) → jr rs
|
||||
* jump_link(rs, rd) → jalr rs (link in rd, default $ra)
|
||||
* nop → sll $0, $0, 0
|
||||
*/
|
||||
#define load_word(rt, base, off) enc_i(op_lw, (base), (rt), (off))
|
||||
#define load_byte(rt, base, off) enc_i(op_lb, (base), (rt), (off))
|
||||
#define load_half(rt, base, off) enc_i(op_lh, (base), (rt), (off))
|
||||
#define load_byte_u(rt, base, off) enc_i(op_lbu, (base), (rt), (off))
|
||||
#define load_half_u(rt, base, off) enc_i(op_lhu, (base), (rt), (off))
|
||||
#define store_word(rt, base, off) enc_i(op_sw, (base), (rt), (off))
|
||||
#define add_ui(rt, rs, imm) enc_i(op_addiu, (rs), (rt), (imm))
|
||||
#define and_i(rt, rs, imm) enc_i(op_andi, (rs), (rt), (imm))
|
||||
// #define and_si and_i
|
||||
#define or_i(rt, rs, imm) enc_i(op_ori, (rs), (rt), (imm))
|
||||
#define xor_i(rt, rs, imm) enc_i(op_xori, (rs), (rt), (imm))
|
||||
#define load_upper_i(rt, imm) enc_i(op_lui, R_0, (rt), (imm))
|
||||
|
||||
#define load_u1 load_byte_u
|
||||
#define load_u2 load_half_u
|
||||
#define load_u4 load_word
|
||||
|
||||
// Ergonomic add to the same register.
|
||||
#define or_i_self(rt_rs, imm) enc_i(op_ori, (rt_rs), (rt_rs), (imm))
|
||||
#define add_ui_self(rt_rs, imm) enc_i(op_addiu, (rt_rs), (rt_rs), (imm))
|
||||
|
||||
/* Logic Opcodes */
|
||||
|
||||
#define and_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_and)
|
||||
#define or_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_or)
|
||||
#define xor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_xor)
|
||||
#define nor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_nor)
|
||||
|
||||
#define or_u_self(rd_rs, rt) enc_r(op_special, (rd_rs), (rt), (rd_rs), 0, fc_or)
|
||||
|
||||
/* Shift family (R-type). shift_lleft/lright/aright: `sll/srl/sra rd, rt, shamt` */
|
||||
#define shift_lleft(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_sll)
|
||||
#define shift_lright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_srl)
|
||||
#define shift_aright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_sra)
|
||||
|
||||
#define shift_lleft_self(rd_rt, shamt) enc_r(op_special, R_0, (rd_rt), (rd_rt), (shamt), fc_sll)
|
||||
|
||||
#define mask_upper(rd, rt, shamt) shift_lleft(rd, rt, shamt), shift_lright(rd, rt, shamt)
|
||||
|
||||
/* jr rs — jump to address in rs. */
|
||||
#define jump_reg(rs) enc_r(op_special, (rs), R_0, R_0, 0, fc_jr)
|
||||
|
||||
/* jalr rs, rd — link in rd (default $ra) and jump to address in rs.
|
||||
* Layout: [op_special][rs:5][rt=0:5][rd:5][shamt=0:5][fc_jalr=0x09] */
|
||||
#define jump_link(rs, rd) enc_r(op_special, (rs), R_0, (rd), 0, fc_jalr)
|
||||
|
||||
/* call_reg rs — jump-and-link to register-held address; link in $ra. */
|
||||
#define call_reg(rs) jump_link((rs), R_RA)
|
||||
|
||||
/* j target — absolute jump within the current 256MB region. */
|
||||
#define jump(off) enc_i(op_j, R_0, R_0, (off))
|
||||
|
||||
/* call_addr off — jump-and-link to immediate address. */
|
||||
#define call_addr(off) enc_i(op_jal, R_0, R_0, (off))
|
||||
|
||||
/* --- Store family (mirrors the load family) --- */
|
||||
#define store_byte(rt, base, off) enc_i(op_sb, (base), (rt), (off))
|
||||
#define store_half(rt, base, off) enc_i(op_sh, (base), (rt), (off))
|
||||
/* store_word already exists above */
|
||||
|
||||
/* --- Arithmetic R-type (signed/unsigned split: _s traps, _u doesn't) ---
|
||||
* add_s rd, rs, rt → add rd, rs, rt (overflow traps)
|
||||
* add_u rd, rs, rt → addu rd, rs, rt (overflow silent)
|
||||
* sub_s / sub_u → sub / subu
|
||||
* mult_s / mult_u → mult / multu (writes HI/LO; result in LO)
|
||||
* div_s / div_u → div / divu (LO = quot, HI = rem)
|
||||
*
|
||||
* NOTE: dsl.h defines `add_s`/`sub_s`/`mut_s`/`gt_s`/etc. as _Generic-based signed integer-arithmetic helpers for U1/U2/U4.
|
||||
* Those live in a different conceptual layer (generic arithmetic on DSL types) and would collide with the instruction encoders here.
|
||||
* The `#undef` below lets the gas-style names below win; if a file needs both, the dsl.h versions can be reached via their long forms
|
||||
* (e.g. `def_signed_op`-style or the underlying `add_s1/s2/s4`). */
|
||||
#undef add_s
|
||||
#undef sub_s
|
||||
#define add_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_add)
|
||||
#define add_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_addu)
|
||||
#define sub_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_sub)
|
||||
#define sub_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_subu)
|
||||
#define mult_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_mult)
|
||||
#define mult_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_multu)
|
||||
#define div_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_div)
|
||||
#define div_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_divu)
|
||||
|
||||
#define add_u_self(rd_rs, rt) add_u(rd_rs, rd_rs, rt)
|
||||
|
||||
/* --- Arithmetic I-type (immediate) --- */
|
||||
#define add_si(rt, rs, imm) enc_i(op_addi, (rs), (rt), (imm))
|
||||
/* add_ui already exists above as add_ui */
|
||||
|
||||
/* --- Set on less than (R-type and I-type) --- */
|
||||
#define set_lt_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_slt)
|
||||
#define set_lt_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_sltu)
|
||||
#define set_lt_si(rt, rs, imm) enc_i(op_slti, (rs), (rt), (imm))
|
||||
#define set_lt_ui(rt, rs, imm) enc_i(op_sltiu, (rs), (rt), (imm))
|
||||
|
||||
/* --- Move from/to HI/LO (mult/div results) --- */
|
||||
#define mov_from_high(rd) enc_r(op_special, R_0, R_0, (rd), 0, fc_mfhi)
|
||||
#define mov_from_low(rd) enc_r(op_special, R_0, R_0, (rd), 0, fc_mflo)
|
||||
#define mov_to_high(rs) enc_r(op_special, (rs), R_0, R_0, 0, fc_mthi)
|
||||
#define mov_to_low(rs) enc_r(op_special, (rs), R_0, R_0, 0, fc_mtlo)
|
||||
|
||||
/* --- Atomic branches (no pseudos like bgt/bge; compose with slt_* + branch_ne) ---
|
||||
* branch_equal rs, rt, off → beq rs, rt, off
|
||||
* branch_ne rs, rt, off → bne rs, rt, off
|
||||
* branch_lt_zero rs, off → bltz rs, off
|
||||
* branch_gt_zero rs, off → bgtz rs, off
|
||||
* branch_le_zero rs, off → blez rs, off
|
||||
* branch_ge_zero rs, off → bgez rs, off
|
||||
* (For `bgez`, the opcode is `op_bcond` with rt=1 to invert the bltz condition.) */
|
||||
|
||||
#define branch_equal(rs, rt, off) enc_i(op_beq, (rs), (rt), (off))
|
||||
#define branch_ne(rs, rt, off) enc_i(op_bne, (rs), (rt), (off))
|
||||
#define branch_lt_zero(rs, off) enc_i(op_bcond, (rs), R_0, (off)) /* bltz is bcond with rt=0 */
|
||||
#define branch_ge_zero(rs, off) enc_i(op_bcond, (rs), 1, (off)) /* bgez is bcond with rt=1 */
|
||||
#define branch_le_zero(rs, off) enc_i(op_blez, (rs), R_0, (off)) /* blez has its own opcode, rt=0 */
|
||||
#define branch_gt_zero(rs, off) enc_i(op_bgtz, (rs), R_0, (off)) /* bgtz has its own opcode, rt=0 */
|
||||
|
||||
/* --- System (kernel) instructions --- */
|
||||
#define syscall() enc_r(op_special, R_0, R_0, R_0, 0, fc_syscall)
|
||||
#define breakpoint() enc_r(op_special, R_0, R_0, R_0, 0, fc_break)
|
||||
|
||||
/* --- Shift-amount alias (matches the gas convention `\p3 = shamt`) --- */
|
||||
#define shift_amount(rd, rt, n) shift_lleft(rd, rt, n)
|
||||
|
||||
/* nop — sll $0, $0, 0 */
|
||||
#define nop shift_lleft(rdiscard, rdiscard, 0)
|
||||
#define nop2 nop, nop
|
||||
|
||||
#define load_imm_1w(rt, imm) add_ui((rt), R_0, (imm))
|
||||
#define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm))
|
||||
|
||||
/* load_imm_2w — unconditional 2-word `li` form: `lui` + (ori | addi).
|
||||
*
|
||||
* Granular companion to `load_imm`: skips the compile-time range checks and always emits 2 .words. Use this when:
|
||||
* - you know `imm` is > 0xFFFF (otherwise you're wasting a word), OR
|
||||
* - `imm` is not a compile-time constant and you want predictable
|
||||
* 2-word emission without the `__builtin_constant_p` branches.
|
||||
*
|
||||
* The lo16 strategy is still chosen at expansion time on the lo half:
|
||||
* lo16 in 0x0000..0x7FFF → addi (sign-ext is harmless, the lui already cleared bits 15..0)
|
||||
* lo16 in 0x8000..0xFFFF → ori (zero-extends to preserve the intended bit pattern)
|
||||
*
|
||||
* For situations where you need to bypass even this choice
|
||||
* (e.g. to force a specific encoding for a known discontiguous high/low pair),
|
||||
* see `load_imm_2w_ori_forced` and `load_imm_2w_addi_forced` below.
|
||||
* Statement-level (not expression-level): emits its own `asm volatile(...)`.
|
||||
*/
|
||||
#define load_imm_2w(rt, imm) do { \
|
||||
if (u4_low(imm) <= 0x7FFFU) { \
|
||||
asm volatile( \
|
||||
asm_words(load_ui((rt), u4_hi(imm), \
|
||||
add_si((rt), (rt), (S2)C_(U2,u4_lo(imm))) \
|
||||
asm_clobber: rlit(R_AT), clb_mem_drain \
|
||||
); \
|
||||
} \
|
||||
else { \
|
||||
asm volatile(asm_words( \
|
||||
load_ui((rt), u4_hi(imm)), \
|
||||
or_i((rt), (rt), C_(U2,u4_lo(imm)) \
|
||||
asm_clobber: rlit(R_AT), clb_mem_drain \
|
||||
); \
|
||||
} \
|
||||
} while (0)
|
||||
|
||||
/* load_imm_2w_ori_forced — force the `lui` + `ori` form regardless of lo16 sign.
|
||||
* Use when you specifically need zero-extension in the lo half. */
|
||||
#define load_imm_2w_ori_forced(rt, imm) do { \
|
||||
asm volatile( \
|
||||
asm_words(load_ui((rt), u4_lo(imm)), \
|
||||
or_i((rt), (rt), C_(U2,u4_hi(imm))) ) \
|
||||
asm_clobber: rlit(R_AT), clb_mem_drain \
|
||||
); \
|
||||
} while (0)
|
||||
|
||||
/* load_imm_2w_addi_forced — force the `lui` + `addi` form regardless of lo16 sign.
|
||||
* Use when you know sign-extension is fine (e.g. lo16 is treated as signed downstream)
|
||||
* and you want a smaller effective instruction (the assembler/MIPS hardware will sign-extend the imm16). */
|
||||
#define load_imm_2w_addi_forced(rt, imm) do { \
|
||||
/*U4 _li2a_imm_ = (U4)(imm);*/ \
|
||||
asm volatile(asm_words( \
|
||||
lui_op((rt), u4_lo(imm)), \
|
||||
add_si((rt), (rt), (S2)C_(U2,u4_hi(imm))) ) \
|
||||
asm_clobber: rlit(R_AT), clb_mem_drain \
|
||||
); \
|
||||
} while (0)
|
||||
|
||||
/* load_imm rt, imm — true `li` semantics (assembler `li` pseudo)
|
||||
*
|
||||
* Dispatches at compile time on the immediate's range, picking the smallest single-instruction form when possible:
|
||||
* imm in 0 .. 0x7FFF → addi rt, $0, imm (1 word)
|
||||
* imm in 0x8000 .. 0xFFFF → ori rt, $0, imm (1 word; sign-bit must be zeroed)
|
||||
* imm in 0x10000 .. 0xFFFFFFFF → lui + (ori | addi) (2 words)
|
||||
*
|
||||
* Statement-level (not expression-level): the macro emits its own `asm volatile(...)` block with 1 or 2 .word constants.
|
||||
* Callers can group multiple `load_imm` calls in a single volatile by using the lower-level encoders directly:
|
||||
* load_imm(R_T4, 0x12345678); // emits 2 .words
|
||||
*
|
||||
* Falls back to a 2-word form if `imm` is not a compile-time constant, but that path is unusual
|
||||
* (load_imm is most useful with literal addresses and magic numbers). */
|
||||
#define load_imm(rt, imm) do { \
|
||||
if (cexpr_(imm) && ((imm) <= 0x7FFFU)) { \
|
||||
/* Small positive: addi rt, $0, imm */ \
|
||||
asm volatile( \
|
||||
asm_words(add_si((rt), R_0, (imm))) \
|
||||
asm_clobber: rlit(R_AT), clb_mem_drain \
|
||||
); \
|
||||
} \
|
||||
else if (cexpr_(imm) && ((U4)(imm) <= 0xFFFFU)) { \
|
||||
/* 0x8000..0xFFFF: ori rt, $0, imm (zero-extends) */ \
|
||||
asm volatile( \
|
||||
asm_words(or_i((rt), R_0, (imm))) \
|
||||
asm_clobber: rlit(R_AT), clb_mem_drain \
|
||||
); \
|
||||
} \
|
||||
else \
|
||||
{ \
|
||||
/* > 16 bits: lui + (ori | addi). \
|
||||
* If lo16 is in [0, 0x7FFF] use addi (sign-ext is harmless \
|
||||
* since the high half cleared bits 15..0). Otherwise ori. */ \
|
||||
if (u4_lo(imm) <= 0x7FFFU) { \
|
||||
asm volatile(asm_words( \
|
||||
load_ui((rt), u4_hi(imm)), \
|
||||
add_si((rt), (rt), (S2)C_(U2,u4_lo(imm))) \
|
||||
asm_clobber: rlit(R_AT), clb_mem_drain \
|
||||
); \
|
||||
} \
|
||||
else { \
|
||||
asm volatile(asm_words( \
|
||||
load_ui((rt), u4_hi(imm)), \
|
||||
or_i((rt), (rt), C_(U2,u4_lo(imm)) \
|
||||
asm_clobber: rlit(R_AT), clb_mem_drain \
|
||||
); \
|
||||
} \
|
||||
} \
|
||||
} while (0 )
|
||||
|
||||
|
||||
/* Standard clobber list for pure-MIPS asm volatile blocks: caller-saved
|
||||
* GPRs that the kernel treats as volatile (v0/v1/t0/t1/ra) plus the "memory" barrier.
|
||||
* The register ids are passed through `rlit` so the R_*_Code `#define`s are stringified into "$N" at expansion time. */
|
||||
#define clbr_volatile_gprs rlit(R_V0), rlit(R_T0), rlit(R_T1), rlit(R_RA), clb_mem_drain
|
||||
|
||||
#define asm_mips_flush_icache() asm volatile( asm_words( \
|
||||
add_ui(rstack_ptr, rstack_ptr, -MipsStackAlignment) \
|
||||
, store_word(rret_addr, rstack_ptr, 4) \
|
||||
, add_ui(rret_0, rdiscard, bios_flushcache) \
|
||||
, add_ui(rtmp_0, rdiscard, bios_table_addr) \
|
||||
, jump_link(rtmp_0, rret_addr) \
|
||||
, nop \
|
||||
, load_word(rret_addr, rstack_ptr, 4) \
|
||||
, jump_reg(rret_addr) \
|
||||
, add_ui(rstack_ptr, rstack_ptr, MipsStackAlignment) \
|
||||
) asm_clobber: clbr_volatile_gprs )
|
||||
@@ -0,0 +1,44 @@
|
||||
/* ============================================================================
|
||||
* duffle DSL — MIPS Vendor Mnemonics (opt-in)
|
||||
* ============================================================================
|
||||
*
|
||||
* Provides the textbook MIPS assembly mnemonics as thin aliases to the duffle macros in mips.h.
|
||||
* The duffle names are primary; this header is for users who prefer the textbook mnemonics.
|
||||
*
|
||||
* USAGE: #include "duffle/mips_vendor_sym.h" // after mips.h
|
||||
*
|
||||
* Mapping (vendor -> duffle):
|
||||
* Shift family:
|
||||
* sll -> shift_lleft (shift left logical)
|
||||
* srl -> shift_lright (shift right logical)
|
||||
* sra -> shift_aright (shift right arithmetic)
|
||||
* (no sllv/srlv/srav; the shift macros take a literal shamt)
|
||||
*
|
||||
* Jump family (1-arg / implicit-rd forms):
|
||||
* jr -> jump_reg (jump register)
|
||||
* j -> jump (jump to immediate address)
|
||||
* jal -> call_addr (jump-and-link to immediate address)
|
||||
* jalr -> call_reg (jump-and-link to register, default $ra)
|
||||
* (for the 2-arg `jalr rs, rd`, use `jump_link(rs, rd)` directly)
|
||||
* ============================================================================ */
|
||||
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "mips.h"
|
||||
#endif
|
||||
|
||||
#ifndef DUFFLE_MIPS_VENDOR_SYM_H
|
||||
#define DUFFLE_MIPS_VENDOR_SYM_H
|
||||
|
||||
/* Shift family */
|
||||
#define sll shift_lleft
|
||||
#define srl shift_lright
|
||||
#define sra shift_aright
|
||||
|
||||
/* Jump family (1-arg / implicit-$ra forms) */
|
||||
#define jr jump_reg
|
||||
#define j jump
|
||||
#define jal call_addr
|
||||
#define jalr call_reg
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,56 @@
|
||||
// word_count.metadata.h
|
||||
// Single source of truth for instruction-word counts.
|
||||
// Used by C (to define compile-time constants) AND Python (to count positions).
|
||||
//
|
||||
// Format: WORD_COUNT(MACRO_NAME, COUNT)
|
||||
// One line per macro that appears in your atom sources.
|
||||
//
|
||||
// This file is encoding-macros-only.
|
||||
// The auto-generated component macros (mac_X) live in duffle/gen/<dir>.macs.h (included separately by the unity build).
|
||||
// The unity build should include THIS file and the .macs.h file in the same TU, with both wrapped
|
||||
// (or the include guard order handled) to avoid WORD_COUNT redeclaration.
|
||||
//
|
||||
// To regenerate: hand-count the instructions in each macro definition.
|
||||
// (You'll only need to do this once per macro — they don't change often.)
|
||||
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
||||
|
||||
WORD_COUNT(nop, 1)
|
||||
WORD_COUNT(load_upper_i, 1)
|
||||
WORD_COUNT(jump_reg, 1)
|
||||
WORD_COUNT(jump_link, 1)
|
||||
WORD_COUNT(call_reg, 1)
|
||||
WORD_COUNT(call_addr, 1)
|
||||
WORD_COUNT(branch_le_zero, 1)
|
||||
WORD_COUNT(branch_equal, 1)
|
||||
WORD_COUNT(add_ui, 1)
|
||||
WORD_COUNT(set_lt_u, 1)
|
||||
WORD_COUNT(set_lt_s, 1)
|
||||
WORD_COUNT(set_lt_si, 1)
|
||||
WORD_COUNT(set_lt_ui, 1)
|
||||
WORD_COUNT(load_word, 1)
|
||||
WORD_COUNT(load_half_u, 1)
|
||||
WORD_COUNT(store_word, 1)
|
||||
WORD_COUNT(add_ui_self, 1)
|
||||
WORD_COUNT(add_u_self, 1)
|
||||
WORD_COUNT(add_u, 1)
|
||||
WORD_COUNT(or_i, 1)
|
||||
WORD_COUNT(or_i_self, 1)
|
||||
WORD_COUNT(or_u, 1)
|
||||
WORD_COUNT(or_u_self, 1)
|
||||
WORD_COUNT(shift_lleft, 1)
|
||||
WORD_COUNT(shift_lleft_self, 1)
|
||||
WORD_COUNT(shift_lright, 1)
|
||||
WORD_COUNT(shift_aright, 1)
|
||||
WORD_COUNT(mask_upper, 2)
|
||||
WORD_COUNT(gte_mv_from_data_r, 1)
|
||||
WORD_COUNT(gte_mv_from_ctrl_r, 1)
|
||||
WORD_COUNT(gte_mv_to_data_r, 1)
|
||||
WORD_COUNT(gte_mv_to_ctrl_r, 1)
|
||||
WORD_COUNT(gte_sw, 1)
|
||||
WORD_COUNT(gte_cmdw_rtpt, 1)
|
||||
WORD_COUNT(gte_cmdw_nclip, 1)
|
||||
WORD_COUNT(gte_avg_sort_z3, 1)
|
||||
WORD_COUNT(sub_u, 1)
|
||||
WORD_COUNT(nop2, 2)
|
||||
|
||||
#undef WORD_COUNT
|
||||
@@ -17,19 +17,19 @@ enum {
|
||||
};
|
||||
|
||||
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
||||
typedef def_farray(OrderingTable_Buffer, 2);
|
||||
typedef Array_(OrderingTable_Buffer, 2);
|
||||
|
||||
typedef B1 PrimitiveBuffer[PrimitiveBuff_Len];
|
||||
typedef def_farray(PrimitiveBuffer, 2);
|
||||
typedef def_struct(PrimitiveArena) {
|
||||
typedef Array_(PrimitiveBuffer, 2);
|
||||
typedef Struct_(PrimitiveArena) {
|
||||
A2_PrimitiveBuffer buf;
|
||||
U4 used;
|
||||
};
|
||||
|
||||
#define Cube_num_verts 8
|
||||
typedef def_farray(V3_S2, Cube_num_verts);
|
||||
typedef Array_(V3_S2, Cube_num_verts);
|
||||
#define Cube_num_faces 6
|
||||
typedef def_farray(V4_S2, Cube_num_faces);
|
||||
typedef Array_(V4_S2, Cube_num_faces);
|
||||
void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
||||
memory_copy(verts, & (A8_V3_S2) {
|
||||
{ -128, -128, -128 },
|
||||
@@ -40,7 +40,7 @@ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
||||
{ 128, 128, -128 },
|
||||
{ 128, 128, 128 },
|
||||
{ -128, 128, 128 }
|
||||
}, size_of(A8_V3_S2) );
|
||||
}, S_(A8_V3_S2) );
|
||||
memory_copy(faces, & (A6_V4_S2) {
|
||||
{ 3, 2, 0, 1 },
|
||||
{ 0, 1, 4, 5 },
|
||||
@@ -48,10 +48,10 @@ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
||||
{ 1, 2, 5, 6 },
|
||||
{ 2, 3, 6, 7 },
|
||||
{ 3, 0, 7, 4 },
|
||||
}, size_of(A6_V4_S2) );
|
||||
}, S_(A6_V4_S2) );
|
||||
return;
|
||||
}
|
||||
typedef def_struct(Ent_Cube) {
|
||||
typedef Struct_(Ent_Cube) {
|
||||
V3_S4 accel;
|
||||
V3_S4 vel;
|
||||
V3_S4 pos;
|
||||
@@ -62,22 +62,22 @@ typedef def_struct(Ent_Cube) {
|
||||
};
|
||||
|
||||
#define Floor_num_verts 4
|
||||
typedef def_farray(V3_S2, Floor_num_verts);
|
||||
typedef Array_(V3_S2, Floor_num_verts);
|
||||
#define Floor_num_faces 2
|
||||
typedef def_farray(V3_S2, Floor_num_faces);
|
||||
typedef Array_(V3_S2, Floor_num_faces);
|
||||
void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
|
||||
memory_copy(verts, &(A4_V3_S2) {
|
||||
{ -900, 0, -900 },
|
||||
{ -900, 0, 900 },
|
||||
{ 900, 0, -900 },
|
||||
{ 900, 0, 900 },
|
||||
}, size_of(A8_V3_S2));
|
||||
}, S_(A8_V3_S2));
|
||||
memory_copy(faces, & (A2_V3_S2) {
|
||||
{ 0, 1, 2 },
|
||||
{ 1, 3, 2 },
|
||||
}, size_of(A2_V3_S2));
|
||||
}, S_(A2_V3_S2));
|
||||
};
|
||||
typedef def_struct(Ent_Floor) {
|
||||
typedef Struct_(Ent_Floor) {
|
||||
V3_S4 accel;
|
||||
V3_S4 pos;
|
||||
V3_S4 scale;
|
||||
@@ -86,7 +86,7 @@ typedef def_struct(Ent_Floor) {
|
||||
A2_V3_S2 faces;
|
||||
};
|
||||
|
||||
typedef def_struct(SMemory) {
|
||||
typedef Struct_(SMemory) {
|
||||
DoubleBuffer screen_buf;
|
||||
A2_OrderingTable_Buffer ordering_tbl;
|
||||
PrimitiveArena primitives;
|
||||
@@ -108,7 +108,7 @@ B1* prim__alloc(U4 type_width, Str8 type_name) {
|
||||
pa->used += type_width;
|
||||
return next;
|
||||
}
|
||||
#define prim_alloc(type) (type*)prim__alloc(size_of(type), txt( stringify(type)))
|
||||
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
|
||||
|
||||
void gp_screen_init_c11(DoubleBuffer* screen_buf, S2* active_buf_id)
|
||||
{
|
||||
|
||||
@@ -5,8 +5,8 @@
|
||||
# include "duffle/gp.h"
|
||||
#endif
|
||||
|
||||
typedef def_struct(DrawEnv_Packed) { U4 tag; U4 code[15]; };
|
||||
typedef def_struct(DrawEnv) {
|
||||
typedef Struct_(DrawEnv_Packed) { U4 tag; U4 code[15]; };
|
||||
typedef Struct_(DrawEnv) {
|
||||
Rect_S2 clip_area;
|
||||
A2_S2 drawing_offset;
|
||||
Rect_S2 texture_window;
|
||||
@@ -17,7 +17,7 @@ typedef def_struct(DrawEnv) {
|
||||
RGB8 initial_bg_color;
|
||||
DrawEnv_Packed dr_env; // reserved
|
||||
};
|
||||
typedef def_struct(DisplayEnv) {
|
||||
typedef Struct_(DisplayEnv) {
|
||||
Rect_S2 display_area;
|
||||
Rect_S2 screen;
|
||||
B1 vinterlace;
|
||||
@@ -25,9 +25,9 @@ typedef def_struct(DisplayEnv) {
|
||||
B1 pad0;
|
||||
B1 pad1;
|
||||
};
|
||||
typedef def_farray(DrawEnv, 2);
|
||||
typedef def_farray(DisplayEnv, 2);
|
||||
typedef def_struct(DoubleBuffer) {
|
||||
typedef Array_(DrawEnv, 2);
|
||||
typedef Array_(DisplayEnv, 2);
|
||||
typedef Struct_(DoubleBuffer) {
|
||||
A2_DrawEnv draw;
|
||||
A2_DisplayEnv display;
|
||||
};
|
||||
@@ -58,7 +58,7 @@ U4 vsync(U4 mode) __asm__("VSync");
|
||||
|
||||
void draw_orderingtbl(U4* buf) __asm__("DrawOTag");
|
||||
|
||||
typedef def_struct(PolyTag) {
|
||||
typedef Struct_(PolyTag) {
|
||||
U4 addr: 24;
|
||||
U4 len: 8;
|
||||
RGB8 color;
|
||||
@@ -106,7 +106,7 @@ typedef def_struct(PolyTag) {
|
||||
// #define setLineF4(p) set_len(p, 6), set_code(p, 0x4c),(p)->pad = 0x55555555
|
||||
// #define setLineG4(p) set_len(p, 9), set_code(p, 0x5c),(p)->pad = 0x55555555, (p)->p2 = 0, (p)->p3 = 0
|
||||
|
||||
typedef def_struct(Poly_F3) {
|
||||
typedef Struct_(Poly_F3) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
@@ -120,14 +120,14 @@ typedef def_struct(Poly_F3) {
|
||||
};
|
||||
};
|
||||
|
||||
typedef def_struct(Poly_G3) {
|
||||
typedef Struct_(Poly_G3) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2;
|
||||
};
|
||||
|
||||
typedef def_struct(Poly_F4) {
|
||||
typedef Struct_(Poly_F4) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
@@ -142,7 +142,7 @@ typedef def_struct(Poly_F4) {
|
||||
};
|
||||
};
|
||||
|
||||
typedef def_struct(Poly_G4) {
|
||||
typedef Struct_(Poly_G4) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
@@ -150,7 +150,7 @@ typedef def_struct(Poly_G4) {
|
||||
V2_S2 p3;
|
||||
};
|
||||
|
||||
typedef def_struct(Tile) {
|
||||
typedef Struct_(Tile) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
@@ -169,7 +169,7 @@ M3_S2* m3s2_scale (M3_S2* mat, V3_S4* vec) __asm__("ScaleMatrix");
|
||||
// Rotation, Translation, Perspective
|
||||
|
||||
S4 rtp_v3s2_raw(V3_S2* vec, S4* xy, S4* pp, S4* flag) __asm__("RotTransPers");
|
||||
FI_ S4 rtp_v3s2(V3_S2* vec, V2_S2* xy, A2_S2* pp, S4* flag) { return rtp_v3s2_raw(vec, cast(S4*R_, & xy->x), cast(S4*R_, pp), r_(flag)); }
|
||||
FI_ S4 rtp_v3s2(V3_S2* vec, V2_S2* xy, A2_S2* pp, S4* flag) { return rtp_v3s2_raw(vec, C_(S4*R_, & xy->x), C_(S4*R_, pp), r_(flag)); }
|
||||
|
||||
S4 rtp_avg_nclip_a3_v3s2_raw(V3_S2* v0, V3_S2* v1, V3_S2* v2, S4* xy1, S4* xy2, S4* xy3, S4* pp, S4* otz, S4* flag) __asm__("RotAverageNclip3");
|
||||
FI_ S4 rtp_avg_nclip_a3_v3s2(
|
||||
@@ -179,8 +179,8 @@ FI_ S4 rtp_avg_nclip_a3_v3s2(
|
||||
){
|
||||
return rtp_avg_nclip_a3_v3s2_raw(
|
||||
v0, v1, v2,
|
||||
cast(S4*R_, xy0), cast(S4*R_, xy1), cast(S4*R_, xy2),
|
||||
cast(S4*R_, pp), cast(S4*R_, otz), cast(S4*R_, flag)
|
||||
C_(S4*R_, xy0), C_(S4*R_, xy1), C_(S4*R_, xy2),
|
||||
C_(S4*R_, pp), C_(S4*R_, otz), C_(S4*R_, flag)
|
||||
);
|
||||
}
|
||||
|
||||
@@ -192,8 +192,8 @@ FI_ S4 rtp_avg_nclip_a4_v3s2(
|
||||
){
|
||||
return rtp_avg_nclip_a4_v3s2_raw(
|
||||
v0, v1, v2, v3,
|
||||
cast(S4*R_, xy0), cast(S4*R_, xy1), cast(S4*R_, xy2), cast(S4*R_, xy3),
|
||||
cast(S4*R_, pp), cast(S4*R_, otz), cast(S4*R_, flag)
|
||||
C_(S4*R_, xy0), C_(S4*R_, xy1), C_(S4*R_, xy2), C_(S4*R_, xy3),
|
||||
C_(S4*R_, pp), C_(S4*R_, otz), C_(S4*R_, flag)
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
|
||||
// Source: C:\projects\Pikuma\ps1\code\gte_hello\hello_gte_tape.c
|
||||
#pragma once
|
||||
|
||||
#pragma region hello_gte_tape
|
||||
|
||||
|
||||
// --- atom: cube_g4_face (87 words) ---
|
||||
|
||||
#define _atom_offset_cull_cube_g4_face_exit 48
|
||||
#define _atom_offset_bounds_chk_cube_g4_face_exit 12
|
||||
|
||||
enum {
|
||||
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
|
||||
atom_offset_bounds_chk_cube_g4_face_exit = _atom_offset_bounds_chk_cube_g4_face_exit,
|
||||
};
|
||||
|
||||
// --- atom: floor_f3_face (58 words) ---
|
||||
|
||||
#define _atom_offset_culling_floor_f3_face_exit 25
|
||||
#define _atom_offset_bounds_chk_floor_f3_face_exit 13
|
||||
|
||||
enum {
|
||||
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
|
||||
atom_offset_bounds_chk_floor_f3_face_exit = _atom_offset_bounds_chk_floor_f3_face_exit,
|
||||
};
|
||||
|
||||
#pragma endregion hello_gte_tape
|
||||
|
||||
+229
-86
@@ -1,6 +1,6 @@
|
||||
#include "stdio.h"
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include "assert.h"
|
||||
#include <assert.h>
|
||||
// #include "libgpu.h"
|
||||
// #include "libetc.h"
|
||||
// #include "libgte.h"
|
||||
@@ -8,30 +8,39 @@
|
||||
#include "duffle/dsl.h"
|
||||
#include "duffle/memory.h"
|
||||
#include "duffle/math.h"
|
||||
|
||||
#include "duffle/gcc_asm.h"
|
||||
#include "duffle/mips.h"
|
||||
#include "duffle/gp.h"
|
||||
#include "duffle/gte.h"
|
||||
|
||||
# include "duffle/gen/duffle.macs.h"
|
||||
# include "duffle/gen/duffle.offsets.h"
|
||||
#include "duffle/atom_dsl.h"
|
||||
#include "duffle/lottes_tape.h"
|
||||
#include "duffle/word_count.metadata.h"
|
||||
|
||||
# include "gen/gte_hello.offsets.h"
|
||||
#include "hello_gte.h"
|
||||
|
||||
enum {
|
||||
PrimitiveBuff_Len = 4096,
|
||||
OrderingTbl_Len = 2048
|
||||
};
|
||||
#include "hello_gte_tape.c"
|
||||
|
||||
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
||||
typedef def_farray(OrderingTable_Buffer, 2);
|
||||
typedef Array_(OrderingTable_Buffer, 2);
|
||||
|
||||
typedef B1 PrimitiveBuffer[PrimitiveBuff_Len];
|
||||
typedef def_farray(PrimitiveBuffer, 2);
|
||||
typedef def_struct(PrimitiveArena) {
|
||||
typedef Array_(PrimitiveBuffer, 2);
|
||||
typedef Struct_(PrimitiveArena) {
|
||||
A2_PrimitiveBuffer buf;
|
||||
U4 used;
|
||||
};
|
||||
|
||||
#define Cube_num_verts 8
|
||||
typedef def_farray(V3_S2, Cube_num_verts);
|
||||
typedef Array_(V3_S2, Cube_num_verts);
|
||||
#define Cube_num_faces 6
|
||||
typedef def_farray(V4_S2, Cube_num_faces);
|
||||
void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
||||
memory_copy(verts, & (A8_V3_S2) {
|
||||
typedef Array_(V4_S2, Cube_num_faces);
|
||||
I_ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
||||
LP_ A8_V3_S2 baked_verts = (A8_V3_S2) {
|
||||
{ -128, -128, -128 },
|
||||
{ 128, -128, -128 },
|
||||
{ 128, -128, 128 },
|
||||
@@ -40,18 +49,20 @@ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
||||
{ 128, 128, -128 },
|
||||
{ 128, 128, 128 },
|
||||
{ -128, 128, 128 }
|
||||
}, size_of(A8_V3_S2) );
|
||||
memory_copy(faces, & (A6_V4_S2) {
|
||||
};
|
||||
LP_ A6_V4_S2 baked_faces = (A6_V4_S2) {
|
||||
{ 3, 2, 0, 1 },
|
||||
{ 0, 1, 4, 5 },
|
||||
{ 4, 5, 7, 6 },
|
||||
{ 1, 2, 5, 6 },
|
||||
{ 2, 3, 6, 7 },
|
||||
{ 3, 0, 7, 4 },
|
||||
}, size_of(A6_V4_S2) );
|
||||
};
|
||||
mem_copy(u4_(verts), u4_(& baked_verts), S_(A8_V3_S2) );
|
||||
mem_copy(u4_(faces), u4_(& baked_faces), S_(A6_V4_S2) );
|
||||
return;
|
||||
}
|
||||
typedef def_struct(Ent_Cube) {
|
||||
typedef Struct_(Ent_Cube) {
|
||||
V3_S4 accel;
|
||||
V3_S4 vel;
|
||||
V3_S4 pos;
|
||||
@@ -62,22 +73,24 @@ typedef def_struct(Ent_Cube) {
|
||||
};
|
||||
|
||||
#define Floor_num_verts 4
|
||||
typedef def_farray(V3_S2, Floor_num_verts);
|
||||
typedef Array_(V3_S2, Floor_num_verts);
|
||||
#define Floor_num_faces 2
|
||||
typedef def_farray(V3_S2, Floor_num_faces);
|
||||
void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
|
||||
memory_copy(verts, &(A4_V3_S2) {
|
||||
typedef Array_(V3_S2, Floor_num_faces);
|
||||
I_ void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
|
||||
LP_ A4_V3_S2 baked_verts = (A4_V3_S2) {
|
||||
{ -900, 0, -900 },
|
||||
{ -900, 0, 900 },
|
||||
{ 900, 0, -900 },
|
||||
{ 900, 0, 900 },
|
||||
}, size_of(A8_V3_S2));
|
||||
memory_copy(faces, & (A2_V3_S2) {
|
||||
};
|
||||
LP_ A2_V3_S2 baked_faces = (A2_V3_S2) {
|
||||
{ 0, 1, 2 },
|
||||
{ 1, 3, 2 },
|
||||
}, size_of(A2_V3_S2));
|
||||
};
|
||||
mem_copy(u4_(verts), u4_(& baked_verts), S_(A4_V3_S2));
|
||||
mem_copy(u4_(faces), u4_(& baked_faces), S_(A2_V3_S2));
|
||||
};
|
||||
typedef def_struct(Ent_Floor) {
|
||||
typedef Struct_(Ent_Floor) {
|
||||
V3_S4 accel;
|
||||
V3_S4 pos;
|
||||
V3_S4 scale;
|
||||
@@ -86,31 +99,44 @@ typedef def_struct(Ent_Floor) {
|
||||
A2_V3_S2 faces;
|
||||
};
|
||||
|
||||
typedef def_struct(SMemory) {
|
||||
enum {
|
||||
Scratchpad_Len = 1024,
|
||||
MemTape_Len = 512,
|
||||
};
|
||||
typedef Struct_(SMemory) {
|
||||
U4 MemTape[MemTape_Len];
|
||||
|
||||
DoubleBuffer screen_buf;
|
||||
A2_OrderingTable_Buffer ordering_tbl;
|
||||
PrimitiveArena primitives;
|
||||
S2 active_buf_id;
|
||||
S4 active_buf_id;
|
||||
|
||||
M3_S2 tform_world;
|
||||
|
||||
Ent_Cube cube;
|
||||
Ent_Floor floor;
|
||||
};
|
||||
global SMemory static_mem;
|
||||
extern SMemory static_mem;
|
||||
|
||||
B1* prim__alloc(U4 type_width, Str8 type_name) {
|
||||
gknown PrimitiveArena* pa = & static_mem.primitives;
|
||||
gknown B1* buf = (B1*) r_(static_mem.primitives.buf)[static_mem.active_buf_id];
|
||||
U4_V scratchpad; // d-cache
|
||||
};
|
||||
global SMemory smem;
|
||||
extern SMemory smem;
|
||||
|
||||
// TODO(Ed):
|
||||
FI_ U4* spad_warm(MipsAtom atom) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
|
||||
gknown PrimitiveArena* pa = & smem.primitives;
|
||||
gknown B1* buf = (B1*) r_(smem.primitives.buf)[smem.active_buf_id];
|
||||
assert(pa->used + type_width < PrimitiveBuff_Len);
|
||||
B1* next = buf + pa->used;
|
||||
B1* next = buf + pa->used;
|
||||
pa->used += type_width;
|
||||
return next;
|
||||
}
|
||||
#define prim_alloc(type) (type*)prim__alloc(size_of(type), txt( stringify(type)))
|
||||
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
|
||||
|
||||
void gp_screen_init_c11(DoubleBuffer* screen_buf, S2* active_buf_id)
|
||||
void gp_screen_init_c11(DoubleBuffer* screen_buf, S4* active_buf_id)
|
||||
{
|
||||
reset_graph(0);
|
||||
|
||||
@@ -136,13 +162,17 @@ void gp_screen_init_c11(DoubleBuffer* screen_buf, S2* active_buf_id)
|
||||
|
||||
// Initialize and setup the GTE geometry offsets
|
||||
geom_init();
|
||||
// NOTE: geom_set_offset/geom_set_screen are kept as-is (the libgte versions
|
||||
// are known to be broken in this PSYQ 4.7 build — see report 2026-07-09).
|
||||
// The user's research wants the C-side non-tape reference to work as a
|
||||
// known-good baseline for comparison against the tape.
|
||||
geom_set_offset(ScreenRes_CenterX, ScreenRes_CenterY);
|
||||
geom_set_screen(ScreenZ);
|
||||
|
||||
set_display_enabled(1); // gp_DisplayEnabled
|
||||
}
|
||||
|
||||
void gp_display_frame(DoubleBuffer* screen_buf, S2* active_buf_id, U4* ordering_buf, PrimitiveArena* pa) {
|
||||
void gp_display_frame(DoubleBuffer* screen_buf, S4* active_buf_id, U4* ordering_buf, PrimitiveArena* pa) {
|
||||
draw_sync(0);
|
||||
vsync(0);
|
||||
displayenv_put(& r_(screen_buf->display)[active_buf_id[0] ]);
|
||||
@@ -157,14 +187,15 @@ void gp_display_frame(DoubleBuffer* screen_buf, S2* active_buf_id, U4* ordering_
|
||||
void render(void) {
|
||||
}
|
||||
|
||||
GCC_OPTIMIZATION_DISABLE
|
||||
void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
{
|
||||
orderingtbl_clear_reverse(ordering_buf, OrderingTbl_Len);
|
||||
|
||||
// Update the position based on acceleration and velocity
|
||||
gknown V3_S4_R pos = & static_mem.cube.pos;
|
||||
gknown V3_S4_R vel = & static_mem.cube.vel;
|
||||
gknown V3_S4_R acc = & static_mem.cube.accel;
|
||||
gknown V3_S4_R pos = & smem.cube.pos;
|
||||
gknown V3_S4_R vel = & smem.cube.vel;
|
||||
gknown V3_S4_R acc = & smem.cube.accel;
|
||||
add_v3s4(vel, acc[0]);
|
||||
add_v3s4_fp(pos, vel[0]);
|
||||
// vel->x += acc->x;
|
||||
@@ -174,7 +205,7 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
// pos->y += vel->y;
|
||||
// pos->z += vel->z;
|
||||
|
||||
if (pos->y + 150 > static_mem.floor.pos.y) vel->y *= -1;
|
||||
if (pos->y + 150 > smem.floor.pos.y) vel->y *= -1;
|
||||
|
||||
// Prep
|
||||
S4 nclip = 0;
|
||||
@@ -182,13 +213,16 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
A2_S2 p; //???
|
||||
S4 flag; //????
|
||||
|
||||
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
|
||||
|
||||
// Draw Cube
|
||||
if (0)
|
||||
{
|
||||
m3s2_rotation (& static_mem.cube.rot, & static_mem.tform_world);
|
||||
m3s2_translation(& static_mem.tform_world, & static_mem.cube.pos);
|
||||
m3s2_scale (& static_mem.tform_world, & static_mem.cube.scale);
|
||||
gte_matrix_set_rotation (& static_mem.tform_world);
|
||||
gte_matrix_set_translation(& static_mem.tform_world);
|
||||
m3s2_rotation (& smem.cube.rot, & smem.tform_world);
|
||||
m3s2_translation(& smem.tform_world, & smem.cube.pos);
|
||||
m3s2_scale (& smem.tform_world, & smem.cube.scale);
|
||||
// gte_matrix_set_rotation (& smem.tform_world);
|
||||
gte_matrix_set_translation(& smem.tform_world);
|
||||
for (U4 face_id = 0; face_id < Cube_num_faces; face_id += 1)
|
||||
{
|
||||
Poly_G4* quad = prim_alloc(Poly_G4); set_poly_g4(quad);
|
||||
@@ -197,11 +231,11 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
quad->c2 = rgb8( 0, 255, 255);
|
||||
quad->c3 = rgb8( 0, 255, 0);
|
||||
|
||||
V4_S2* face = & static_mem.cube.faces[face_id];
|
||||
V3_S2* p0 = & static_mem.cube.verts[face->x];
|
||||
V3_S2* p1 = & static_mem.cube.verts[face->y];
|
||||
V3_S2* p2 = & static_mem.cube.verts[face->z];
|
||||
V3_S2* p3 = & static_mem.cube.verts[face->w];
|
||||
V4_S2* face = & smem.cube.faces[face_id];
|
||||
V3_S2* p0 = & smem.cube.verts[face->x];
|
||||
V3_S2* p1 = & smem.cube.verts[face->y];
|
||||
V3_S2* p2 = & smem.cube.verts[face->z];
|
||||
V3_S2* p3 = & smem.cube.verts[face->w];
|
||||
|
||||
nclip = rtp_avg_nclip_a4_v3s2(
|
||||
p0, p1, p2, p3,
|
||||
@@ -216,72 +250,181 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
orderingtbl_add_primitive(ordering_buf[orderingtbl_z], quad);
|
||||
}
|
||||
}
|
||||
// static_mem.cube.rot.x += 6;
|
||||
// static_mem.cube.rot.y += 8;
|
||||
// static_mem.cube.rot.z += 12;
|
||||
static_mem.cube.rot.y += 20;
|
||||
// smem.cube.rot.x += 6;
|
||||
// smem.cube.rot.y += 8;
|
||||
// smem.cube.rot.z += 12;
|
||||
smem.cube.rot.y += 30;
|
||||
}
|
||||
// Draw cube (tape method) - two triangles per face
|
||||
if (1)
|
||||
{
|
||||
m3s2_rotation (& smem.cube.rot, & smem.tform_world);
|
||||
m3s2_translation(& smem.tform_world, & smem.cube.pos);
|
||||
m3s2_scale (& smem.tform_world, & smem.cube.scale);
|
||||
gte_matrix_set_rotation (& smem.tform_world);
|
||||
gte_matrix_set_translation(& smem.tform_world);
|
||||
|
||||
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||
U4 prim_cursor = prim_base + pa->used;
|
||||
|
||||
tb.used = 0; tb_scope(& tb) {
|
||||
tb_emit(& tb, rbind_cube_g4_face);
|
||||
tb_data(& tb, prim_cursor);
|
||||
tb_data(& tb, u4_(smem.cube.faces));
|
||||
tb_data(& tb, u4_(smem.cube.verts));
|
||||
tb_data(& tb, u4_(ordering_buf));
|
||||
|
||||
for (U4 i = 0; i < Cube_num_faces; i++) {
|
||||
// Two triangles per quad face: (x,y,z) and (x,z,w)
|
||||
tb_emit(& tb, cube_g4_face);
|
||||
}
|
||||
|
||||
tb_emit(& tb, sync_primitive_arena);
|
||||
tb_data(& tb, u4_(& pa->used));
|
||||
tb_data(& tb, prim_base);
|
||||
}
|
||||
tape_run(tb_slice(tb));
|
||||
|
||||
smem.cube.rot.y += 30;
|
||||
}
|
||||
// Draw Floor
|
||||
if (0)
|
||||
{
|
||||
m3s2_rotation (& static_mem.floor.rot, & static_mem.tform_world);
|
||||
m3s2_translation(& static_mem.tform_world, & static_mem.floor.pos);
|
||||
m3s2_scale (& static_mem.tform_world, & static_mem.floor.scale);
|
||||
gte_matrix_set_rotation (& static_mem.tform_world);
|
||||
gte_matrix_set_translation(& static_mem.tform_world);
|
||||
m3s2_rotation (& smem.floor.rot, & smem.tform_world);
|
||||
m3s2_translation(& smem.tform_world, & smem.floor.pos);
|
||||
m3s2_scale (& smem.tform_world, & smem.floor.scale);
|
||||
gte_matrix_set_rotation (& smem.tform_world);
|
||||
gte_matrix_set_translation(& smem.tform_world);
|
||||
for (U4 face_id = 0; face_id < Floor_num_faces; face_id += 1)
|
||||
{
|
||||
Poly_F3* tri = prim_alloc(Poly_F3); set_poly_f3(tri);
|
||||
tri->color = rgb8(255, 255, 255);
|
||||
|
||||
V3_S2* face = & static_mem.floor.faces[face_id];
|
||||
V3_S2* p0 = & static_mem.floor.verts[face->x];
|
||||
V3_S2* p1 = & static_mem.floor.verts[face->y];
|
||||
V3_S2* p2 = & static_mem.floor.verts[face->z];
|
||||
V3_S2* face = & smem.floor.faces[face_id];
|
||||
register V3_S2* p0 rgcc(R_T4) = & smem.floor.verts[face->x];
|
||||
register V3_S2* p1 rgcc(R_T5) = & smem.floor.verts[face->y];
|
||||
register V3_S2* p2 rgcc(R_T6) = & smem.floor.verts[face->z];
|
||||
|
||||
nclip = rtp_avg_nclip_a3_v3s2(p0, p1, p2
|
||||
, & tri->p0, & tri->p1, & tri->p2
|
||||
, & p, & orderingtbl_z, & flag
|
||||
gte_load_v0(p0, R_T4);
|
||||
/*
|
||||
asm volatile( ".word " "%0" ", %1" : :
|
||||
"i"(((op_lwc2 & OPCODE_MASK) << OPCODE_SHIFT) | ((R_T4 & REG_MASK) << RS_SHIFT) | ((gte_in_v0_xy & REG_MASK) << RT_SHIFT) | (0 & IMM_MASK)),
|
||||
"i"(((op_lwc2 & OPCODE_MASK) << OPCODE_SHIFT) | ((R_T4 & REG_MASK) << RS_SHIFT) | ((gte_in_v0_z & REG_MASK) << RT_SHIFT) | (GTE_Z_Offset & IMM_MASK)),
|
||||
"r"(p0) :
|
||||
"$2", "$8", "$9", "$31", "memory"
|
||||
);
|
||||
if (nclip <= 0) {
|
||||
continue;
|
||||
}
|
||||
*/
|
||||
gte_load_v1(p1, R_T5);
|
||||
gte_load_v2(p2, R_T6);
|
||||
|
||||
if ((orderingtbl_z > 0) && (orderingtbl_z < OrderingTbl_Len)) {
|
||||
orderingtbl_add_primitive(ordering_buf[orderingtbl_z], tri);
|
||||
gte_rtpt();
|
||||
gte_nclip();
|
||||
gte_stotz(& nclip);
|
||||
|
||||
// nclip = rtp_avg_nclip_a3_v3s2(p0, p1, p2
|
||||
// , & tri->p0, & tri->p1, & tri->p2
|
||||
// , & p, & orderingtbl_z, & flag
|
||||
// );
|
||||
// if (nclip <= 0) {
|
||||
// continue;
|
||||
// }
|
||||
|
||||
if (nclip > 0 ) {
|
||||
gte_stsxy3(& tri->p0, & tri->p1, & tri->p2);
|
||||
gte_avsz3();
|
||||
gte_stotz(& orderingtbl_z);
|
||||
|
||||
if ((orderingtbl_z > 0) && (orderingtbl_z < OrderingTbl_Len)) {
|
||||
orderingtbl_add_primitive(ordering_buf[orderingtbl_z], tri);
|
||||
}
|
||||
}
|
||||
}
|
||||
static_mem.floor.rot.y += 5;
|
||||
smem.floor.rot.y += 5;
|
||||
}
|
||||
// Draw floor tape method
|
||||
if (1)
|
||||
{
|
||||
m3s2_rotation (& smem.floor.rot, & smem.tform_world);
|
||||
m3s2_translation(& smem.tform_world, & smem.floor.pos);
|
||||
m3s2_scale (& smem.tform_world, & smem.floor.scale);
|
||||
|
||||
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||
U4 prim_cursor = prim_base + pa->used;
|
||||
|
||||
// TODO(Ed): We should do a bounds check beforehand to confirm pa can hold all tris?
|
||||
// The tape atoms in-flight should not need to care.
|
||||
|
||||
// Prepare the tape. (Push protocol to tape)
|
||||
tb.used = 0; tb_scope(& tb) {
|
||||
tb_emit(& tb, set_gte_world);
|
||||
tb_data(& tb, u4_(& smem.tform_world));
|
||||
|
||||
tb_emit(& tb, rbind_floor_f3_face);
|
||||
// TODO(Ed): Just use a single context struct ref
|
||||
tb_data(& tb, prim_cursor);
|
||||
tb_data(& tb, u4_(smem.floor.faces));
|
||||
tb_data(& tb, u4_(smem.floor.verts));
|
||||
tb_data(& tb, u4_(ordering_buf));
|
||||
for (U4 i = 0; i < Floor_num_faces; i++) {
|
||||
tb_emit(& tb, floor_f3_face);
|
||||
}
|
||||
// After floor_f3_face iterations complete, the primitive arena's used counter needs updating.
|
||||
tb_emit(& tb, sync_primitive_arena);
|
||||
tb_data(& tb, u4_(& pa->used));
|
||||
tb_data(& tb, prim_base);
|
||||
}
|
||||
tape_run(tb_slice(tb));// Fire off the tape.
|
||||
|
||||
// C-side state (pa->used) has already been updated by the tape!
|
||||
smem.floor.rot.y += 5;
|
||||
}
|
||||
// --- TAPE DIAGNOSTICS ---
|
||||
if (0)
|
||||
{
|
||||
LP_ U4 mem_temp_tape[512]; FArena tape_arena; farena_init(& tape_arena, slice_ut_arr(mem_temp_tape));
|
||||
TapeBuilder tb = tb_make_old(& tape_arena); tb_scope(& tb) {
|
||||
// Skip set_gte_world atom for diagnostics to isolate the triangle loop
|
||||
for (U4 i = 0; i < Floor_num_faces; i++) {
|
||||
// tb_emit(& tb, code_diag_yield);
|
||||
// tb_emit(& tb, code_diag_color);
|
||||
// tb_emit(& tb, code_diag_gte);
|
||||
}
|
||||
}
|
||||
B1* prim_cursor = (B1*)r_(pa->buf)[smem.active_buf_id] + pa->used;
|
||||
tape_run(tb_slice(tb));
|
||||
pa->used = (U4)prim_cursor - (U4)r_(pa->buf)[smem.active_buf_id];
|
||||
}
|
||||
}
|
||||
GCC_OPTIMIZATION_ENABLE
|
||||
|
||||
int main(void)
|
||||
{
|
||||
static_mem = (SMemory){0};
|
||||
static_mem.primitives.used = 0;
|
||||
ent_cube128_init(& static_mem.cube.verts, & static_mem.cube.faces); {
|
||||
Ent_Cube* cube = & static_mem.cube;
|
||||
smem = (SMemory){0};
|
||||
smem.scratchpad = C_(U4_V, 0x1F800000);
|
||||
smem.primitives.used = 0;
|
||||
ent_cube128_init(& smem.cube.verts, & smem.cube.faces); {
|
||||
Ent_Cube* cube = & smem.cube;
|
||||
cube->rot = v3s2(0, 0, 0);
|
||||
// cube->pos = v3s4(0, 0, 900);
|
||||
cube->scale = v3s4_fp_one();
|
||||
cube->accel = v3s4(0, 1, 0);
|
||||
cube->pos = v3s4(0, -400, 1800);
|
||||
}
|
||||
ent_floor_init(& static_mem.floor.verts, & static_mem.floor.faces); {
|
||||
Ent_Floor* floor = & static_mem.floor;
|
||||
ent_floor_init(& smem.floor.verts, & smem.floor.faces); {
|
||||
Ent_Floor* floor = & smem.floor;
|
||||
floor->rot = v3s2(0, 0, 0);
|
||||
floor->pos = v3s4(0, 450, 1800);
|
||||
floor->scale = v3s4_fp_one();
|
||||
}
|
||||
// gknown gp_screen_init();
|
||||
gp_screen_init_c11(& static_mem.screen_buf, & static_mem.active_buf_id);
|
||||
while (1)
|
||||
{
|
||||
gknown S2* active_buf_id = & static_mem.active_buf_id;
|
||||
gknown U4* ordering_buf = r_(static_mem.ordering_tbl)[active_buf_id[0]];
|
||||
gknown PrimitiveArena* pa = & static_mem.primitives;
|
||||
gp_screen_init_c11(& smem.screen_buf, & smem.active_buf_id);
|
||||
while (1) {
|
||||
gknown S4* active_buf_id = & smem.active_buf_id;
|
||||
gknown U4* ordering_buf = r_(smem.ordering_tbl)[active_buf_id[0]];
|
||||
gknown PrimitiveArena* pa = & smem.primitives;
|
||||
update(pa, ordering_buf);
|
||||
render();
|
||||
gp_display_frame(& static_mem.screen_buf, active_buf_id, ordering_buf, pa);
|
||||
gp_display_frame(& smem.screen_buf, active_buf_id, ordering_buf, pa);
|
||||
};
|
||||
return 0;
|
||||
}
|
||||
|
||||
+17
-105
@@ -5,8 +5,13 @@
|
||||
# include "duffle/gp.h"
|
||||
#endif
|
||||
|
||||
typedef def_struct(DrawEnv_Packed) { U4 tag; U4 code[15]; };
|
||||
typedef def_struct(DrawEnv) {
|
||||
enum {
|
||||
PrimitiveBuff_Len = 4096,
|
||||
OrderingTbl_Len = 2048
|
||||
};
|
||||
|
||||
typedef Struct_(DrawEnv_Packed) { U4 tag; U4 code[15]; };
|
||||
typedef Struct_(DrawEnv) {
|
||||
Rect_S2 clip_area;
|
||||
A2_S2 drawing_offset;
|
||||
Rect_S2 texture_window;
|
||||
@@ -17,7 +22,7 @@ typedef def_struct(DrawEnv) {
|
||||
RGB8 initial_bg_color;
|
||||
DrawEnv_Packed dr_env; // reserved
|
||||
};
|
||||
typedef def_struct(DisplayEnv) {
|
||||
typedef Struct_(DisplayEnv) {
|
||||
Rect_S2 display_area;
|
||||
Rect_S2 screen;
|
||||
B1 vinterlace;
|
||||
@@ -25,9 +30,9 @@ typedef def_struct(DisplayEnv) {
|
||||
B1 pad0;
|
||||
B1 pad1;
|
||||
};
|
||||
typedef def_farray(DrawEnv, 2);
|
||||
typedef def_farray(DisplayEnv, 2);
|
||||
typedef def_struct(DoubleBuffer) {
|
||||
typedef Array_(DrawEnv, 2);
|
||||
typedef Array_(DisplayEnv, 2);
|
||||
typedef Struct_(DoubleBuffer) {
|
||||
A2_DrawEnv draw;
|
||||
A2_DisplayEnv display;
|
||||
};
|
||||
@@ -58,106 +63,13 @@ U4 vsync(U4 mode) __asm__("VSync");
|
||||
|
||||
void draw_orderingtbl(U4* buf) __asm__("DrawOTag");
|
||||
|
||||
typedef def_struct(PolyTag) {
|
||||
U4 addr: 24;
|
||||
U4 len: 8;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
};
|
||||
|
||||
/*
|
||||
* Primitive Handling Macros
|
||||
*/
|
||||
|
||||
#define set_len( p, _len) (((PolyTag*R_)(p))->len = (B1)(_len))
|
||||
#define set_addr(p, _addr) (((PolyTag*R_)(p))->addr = (U4)(_addr))
|
||||
#define set_code(p, _code) (((PolyTag*R_)(p))->code = (B1)(_code))
|
||||
|
||||
#define get_len(p) (B1)(((PolyTag*R_)(p))->len)
|
||||
#define get_code(p) (B1)(((PolyTag*R_)(p))->code)
|
||||
#define get_addr(p) (U4)(((PolyTag*R_)(p))->addr)
|
||||
|
||||
#define orderingtbl_add_primitive(ot, p) set_addr(p, get_addr(ot)), set_addr(ot, p)
|
||||
#define orderingtbl_add_primitives(ot, p0, p1) set_addr(p1, get_addr(ot)), set_addr(ot, p0)
|
||||
|
||||
/* Primitive Length Code */
|
||||
|
||||
#define set_poly_f3(p) set_len(p, 4), set_code(p, 0x20)
|
||||
#define set_poly_ft3(p) set_len(p, 7), set_code(p, 0x24)
|
||||
#define set_poly_g3(p) set_len(p, 6), set_code(p, 0x30)
|
||||
#define set_poly_gt3(p) set_len(p, 9), set_code(p, 0x34)
|
||||
#define set_poly_f4(p) set_len(p, 5), set_code(p, 0x28)
|
||||
#define set_poly_ft4(p) set_len(p, 9), set_code(p, 0x2c)
|
||||
#define set_poly_g4(p) set_len(p, 8), set_code(p, 0x38)
|
||||
#define set_poly_gt4(p) set_len(p, 12), set_code(p, 0x3c)
|
||||
|
||||
// #define setSprt8(p) setlen(p, 3), setcode(p, 0x74)
|
||||
// #define setSprt16(p) setlen(p, 3), setcode(p, 0x7c)
|
||||
// #define setSprt(p) setlen(p, 4), setcode(p, 0x64)
|
||||
|
||||
// #define setTile1(p) set_len(p, 2), set_code(p, 0x68)
|
||||
// #define setTile8(p) set_len(p, 2), set_code(p, 0x70)
|
||||
// #define setTile16(p) set_len(p, 2), set_code(p, 0x78)
|
||||
#define set_tile(p) set_len(p, 3), set_code(p, 0x60)
|
||||
// #define setLineF2(p) set_len(p, 3), set_code(p, 0x40)
|
||||
// #define setLineG2(p) set_len(p, 4), set_code(p, 0x50)
|
||||
// #define setLineF3(p) set_len(p, 5), set_code(p, 0x48),(p)->pad = 0x55555555
|
||||
// #define setLineG3(p) set_len(p, 7), set_code(p, 0x58),(p)->pad = 0x55555555, (p)->p2 = 0
|
||||
// #define setLineF4(p) set_len(p, 6), set_code(p, 0x4c),(p)->pad = 0x55555555
|
||||
// #define setLineG4(p) set_len(p, 9), set_code(p, 0x5c),(p)->pad = 0x55555555, (p)->p2 = 0, (p)->p3 = 0
|
||||
|
||||
typedef def_struct(Poly_F3) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
union {
|
||||
struct {
|
||||
V2_S2 p0;
|
||||
V2_S2 p1;
|
||||
V2_S2 p2;
|
||||
};
|
||||
A3_V2_S2 points;
|
||||
};
|
||||
};
|
||||
|
||||
typedef def_struct(Poly_G3) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2;
|
||||
};
|
||||
|
||||
typedef def_struct(Poly_F4) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
union {
|
||||
struct {
|
||||
V2_S2 p0;
|
||||
V2_S2 p1;
|
||||
V2_S2 p2;
|
||||
V2_S2 p3;
|
||||
};
|
||||
A4_V2_S2 points;
|
||||
};
|
||||
};
|
||||
|
||||
typedef def_struct(Poly_G4) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2; RGB8 c3; B1 pad3;
|
||||
V2_S2 p3;
|
||||
};
|
||||
|
||||
typedef def_struct(Tile) {
|
||||
typedef Struct_(Tile) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
Rect_S2 rect;
|
||||
};
|
||||
|
||||
|
||||
/*
|
||||
Linear Algebra
|
||||
*/
|
||||
@@ -169,7 +81,7 @@ M3_S2* m3s2_scale (M3_S2* mat, V3_S4* vec) __asm__("ScaleMatrix");
|
||||
// Rotation, Translation, Perspective
|
||||
|
||||
S4 rtp_v3s2_raw(V3_S2* vec, S4* xy, S4* pp, S4* flag) __asm__("RotTransPers");
|
||||
FI_ S4 rtp_v3s2(V3_S2* vec, V2_S2* xy, A2_S2* pp, S4* flag) { return rtp_v3s2_raw(vec, cast(S4*R_, & xy->x), cast(S4*R_, pp), r_(flag)); }
|
||||
FI_ S4 rtp_v3s2(V3_S2* vec, V2_S2* xy, A2_S2* pp, S4* flag) { return rtp_v3s2_raw(vec, C_(S4*R_, & xy->x), C_(S4*R_, pp), r_(flag)); }
|
||||
|
||||
S4 rtp_avg_nclip_a3_v3s2_raw(V3_S2* v0, V3_S2* v1, V3_S2* v2, S4* xy1, S4* xy2, S4* xy3, S4* pp, S4* otz, S4* flag) __asm__("RotAverageNclip3");
|
||||
FI_ S4 rtp_avg_nclip_a3_v3s2(
|
||||
@@ -179,8 +91,8 @@ FI_ S4 rtp_avg_nclip_a3_v3s2(
|
||||
){
|
||||
return rtp_avg_nclip_a3_v3s2_raw(
|
||||
v0, v1, v2,
|
||||
cast(S4*R_, xy0), cast(S4*R_, xy1), cast(S4*R_, xy2),
|
||||
cast(S4*R_, pp), cast(S4*R_, otz), cast(S4*R_, flag)
|
||||
C_(S4*R_, xy0), C_(S4*R_, xy1), C_(S4*R_, xy2),
|
||||
C_(S4*R_, pp), C_(S4*R_, otz), C_(S4*R_, flag)
|
||||
);
|
||||
}
|
||||
|
||||
@@ -192,8 +104,8 @@ FI_ S4 rtp_avg_nclip_a4_v3s2(
|
||||
){
|
||||
return rtp_avg_nclip_a4_v3s2_raw(
|
||||
v0, v1, v2, v3,
|
||||
cast(S4*R_, xy0), cast(S4*R_, xy1), cast(S4*R_, xy2), cast(S4*R_, xy3),
|
||||
cast(S4*R_, pp), cast(S4*R_, otz), cast(S4*R_, flag)
|
||||
C_(S4*R_, xy0), C_(S4*R_, xy1), C_(S4*R_, xy2), C_(S4*R_, xy3),
|
||||
C_(S4*R_, pp), C_(S4*R_, otz), C_(S4*R_, flag)
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
@@ -1,170 +0,0 @@
|
||||
// .include "./toolchain/pcsx-redux/src/mips/common/crt0/crt0.s"
|
||||
|
||||
.include "./asmdd/dsl.s"
|
||||
.include "./asmdd/math.s"
|
||||
.include "./asmdd/io.s"
|
||||
.include "./asmdd/gp.s"
|
||||
|
||||
# DrawEnv_Packed { U4 tag; U4 code[15]; }
|
||||
.equ DrawEnv_Packed_tag, 0
|
||||
.equ DrawEnv_Packed_code, DrawEnv_Packed_tag + U4
|
||||
.equ DrawEnv_Packed, 64
|
||||
# DrawEnv { Rect_S2 clip; V2_S2 ofs; Rect_S2 tw; U2 tpage; U8 dtd; U8 dfe; U8 tme; U8 r0,g0,b0; DR_ENV dr_env; }
|
||||
.equ DrawEnv_clip_area, /* 0 */ Rect_S2 * 0
|
||||
.equ DrawEnv_drawing_offset, /* 8 */ V2_S2 * 0 + Rect_S2
|
||||
.equ DrawEnv_texture_window, /* 12 */ Rect_S2 * 0 + A2_S2 + DrawEnv_drawing_offset
|
||||
.equ DrawEnv_texture_page, /* 20 */ S1 * 0 + Rect_S2 + DrawEnv_texture_window
|
||||
.equ DrawEnv_flag_dither, /* 22 */ B1 * 0 + S2 + DrawEnv_texture_page
|
||||
.equ DrawEnv_flag_draw_on_display, /* 23 */ B1 * 0 + B1 + DrawEnv_flag_dither
|
||||
.equ DrawEnv_enable_auto_clear, /* 24 */ B1 * 0 + B1 + DrawEnv_flag_draw_on_display
|
||||
.equ DrawEnv_initial_bg_color, /* 25 */ RGB8 * 0 + B1 + DrawEnv_enable_auto_clear
|
||||
.equ DrawEnv_dr_env, /* 28 */ DrawEnv_Packed * 0 + RGB8 + DrawEnv_initial_bg_color
|
||||
.equ DrawEnv, /* 92 */ DrawEnv_dr_env + DrawEnv_Packed
|
||||
# DisplayEnv { Rect_S16 disp; Rect_S16 screen; U8 isinter; U8 isrgb24; U8 pad[2]; }
|
||||
.equ DisplayEnv_display_area, Rect_S2 * 0
|
||||
.equ DisplayEnv_screen, Rect_S2 * 0 + Rect_S2 + DisplayEnv_display_area
|
||||
.equ DisplayEnv_vinterlace, B1 * 0 + Rect_S2 + DisplayEnv_screen
|
||||
.equ DisplayEnv_color24, B1 * 0 + B1 + DisplayEnv_vinterlace
|
||||
.equ DisplayEnv_pad0, B1 * 0 + B1 + DisplayEnv_color24
|
||||
.equ DisplayEnv_pad1, B1 * 0 + B1 + DisplayEnv_pad0
|
||||
.equ DisplayEnv, DisplayEnv_pad1 + B1
|
||||
# DoubleBuffer { DrawEnv draw[2]; DisplayEnv display[2]; }
|
||||
.equ DoubleBuffer_draw, 0
|
||||
.equ DoubleBuffer_draw_0, (DrawEnv * 0)
|
||||
.equ DoubleBuffer_draw_1, (DrawEnv * 1)
|
||||
.equ DoubleBuffer_display, (DrawEnv * 2)
|
||||
.equ DoubleBuffer_display_0, (DisplayEnv * 0) + DoubleBuffer_display
|
||||
.equ DoubleBuffer_display_1, (DisplayEnv * 1) + DoubleBuffer_display
|
||||
.equ DoubleBuffer, (DisplayEnv * 2) + DoubleBuffer_display
|
||||
# Screen Constants
|
||||
.equ ScreenRes_X, 320
|
||||
.equ ScreenRes_Y, 240
|
||||
.equ ScreenRes_CenterX, (ScreenRes_X >> 1)
|
||||
.equ ScreenRes_CenterY, (ScreenRes_Y >> 1)
|
||||
|
||||
.equ SMemory_screen_buf, DoubleBuffer * 0
|
||||
.equ SMemory_active_screen_buf, S2 * 0 + DoubleBuffer
|
||||
|
||||
.equ CF_Shadow, 16
|
||||
|
||||
.extern ResetGraph
|
||||
.equ ResetGraph_mode, rarg_0
|
||||
|
||||
.extern SetDispMask
|
||||
.equ SetDispMask_mask, rarg_0
|
||||
|
||||
.extern PutDispEnv
|
||||
.extern PutDrawEnv
|
||||
.equ PutDispEnv_env, rarg_0
|
||||
.equ PutDrawEnv_env, rarg_0
|
||||
|
||||
.extern SetDefDispEnv
|
||||
.equ SetDefDispEnv_env, rarg_0
|
||||
.equ SetDefDispEnv_x, rarg_1
|
||||
.equ SetDefDispEnv_y, rarg_2
|
||||
.equ SetDefDispEnv_w, rarg_3
|
||||
.equ SetDefDispEnv_h, CF_Shadow
|
||||
.set SetDefDispEnv_sp_size, CF_Shadow + S4
|
||||
|
||||
.extern SetDefDrawEnv
|
||||
.equ SetDefDrawEnv_env, rarg_0
|
||||
.equ SetDefDrawEnv_x, rarg_1
|
||||
.equ SetDefDrawEnv_y, rarg_2
|
||||
.equ SetDefDrawEnv_w, rarg_3
|
||||
.equ SetDefDrawEnv_h, CF_Shadow
|
||||
.set SetDefDrawEnv_sp_size, CF_Shadow + S4
|
||||
|
||||
.extern SetGeomOffset
|
||||
.equ SetGeomOffset_x, rarg_0
|
||||
.equ SetGeomOffset_y, rarg_1
|
||||
|
||||
.extern SetGeomScreen
|
||||
.equ SetGeomScreen_h, rarg_0
|
||||
|
||||
.global gp_screen_init_asm
|
||||
.type gp_screen_init_asm, @function
|
||||
gp_screen_init_asm:
|
||||
.equiv rio_offset, rtmp_0
|
||||
load_imm rtmp_0, IO_BASE_ADDR
|
||||
#define gp0 gpio_port0(rio_offset)
|
||||
#define gp1 gpio_port1(rio_offset)
|
||||
|
||||
def_cf_sp_size 0x18; // Should be enough for all calls within this proc, for some reason SetDefDispEnv needs the offset to be CF_Shadow..
|
||||
stack_alloc cf_ssize
|
||||
store_word rret_addr, 0($sp)
|
||||
|
||||
// Note(Ed): Cannot be used psyq manages things related to vblank and other things so the api must be called instead
|
||||
// gcmd_push gp1, rtmp_1, gp_Reset // ResetGraph(0)
|
||||
// gcmd_push gp1, rtmp_1, gp_DisplayEnabled // SetDispMask(1)
|
||||
load_imm ResetGraph_mode, gp_Reset; jump_nlink ResetGraph
|
||||
load_imm SetDispMask_mask, 1; jump_nlink SetDispMask
|
||||
|
||||
// First buffer area
|
||||
load_addr rtmp_0, static_mem; add_ui SetDefDispEnv_env, rtmp_0, SMemory_screen_buf + DoubleBuffer_display_0
|
||||
move SetDefDispEnv_x, $zero
|
||||
move SetDefDispEnv_y, $zero
|
||||
load_imm SetDefDispEnv_w, ScreenRes_X
|
||||
load_imm rtmp_0, ScreenRes_Y; store_word rtmp_0, SetDefDispEnv_h($sp)
|
||||
jump_nlink SetDefDispEnv
|
||||
load_addr rtmp_0, static_mem; add_ui SetDefDrawEnv_env, rtmp_0, SMemory_screen_buf + DoubleBuffer_draw_0
|
||||
move SetDefDrawEnv_x, $zero
|
||||
load_imm SetDefDrawEnv_y, ScreenRes_Y
|
||||
load_imm SetDefDrawEnv_w, ScreenRes_X
|
||||
load_imm rtmp_0, ScreenRes_Y; store_word rtmp_0, SetDefDrawEnv_h($sp)
|
||||
jump_nlink SetDefDrawEnv
|
||||
// Second buffer area
|
||||
load_addr rtmp_0, static_mem; add_ui SetDefDispEnv_env, rtmp_0, SMemory_screen_buf + DoubleBuffer_display_1
|
||||
move SetDefDispEnv_x, $zero
|
||||
load_imm SetDefDispEnv_y, ScreenRes_Y
|
||||
load_imm SetDefDispEnv_w, ScreenRes_X
|
||||
load_imm rtmp_0, ScreenRes_Y; store_word rtmp_0, SetDefDispEnv_h($sp)
|
||||
jump_nlink SetDefDispEnv
|
||||
load_addr rtmp_0, static_mem; add_ui SetDefDrawEnv_env, rtmp_0, SMemory_screen_buf + DoubleBuffer_draw_1
|
||||
move SetDefDrawEnv_x, $zero
|
||||
move SetDefDrawEnv_y, $zero
|
||||
load_imm SetDefDrawEnv_w, ScreenRes_X
|
||||
load_imm rtmp_0, ScreenRes_Y; store_word rtmp_0, SetDefDrawEnv_h($sp)
|
||||
jump_nlink SetDefDrawEnv
|
||||
|
||||
// Set the back/drawing buffer
|
||||
load_imm rtmp_1, true
|
||||
load_addr rtmp_0, static_mem; // At SMemory_screen_buf
|
||||
store_word rtmp_1, DoubleBuffer_draw_0 + DrawEnv_enable_auto_clear(rtmp_0)
|
||||
store_word rtmp_1, DoubleBuffer_draw_1 + DrawEnv_enable_auto_clear(rtmp_0)
|
||||
|
||||
// Set background clear color
|
||||
load_imm rtmp_1, 28; load_imm rtmp_2, 22; load_imm rtmp_3, 25
|
||||
// 63, 0, 127
|
||||
store_byte rtmp_2, DoubleBuffer_draw_0 + DrawEnv_initial_bg_color + RGB8_r(rtmp_0)
|
||||
store_byte rtmp_1, DoubleBuffer_draw_0 + DrawEnv_initial_bg_color + RGB8_g(rtmp_0)
|
||||
store_byte rtmp_3, DoubleBuffer_draw_0 + DrawEnv_initial_bg_color + RGB8_b(rtmp_0)
|
||||
// 127, 63, 0
|
||||
store_byte rtmp_3, DoubleBuffer_draw_1 + DrawEnv_initial_bg_color + RGB8_r(rtmp_0)
|
||||
store_byte rtmp_2, DoubleBuffer_draw_1 + DrawEnv_initial_bg_color + RGB8_g(rtmp_0)
|
||||
store_byte rtmp_1, DoubleBuffer_draw_1 + DrawEnv_initial_bg_color + RGB8_b(rtmp_0)
|
||||
load_addr rtmp_0, static_mem; store_word rtmp_1, SMemory_active_screen_buf(rtmp_0)
|
||||
|
||||
load_addr rtmp_1, static_mem; load_half rtmp_1, SMemory_active_screen_buf(rtmp_1); // rtmp_1 = active_screen_buffer
|
||||
load_imm rtmp_2, DisplayEnv; mult_u rtmp_1, rtmp_2; mov_from_low rtmp_2 // rtmp_2 = DisplayEnv.type_size * active_screen_Buffer (rtmp_1)
|
||||
add_ui rtmp_2, rtmp_2, DoubleBuffer_display // rtmp_2 += DoubleBuffer.display
|
||||
load_addr rtmp_0, static_mem; add_u PutDispEnv_env, rtmp_0, rtmp_2 // rarg_0 = rtmp_0 (screen_buffer) + rtmp_2 (.display[active_screen-buffer])
|
||||
jump_nlink PutDispEnv
|
||||
load_addr rtmp_1, static_mem; load_half rtmp_1, SMemory_active_screen_buf(rtmp_1);
|
||||
load_imm rtmp_2, DrawEnv; mult_u rtmp_1, rtmp_2; mov_from_low rtmp_2;
|
||||
add_ui rtmp_2, rtmp_2, DoubleBuffer_draw
|
||||
load_addr rtmp_0, static_mem; add_u PutDrawEnv_env, rtmp_0, rtmp_2
|
||||
jump_nlink PutDrawEnv
|
||||
|
||||
// Initialize and setup the GTE geometry offsets
|
||||
jump_nlink InitGeom
|
||||
load_imm SetGeomOffset_x, ScreenRes_CenterX
|
||||
load_imm SetGeomOffset_y, ScreenRes_CenterY
|
||||
jump_nlink SetGeomOffset
|
||||
load_imm SetGeomScreen_h, ScreenRes_CenterX
|
||||
jump_nlink SetGeomScreen
|
||||
|
||||
load_word rret_addr, 0($sp)
|
||||
stack_release cf_ssize
|
||||
jump_reg rret_addr;
|
||||
.Lgp_screen_init_end:
|
||||
.size gp_screen_init_asm, . - gp_screen_init_asm
|
||||
@@ -0,0 +1,156 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# include "duffle/gen/duffle.macs.h"
|
||||
# include "duffle/gen/duffle.offsets.h"
|
||||
# include "duffle/atom_dsl.h"
|
||||
# include "duffle/lottes_tape.h"
|
||||
# include "duffle/word_count.metadata.h"
|
||||
# include "gen/gte_hello.offsets.h"
|
||||
# include "hello_gte.h"
|
||||
#endif
|
||||
|
||||
#pragma region MACs (Mips Atom components)
|
||||
|
||||
|
||||
|
||||
#pragma endregion MACs
|
||||
|
||||
#pragma region Baked Atoms
|
||||
|
||||
typedef Struct_(Binds_CubeTri) {
|
||||
U4 PrimCursor;
|
||||
V4_S2* FaceCursor;
|
||||
V3_S2* VertBase;
|
||||
U4* OtBase;
|
||||
};
|
||||
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)
|
||||
, atom_reads(R_TapePtr)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
){
|
||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
|
||||
internal
|
||||
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
||||
atom_writes(R_PrimCursor, R_FaceCursor)
|
||||
){
|
||||
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
||||
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
||||
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
||||
|
||||
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
|
||||
nop2, gte_cmdw_rotate_translate_perspective_triple,
|
||||
nop2, gte_cmdw_nclip,
|
||||
|
||||
nop2, gte_mv_from_data_r(R_T0, C2_MAC0),
|
||||
nop,
|
||||
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)), nop,
|
||||
|
||||
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
||||
mac_format_g4_color(
|
||||
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
||||
/* c1 yellow */ 0xFF, 0xFF, 0x00,
|
||||
/* c2 cyan */ 0x00, 0xFF, 0xFF,
|
||||
/* c3 green */ 0x00, 0xFF, 0x00),
|
||||
mac_gte_store_g4_p012_post_rtpt_pre_rtps(),
|
||||
|
||||
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
||||
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
|
||||
nop2, gte_cmdw_rotate_translate_perspective_single,
|
||||
mac_gte_store_g4_p3_post_rtps(),
|
||||
|
||||
nop2, gte_cmdw_avg_sort_z4,
|
||||
nop2, gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||
|
||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
|
||||
mac_insert_ot_tag_g4(),
|
||||
|
||||
atom_label(cube_g4_face_exit)
|
||||
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
|
||||
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_FloorTri) {
|
||||
U4 PrimCursor;
|
||||
V3_S2* FaceCursor;
|
||||
V3_S2* VertBase;
|
||||
U4* OtBase;
|
||||
};
|
||||
internal
|
||||
MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(floor_f3)
|
||||
, atom_reads(R_TapePtr)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
){
|
||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
atom_dbg_skip
|
||||
internal
|
||||
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
||||
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursr)
|
||||
) {
|
||||
mac_load_tri_indices( R_T0, R_T1, R_T2),
|
||||
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
|
||||
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
|
||||
gte_cmdw_nclip,
|
||||
|
||||
/* Culling (Branch forward if Backface) */
|
||||
gte_mv_from_data_r(R_T0, C2_MAC0),
|
||||
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
|
||||
/* Format Primitive */
|
||||
mac_format_f3_color(0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
|
||||
mac_gte_store_f3_post_rtpt(),
|
||||
|
||||
/* Calculate Depth */
|
||||
gte_avg_sort_z3,
|
||||
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||
/* Bounds Check OTZ < 2048 (Branch forward to skip insertion) */
|
||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
|
||||
/* Insert into Ordering Table Linked List */
|
||||
mac_insert_ot_tag_f3(),
|
||||
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
|
||||
// Note(Ed): No bounds checking, should be checked before atom runs.
|
||||
|
||||
/* Advance Input Cursor & Yield (Both branch targets land here) */
|
||||
atom_label(floor_f3_face_exit)
|
||||
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
|
||||
internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitiveArena)
|
||||
, atom_reads( R_TapePtr, R_PrimCursor)
|
||||
, atom_writes(R_TapePtr)
|
||||
){
|
||||
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
|
||||
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
|
||||
/* Calculate byte offset and store directly back to RAM */
|
||||
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
|
||||
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
#pragma endregion Baked Atoms
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 220 KiB |
@@ -6,19 +6,34 @@ A rest from the usual.
|
||||
|
||||
## Dependencies
|
||||
|
||||
I will be programming from a Windows 11 machine:
|
||||
I will be programming from a Windows 11 machine (may eventually try this on the Steam Deck...):
|
||||

|
||||
|
||||
```ps1
|
||||
# not really used yet for scripts (may never)
|
||||
scoop install lua
|
||||
```
|
||||
|
||||
[armips](https://github.com/Kingcom/armips)
|
||||
|
||||
* Supports doing bare-metal assembly for the ps1
|
||||
* `scoop install armips` or just clone and build..
|
||||
|
||||
[luajit-2.1](https://github.com/LuaJIT/LuaJIT.git)
|
||||
|
||||
```
|
||||
scoop install luajit
|
||||
```
|
||||
|
||||
* Used for lua scripts
|
||||
* Particularly, ps1_meta.lua which is a staged metaprogram pass for the custom C11 Assembly DSL used in this codebase.
|
||||
|
||||
[lpeg](https://github.com/roberto-ieru/LPeg.git)
|
||||
|
||||
* Lua is slow (even jitted) so this helps.
|
||||
|
||||
[lfs (LuaFileSystem)](https://github.com/lunarmodules/luafilesystem)
|
||||
|
||||
* Native directory enumeration + `mkdir` for the build scripts.
|
||||
* Used by `passes/word_count_eval.lua :: scan_dir` (native walk vs. `dir /b /s` subprocess,
|
||||
~2ms vs. ~56ms) and by `duffle.lua :: ensure_dir` + `to_absolute_path` (avoids
|
||||
`cmd.exe mkdir` + `cd` shell spawns, ~50ms each).
|
||||
|
||||
[pscx-redux](https://github.com/grumpycoders/pcsx-redux/): A collection of tools, research, hardware design, and libraries aiming at development and reverse engineering on the PlayStation 1.
|
||||
|
||||
* Used as the runtime sandbox emulated the ps1
|
||||
@@ -57,3 +72,4 @@ scoop install lua
|
||||

|
||||

|
||||

|
||||

|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
-- gte_debug.lua — defensive version + prints error context.
|
||||
local ok, err = pcall(function()
|
||||
print("[debug] PCSX exists:", PCSX ~= nil)
|
||||
print("[debug] PCSX.WebServer exists:", PCSX and PCSX.WebServer ~= nil)
|
||||
print("[debug] PCSX.WebServer.Handlers exists:", PCSX and PCSX.WebServer and PCSX.WebServer.Handlers ~= nil)
|
||||
if not PCSX.WebServer then
|
||||
print("[debug] creating PCSX.WebServer...")
|
||||
PCSX.WebServer = {}
|
||||
end
|
||||
if not PCSX.WebServer.Handlers then
|
||||
print("[debug] creating PCSX.WebServer.Handlers...")
|
||||
PCSX.WebServer.Handlers = {}
|
||||
end
|
||||
print("[debug] type of Handlers:", type(PCSX.WebServer.Handlers))
|
||||
|
||||
PCSX.WebServer.Handlers.gte = function(req)
|
||||
local r = PCSX.getRegisters()
|
||||
local out = { "pc=0x" .. string.format("%x", r.pc) }
|
||||
for i = 0, 31 do
|
||||
out[#out + 1] = string.format("D[%d]=0x%08x C[%d]=0x%08x",
|
||||
i, r.CP2D.r[i], i, r.CP2C.r[i])
|
||||
end
|
||||
return table.concat(out, "\n")
|
||||
end
|
||||
print("[debug] handler registered")
|
||||
end)
|
||||
|
||||
if not ok then
|
||||
print("[debug] ERROR: " .. tostring(err))
|
||||
end
|
||||
@@ -0,0 +1,217 @@
|
||||
--- audit_lua_nesting.lua — Walk Lua source files and flag any block nesting deeper than 5 levels.
|
||||
---
|
||||
--- Usage:
|
||||
--- luajit scripts/audit_lua_nesting.lua scripts/duffle.lua scripts/ps1_meta.lua
|
||||
--- luajit scripts/audit_lua_nesting.lua scripts/passes/
|
||||
---
|
||||
--- Output: for each file, a list of {line, depth} entries where depth > 5.
|
||||
--- Returns exit code 1 if any violations found, 0 if clean.
|
||||
---
|
||||
--- **Implementation**: a hand-rolled depth tracker that counts:
|
||||
--- - `do`, `function`, `if`, `for`, `while`, `repeat` -> depth +1
|
||||
--- - `end`, `until` -> depth -1
|
||||
--- - `else`, `elseif` -> depth unchanged
|
||||
---
|
||||
--- **Caveats**: doesn't fully handle string/comment state (will miscount braces inside multi-line strings or block comments).
|
||||
--- For our metaprogram files (no embedded code generation), this is acceptable.
|
||||
|
||||
local M = {}
|
||||
|
||||
local BLOCK_OPEN = {
|
||||
["do"] = true,
|
||||
["function"] = true,
|
||||
["if"] = true,
|
||||
["for"] = true,
|
||||
["while"] = true,
|
||||
["repeat"] = true,
|
||||
}
|
||||
|
||||
local function is_block_close(token) return token == "end" or token == "until" end
|
||||
|
||||
-- (internal) Walk one source file and return a list of
|
||||
-- {line, depth, token} entries where depth > max_nesting.
|
||||
local function audit_file(path, max_nesting)
|
||||
local f = io.open(path, "r")
|
||||
if not f then error("Cannot open " .. path) end
|
||||
local content = f:read("*a")
|
||||
f:close()
|
||||
|
||||
local violations = {}
|
||||
local depth = 0
|
||||
local line = 1
|
||||
local pos = 1
|
||||
local src_len = #content
|
||||
local token_idx = 0
|
||||
|
||||
local function read_ident_at(start_pos)
|
||||
local ident_start = start_pos
|
||||
if ident_start > src_len then return nil end
|
||||
local first_ch = content:sub(ident_start, ident_start)
|
||||
if not (first_ch:match("[%a_]")) then return nil end
|
||||
local scan = start_pos + 1
|
||||
while scan <= src_len do
|
||||
local ch = content:sub(scan, scan)
|
||||
if not (ch:match("[%w_]")) then break end
|
||||
scan = scan + 1
|
||||
end
|
||||
return content:sub(ident_start, scan - 1), scan
|
||||
end
|
||||
|
||||
-- Skip past a string literal or comment starting at `start_pos`.
|
||||
-- Returns the position just past the construct, or nil if `start_pos`
|
||||
-- is not the start of a string/comment.
|
||||
local function skip_string_or_comment(start_pos)
|
||||
local ch = content:sub(start_pos, start_pos)
|
||||
if ch == '"' or ch == "'" then
|
||||
local scan = start_pos + 1
|
||||
while scan <= src_len do
|
||||
local c = content:sub(scan, scan)
|
||||
if c == "\\" then scan = scan + 2
|
||||
elseif c == ch then return scan + 1
|
||||
else scan = scan + 1
|
||||
end
|
||||
end
|
||||
return src_len + 1
|
||||
elseif ch == "-" and content:sub(start_pos + 1, start_pos + 1) == "-" then
|
||||
local scan = start_pos + 2
|
||||
if content:sub(scan, scan + 1) == "[[" and content:sub(scan + 2, scan + 3) == "[" then
|
||||
-- Long bracket comment [==[ ... ]==]
|
||||
scan = scan + 2
|
||||
local eq = ""
|
||||
while content:sub(scan, scan) == "=" do
|
||||
eq = eq .. "="
|
||||
scan = scan + 1
|
||||
end
|
||||
local close_marker = "]" .. eq .. "]"
|
||||
local close_pos = content:find(close_marker, scan, true)
|
||||
if close_pos then
|
||||
return close_pos + #close_marker
|
||||
else
|
||||
return src_len + 1
|
||||
end
|
||||
else
|
||||
while scan <= src_len and content:sub(scan, scan) ~= "\n" do scan = scan + 1 end
|
||||
return scan + 1
|
||||
end
|
||||
elseif ch == "[" and content:sub(start_pos + 1, start_pos + 1) == "[" then
|
||||
local scan = start_pos + 2
|
||||
local eq = ""
|
||||
while content:sub(scan, scan) == "=" do
|
||||
eq = eq .. "="
|
||||
scan = scan + 1
|
||||
end
|
||||
local close_marker = "]" .. eq .. "]"
|
||||
local close_pos = content:find(close_marker, scan, true)
|
||||
if close_pos then
|
||||
return close_pos + #close_marker
|
||||
else
|
||||
return src_len + 1
|
||||
end
|
||||
end
|
||||
return nil
|
||||
end
|
||||
|
||||
while pos <= src_len do
|
||||
local ch = content:sub(pos, pos)
|
||||
if ch == "\n" then line = line + 1 end
|
||||
|
||||
local skip_to = skip_string_or_comment(pos)
|
||||
if skip_to then
|
||||
for scan = pos, skip_to - 1 do
|
||||
if content:sub(scan, scan) == "\n" then line = line + 1 end
|
||||
end
|
||||
pos = skip_to
|
||||
elseif ch:match("[%a_]") then
|
||||
local tok, next_pos = read_ident_at(pos)
|
||||
token_idx = token_idx + 1
|
||||
if BLOCK_OPEN[tok] then
|
||||
depth = depth + 1
|
||||
if depth > max_nesting then
|
||||
violations[#violations + 1] = {
|
||||
line = line,
|
||||
depth = depth,
|
||||
token = tok,
|
||||
}
|
||||
end
|
||||
elseif is_block_close(tok) then
|
||||
depth = depth - 1
|
||||
end
|
||||
pos = next_pos
|
||||
else
|
||||
pos = pos + 1
|
||||
end
|
||||
end
|
||||
|
||||
return violations
|
||||
end
|
||||
|
||||
--- Audit one file. Returns nil if clean, else a list of violations.
|
||||
--- @param path string
|
||||
--- @param max_nesting integer -- default 5
|
||||
--- @return table|nil
|
||||
function M.audit(path, max_nesting)
|
||||
local violations = audit_file(path, max_nesting or 5)
|
||||
if #violations == 0 then return nil end
|
||||
return violations
|
||||
end
|
||||
|
||||
-- Module CLI.
|
||||
if arg and arg[1] then
|
||||
local max_nesting = 5
|
||||
local files = {}
|
||||
for arg_idx = 1, #arg do
|
||||
if arg[arg_idx] == "--max" and arg[arg_idx + 1] then
|
||||
max_nesting = tonumber(arg[arg_idx + 1]) or 5
|
||||
else
|
||||
files[#files + 1] = arg[arg_idx]
|
||||
end
|
||||
end
|
||||
|
||||
-- Accept either a directory or a file path. Directory args are
|
||||
-- expanded via lfs.dir (native, no subprocess).
|
||||
local lfs = require("lfs")
|
||||
local function is_dir(p)
|
||||
return lfs.attributes(p, "mode") == "directory"
|
||||
end
|
||||
local function list_lua(dir)
|
||||
local out = {}
|
||||
if not is_dir(dir) then return out end
|
||||
for entry in lfs.dir(dir) do
|
||||
if entry:match("%.lua$") then
|
||||
out[#out + 1] = dir .. "/" .. entry
|
||||
end
|
||||
end
|
||||
return out
|
||||
end
|
||||
|
||||
local to_check = {}
|
||||
for _, f in ipairs(files) do
|
||||
if is_dir(f) then
|
||||
for _, sub in ipairs(list_lua(f)) do to_check[#to_check + 1] = sub end
|
||||
else
|
||||
to_check[#to_check + 1] = f
|
||||
end
|
||||
end
|
||||
|
||||
local total_violations = 0
|
||||
for _, f in ipairs(to_check) do
|
||||
local v = M.audit(f, max_nesting)
|
||||
if v then
|
||||
io.write(string.format("\n%s\n", f))
|
||||
for _, x in ipairs(v) do
|
||||
io.write(string.format(" line %d: depth %d (after '%s')\n", x.line, x.depth, x.token))
|
||||
end
|
||||
total_violations = total_violations + #v
|
||||
end
|
||||
end
|
||||
|
||||
if total_violations == 0 then
|
||||
io.write("OK: no files exceed max nesting of " .. max_nesting .. "\n")
|
||||
os.exit(0)
|
||||
else
|
||||
io.write(string.format("\n%d nesting violation(s) found.\n", total_violations))
|
||||
os.exit(1)
|
||||
end
|
||||
end
|
||||
|
||||
return M
|
||||
+107
-107
@@ -7,123 +7,123 @@ Converts a raw binary file to PlayStation 1 (PS-X) executable format.
|
||||
]]
|
||||
|
||||
function file_size(filename)
|
||||
local file = io.open(filename, "rb")
|
||||
if not file then return nil end
|
||||
local size = file:seek("end")
|
||||
file:close()
|
||||
return size
|
||||
local file = io.open(filename, "rb")
|
||||
if not file then return nil end
|
||||
local size = file:seek("end")
|
||||
file:close()
|
||||
return size
|
||||
end
|
||||
|
||||
function main(args)
|
||||
if #args ~= 2 then
|
||||
io.stderr:write(usage)
|
||||
os.exit(1)
|
||||
end
|
||||
if #args ~= 2 then
|
||||
io.stderr:write(usage)
|
||||
os.exit(1)
|
||||
end
|
||||
|
||||
-- print(string.format("Input file: %s", args[1]))
|
||||
-- print(string.format("Output file: %s", args[2]))
|
||||
|
||||
-- PS1 executables have a maximum size limit of 2MB
|
||||
local max_size = 0x200000
|
||||
-- print(string.format("\nChecking input file size (max: %d bytes)...", max_size))
|
||||
|
||||
local infile_size = file_size(args[1])
|
||||
if not infile_size then
|
||||
io.stderr:write("Error: Cannot open input file " .. args[1] .. "\n")
|
||||
os.exit(1)
|
||||
end
|
||||
|
||||
-- print(string.format("Input file size: %d bytes", infile_size))
|
||||
|
||||
if infile_size > max_size then
|
||||
io.stderr:write(string.format("Error: Input file %s longer than %d bytes\n", args[1], max_size))
|
||||
os.exit(1)
|
||||
end
|
||||
-- print(string.format("Input file: %s", args[1]))
|
||||
-- print(string.format("Output file: %s", args[2]))
|
||||
|
||||
-- PS1 executables have a maximum size limit of 2MB
|
||||
local max_size = 0x200000
|
||||
-- print(string.format("\nChecking input file size (max: %d bytes)...", max_size))
|
||||
|
||||
local infile_size = file_size(args[1])
|
||||
if not infile_size then
|
||||
io.stderr:write("Error: Cannot open input file " .. args[1] .. "\n")
|
||||
os.exit(1)
|
||||
end
|
||||
|
||||
-- print(string.format("Input file size: %d bytes", infile_size))
|
||||
|
||||
if infile_size > max_size then
|
||||
io.stderr:write(string.format("Error: Input file %s longer than %d bytes\n", args[1], max_size))
|
||||
os.exit(1)
|
||||
end
|
||||
|
||||
-- print("\nOpening files...")
|
||||
local ofile = io.open(args[2], "wb")
|
||||
if not ofile then
|
||||
io.stderr:write("Error: Cannot open output file " .. args[2] .. "\n")
|
||||
os.exit(1)
|
||||
end
|
||||
|
||||
local ifile = io.open(args[1], "rb")
|
||||
if not ifile then
|
||||
io.stderr:write("Error: Cannot open input file " .. args[1] .. "\n")
|
||||
os.exit(1)
|
||||
end
|
||||
-- print("\nOpening files...")
|
||||
local ofile = io.open(args[2], "wb")
|
||||
if not ofile then
|
||||
io.stderr:write("Error: Cannot open output file " .. args[2] .. "\n")
|
||||
os.exit(1)
|
||||
end
|
||||
|
||||
local ifile = io.open(args[1], "rb")
|
||||
if not ifile then
|
||||
io.stderr:write("Error: Cannot open input file " .. args[1] .. "\n")
|
||||
os.exit(1)
|
||||
end
|
||||
|
||||
-- PS1 executables start with "PS-X EXE" magic string
|
||||
-- print("Writing PS-X executable header...")
|
||||
ofile:write("PS-X EXE")
|
||||
|
||||
-- Write entry point address (where the PS1 will jump to start execution)
|
||||
-- 0x80010000 is a standard entry point in PS1 RAM
|
||||
ofile:seek("set", 0x10)
|
||||
ofile:write(string.pack("<I4", 0x80010000))
|
||||
-- print(" Entry point: 0x80010000")
|
||||
|
||||
-- Initial GP/R28 register value (Global Pointer for data addressing)
|
||||
-- 0xFFFFFFFF means it will be set by crt0.S startup code
|
||||
ofile:write(string.pack("<I4", 0xFFFFFFFF))
|
||||
|
||||
-- Destination address in RAM where the executable will be loaded
|
||||
ofile:write(string.pack("<I4", 0x80010000))
|
||||
-- print(" Load address: 0x80010000")
|
||||
|
||||
-- Initial stack pointer (SP/R29) and frame pointer (FP/R30)
|
||||
-- 0x801FFF00 points near the top of the 2MB main RAM
|
||||
ofile:seek("set", 0x30)
|
||||
ofile:write(string.pack("<I4", 0x801FFF00))
|
||||
-- print(" Stack pointer: 0x801FFF00")
|
||||
|
||||
-- PS1 executables have an 0x800 (2048) byte header
|
||||
-- Zero fill the rest of the header
|
||||
ofile:seek("set", 0x800)
|
||||
-- print(" Header padding complete (2048 bytes)")
|
||||
-- PS1 executables start with "PS-X EXE" magic string
|
||||
-- print("Writing PS-X executable header...")
|
||||
ofile:write("PS-X EXE")
|
||||
|
||||
-- Write entry point address (where the PS1 will jump to start execution)
|
||||
-- 0x80010000 is a standard entry point in PS1 RAM
|
||||
ofile:seek("set", 0x10)
|
||||
ofile:write(string.pack("<I4", 0x80010000))
|
||||
-- print(" Entry point: 0x80010000")
|
||||
|
||||
-- Initial GP/R28 register value (Global Pointer for data addressing)
|
||||
-- 0xFFFFFFFF means it will be set by crt0.S startup code
|
||||
ofile:write(string.pack("<I4", 0xFFFFFFFF))
|
||||
|
||||
-- Destination address in RAM where the executable will be loaded
|
||||
ofile:write(string.pack("<I4", 0x80010000))
|
||||
-- print(" Load address: 0x80010000")
|
||||
|
||||
-- Initial stack pointer (SP/R29) and frame pointer (FP/R30)
|
||||
-- 0x801FFF00 points near the top of the 2MB main RAM
|
||||
ofile:seek("set", 0x30)
|
||||
ofile:write(string.pack("<I4", 0x801FFF00))
|
||||
-- print(" Stack pointer: 0x801FFF00")
|
||||
|
||||
-- PS1 executables have an 0x800 (2048) byte header
|
||||
-- Zero fill the rest of the header
|
||||
ofile:seek("set", 0x800)
|
||||
-- print(" Header padding complete (2048 bytes)")
|
||||
|
||||
-- Copy the actual program binary data after the header
|
||||
-- print("\nCopying program data...")
|
||||
local buffer_size = 0x2000 -- 8KB chunks for efficient copying
|
||||
local bytes_copied = 0
|
||||
|
||||
for i = 0, math.ceil(infile_size / buffer_size) - 1 do
|
||||
local buffer = ifile:read(buffer_size)
|
||||
if buffer then
|
||||
ofile:write(buffer)
|
||||
bytes_copied = bytes_copied + #buffer
|
||||
-- Show progress every 64KB
|
||||
if bytes_copied % 0x10000 == 0 or bytes_copied == infile_size then
|
||||
print(string.format(" Copied %d/%d bytes (%.1f%%)",
|
||||
bytes_copied, infile_size, (bytes_copied / infile_size) * 100))
|
||||
end
|
||||
end
|
||||
end
|
||||
-- Copy the actual program binary data after the header
|
||||
-- print("\nCopying program data...")
|
||||
local buffer_size = 0x2000 -- 8KB chunks for efficient copying
|
||||
local bytes_copied = 0
|
||||
|
||||
for i = 0, math.ceil(infile_size / buffer_size) - 1 do
|
||||
local buffer = ifile:read(buffer_size)
|
||||
if buffer then
|
||||
ofile:write(buffer)
|
||||
bytes_copied = bytes_copied + #buffer
|
||||
-- Show progress every 64KB
|
||||
if bytes_copied % 0x10000 == 0 or bytes_copied == infile_size then
|
||||
print(string.format(" Copied %d/%d bytes (%.1f%%)",
|
||||
bytes_copied, infile_size, (bytes_copied / infile_size) * 100))
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
-- PS1 executables must be padded to 0x800 (2048) byte boundaries
|
||||
print("\nAligning to 2048-byte boundary...")
|
||||
local exe_size = ofile:seek()
|
||||
if exe_size % 0x800 ~= 0 then
|
||||
local padding = 0x800 - (exe_size % 0x800)
|
||||
exe_size = exe_size + padding
|
||||
ofile:seek("set", exe_size - 1)
|
||||
ofile:write(string.pack("B", 0))
|
||||
print(string.format(" Added %d bytes of padding", padding))
|
||||
else
|
||||
print(" No padding needed")
|
||||
end
|
||||
-- PS1 executables must be padded to 0x800 (2048) byte boundaries
|
||||
print("\nAligning to 2048-byte boundary...")
|
||||
local exe_size = ofile:seek()
|
||||
if exe_size % 0x800 ~= 0 then
|
||||
local padding = 0x800 - (exe_size % 0x800)
|
||||
exe_size = exe_size + padding
|
||||
ofile:seek("set", exe_size - 1)
|
||||
ofile:write(string.pack("B", 0))
|
||||
print(string.format(" Added %d bytes of padding", padding))
|
||||
else
|
||||
print(" No padding needed")
|
||||
end
|
||||
|
||||
-- Write the size of the executable (excluding the 0x800 byte header)
|
||||
-- This goes at offset 0x1C in the header
|
||||
ofile:seek("set", 0x1C)
|
||||
ofile:write(string.pack("<I4", exe_size - 0x800))
|
||||
-- print(string.format("\nProgram size field set to: %d bytes", exe_size - 0x800))
|
||||
-- Write the size of the executable (excluding the 0x800 byte header)
|
||||
-- This goes at offset 0x1C in the header
|
||||
ofile:seek("set", 0x1C)
|
||||
ofile:write(string.pack("<I4", exe_size - 0x800))
|
||||
-- print(string.format("\nProgram size field set to: %d bytes", exe_size - 0x800))
|
||||
|
||||
ifile:close()
|
||||
ofile:close()
|
||||
|
||||
-- print(string.format("\nSuccess! PS1 executable created: %s", args[2]))
|
||||
print(string.format("Total file size: %d bytes\n", exe_size))
|
||||
ifile:close()
|
||||
ofile:close()
|
||||
|
||||
-- print(string.format("\nSuccess! PS1 executable created: %s", args[2]))
|
||||
print(string.format("Total file size: %d bytes\n", exe_size))
|
||||
end
|
||||
|
||||
-- Run main with command line arguments
|
||||
|
||||
+198
-46
@@ -24,6 +24,7 @@ $f_define = "-D"
|
||||
$f_include = "-I"
|
||||
$f_output = "-o"
|
||||
$f_std_c11 = "-std=c11"
|
||||
$f_std_c23 = "-std=c23"
|
||||
|
||||
# Warning Flags
|
||||
$f_wall = "-Wall"
|
||||
@@ -80,21 +81,21 @@ $path_psyq = join-path $path_toolchain 'psyq-4_7'
|
||||
$path_psyq_iwyu = join-path $path_toolchain 'psyq_iwyu'
|
||||
$path_psyq_imyu_inc = join-path $path_psyq_iwyu 'include'
|
||||
|
||||
function assemble-unit { param(
|
||||
function assemble-unit { param(
|
||||
[string] $unit,
|
||||
[string] $link_module,
|
||||
[string[]]$include_paths,
|
||||
[string[]]$user_assemble_args
|
||||
)
|
||||
$assemble_args = @(
|
||||
$f_arch_mips1,
|
||||
$f_arch_abi32,
|
||||
$f_arch_mips1,
|
||||
$f_arch_abi32,
|
||||
$f_arch_fp32,
|
||||
$f_arch_little_endian,
|
||||
$f_arch_little_endian,
|
||||
$f_arch_no_abicalls,
|
||||
$f_arch_no_pic,
|
||||
$f_arch_no_llsc,
|
||||
$f_arch_no_shared,
|
||||
$f_arch_no_pic,
|
||||
$f_arch_no_llsc,
|
||||
$f_arch_no_shared,
|
||||
$f_arch_no_stack_prot
|
||||
)
|
||||
$assemble_args += $f_no_stdlib
|
||||
@@ -129,16 +130,17 @@ function compile-unit { param(
|
||||
$compile_args += $f_no_strict_alias
|
||||
$compile_args += @(
|
||||
$f_arch_mips1,
|
||||
$f_arch_abi32,
|
||||
$f_arch_abi32,
|
||||
$f_arch_fp32,
|
||||
$f_arch_little_endian,
|
||||
$f_arch_no_abicalls,
|
||||
$f_arch_no_gpopt,
|
||||
$f_arch_no_pic,
|
||||
$f_arch_no_llsc,
|
||||
$f_arch_no_shared,
|
||||
$f_arch_no_abicalls,
|
||||
$f_arch_no_gpopt,
|
||||
$f_arch_no_pic,
|
||||
$f_arch_no_llsc,
|
||||
$f_arch_no_shared,
|
||||
$f_arch_no_stack_prot
|
||||
)
|
||||
$compile_args += $f_std_c11
|
||||
$compile_args += ($f_include + $path_psyq_imyu_inc)
|
||||
$compile_args += ($f_include + $path_nugget)
|
||||
|
||||
@@ -152,11 +154,7 @@ function compile-unit { param(
|
||||
& $Compiler $compile_args
|
||||
if ($LASTEXITCODE -ne 0) { write-error "Compilation failed for $unit. Aborting."; exit 1 }
|
||||
}
|
||||
function link-modules { param(
|
||||
[string[]]$link_modules,
|
||||
[string] $elf,
|
||||
[string[]]$user_link_args
|
||||
)
|
||||
function link-modules { param([string[]]$link_modules, [string] $elf, [string[]]$user_link_args)
|
||||
$link_args = @()
|
||||
|
||||
$link_args += $f_no_stdlib
|
||||
@@ -183,19 +181,19 @@ function link-modules { param(
|
||||
|
||||
$link_args += ($f_link_pass_through_prefix + $f_link_start_group)
|
||||
$libraries = @(
|
||||
"api",
|
||||
"c",
|
||||
"c2",
|
||||
"card",
|
||||
"cd",
|
||||
"comb",
|
||||
"ds",
|
||||
"etc",
|
||||
"gpu",
|
||||
"gs",
|
||||
"gte",
|
||||
"gun",
|
||||
"hmd",
|
||||
"api",
|
||||
"c",
|
||||
"c2",
|
||||
"card",
|
||||
"cd",
|
||||
"comb",
|
||||
"ds",
|
||||
"etc",
|
||||
"gpu",
|
||||
"gs",
|
||||
"gte",
|
||||
"gun",
|
||||
"hmd",
|
||||
"math",
|
||||
"mcrd",
|
||||
"mcx",
|
||||
@@ -224,10 +222,7 @@ function link-modules { param(
|
||||
& mipsel-none-elf-objdump.exe -W $elf >> $dasm
|
||||
if ($LASTEXITCODE -ne 0) { write-error "Linking failed. Aborting."; exit 1 }
|
||||
}
|
||||
function make-binary { param(
|
||||
[string]$elf,
|
||||
[string]$exe
|
||||
)
|
||||
function make-binary { param([string]$elf, [string]$exe)
|
||||
Write-Host "--- Creating Binary ---" -ForegroundColor Cyan
|
||||
write-host "Converting $elf to PS-EXE -> '$exe'"
|
||||
$objcopy_args = ($f_objcopy_format + "binary"), $elf, $exe
|
||||
@@ -235,6 +230,52 @@ function make-binary { param(
|
||||
if ($LASTEXITCODE -ne 0) { Write-Error "Objcopy failed. Aborting."; exit 1 }
|
||||
}
|
||||
|
||||
function ps1-meta { param(
|
||||
[string]$unity_root,
|
||||
[string[]]$sources,
|
||||
[Parameter(Mandatory=$true)][string]$metadata,
|
||||
[string]$out_root = (join-path $path_build 'gen'),
|
||||
[string[]]$passes = @('--pre-link'),
|
||||
[string[]]$extra_args = @()
|
||||
)
|
||||
# `--unity-root` and `--source` are
|
||||
# mutually exclusive. Exactly one of `$unity_root` / `$sources` must
|
||||
# be supplied; the other must be absent.
|
||||
if ($null -ne $unity_root -and $unity_root -ne '')
|
||||
{
|
||||
if ($null -ne $sources -and $sources.Count -gt 0) {
|
||||
write-error 'ps1-meta: -unity_root and -sources are mutually exclusive'
|
||||
exit 2
|
||||
}
|
||||
}
|
||||
elseif ($null -eq $sources -or $sources.Count -eq 0) {
|
||||
write-error 'ps1-meta: either -unity_root <file> or -sources <file...> is required'
|
||||
exit 2
|
||||
}
|
||||
|
||||
$script = join-path $path_scripts 'ps1_meta.lua'
|
||||
$input_summary = if ($null -ne $unity_root -and $unity_root -ne '') {
|
||||
"unity=$unity_root"
|
||||
}
|
||||
else {
|
||||
"$($sources.Count) source(s)"
|
||||
}
|
||||
write-host "ps1-meta $input_summary, passes=$($passes -join ',')" ` -ForegroundColor Magenta
|
||||
|
||||
$arg_list = @($passes) + @('--metadata', $metadata) + @('--out-root', $out_root) + @($extra_args)
|
||||
if ($null -ne $unity_root -and $unity_root -ne '') {
|
||||
$arg_list += @('--unity-root', $unity_root)
|
||||
}
|
||||
else {
|
||||
foreach ($s in $sources) { $arg_list += @('--source', $s) }
|
||||
}
|
||||
& luajit $script @arg_list
|
||||
if ($LASTEXITCODE -ne 0) {
|
||||
write-error "ps1-meta failed (exit $LASTEXITCODE). Aborting."
|
||||
exit $LASTEXITCODE
|
||||
}
|
||||
}
|
||||
|
||||
function build-hello_psyqo {
|
||||
$includes += @()
|
||||
|
||||
@@ -279,7 +320,7 @@ function build-graphis_hello {
|
||||
|
||||
$src_asm_crt = join-path $path_nugget_common 'crt0/crt0.s'
|
||||
$module_asm_crt = join-path $path_build 'crt0.o'
|
||||
# assemble-unit $src_asm_crt $module_asm_crt $includes $assemble_args
|
||||
assemble-unit $src_asm_crt $module_asm_crt $includes $assemble_args
|
||||
|
||||
$src_asm = join-path $path_module 'hello_gpu.s'
|
||||
$module_asm = join-path $path_build 'hello_gpu.o'
|
||||
@@ -312,7 +353,13 @@ function build-graphis_hello {
|
||||
function build-gte_hello {
|
||||
$includes += @()
|
||||
|
||||
$path_module = join-path $path_code 'gte_hello'
|
||||
$path_module = join-path $path_code 'gte_hello'
|
||||
$path_duffle = join-path $path_code 'duffle'
|
||||
$path_atom_metadata = join-path $path_duffle 'word_count.metadata.h'
|
||||
$path_build_gen = join-path $path_build 'gen'
|
||||
|
||||
$src_c = join-path $path_module 'hello_gte.c'
|
||||
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen
|
||||
|
||||
$assemble_args = @()
|
||||
$assemble_args += $f_debug
|
||||
@@ -321,21 +368,20 @@ function build-gte_hello {
|
||||
|
||||
$src_asm_crt = join-path $path_nugget_common 'crt0/crt0.s'
|
||||
$module_asm_crt = join-path $path_build 'crt0.o'
|
||||
# assemble-unit $src_asm_crt $module_asm_crt $includes $assemble_args
|
||||
assemble-unit $src_asm_crt $module_asm_crt $includes $assemble_args
|
||||
|
||||
$src_asm = join-path $path_module 'hello_gte.s'
|
||||
$module_asm = join-path $path_build 'hello_gte.o'
|
||||
# $src_asm = join-path $path_module 'hello_gte.s'
|
||||
# $module_asm = join-path $path_build 'hello_gte.o'
|
||||
|
||||
assemble-unit $src_asm $module_asm $includes $assemble_args
|
||||
# assemble-unit $src_asm $module_asm $includes $assemble_args
|
||||
|
||||
$src_c = join-path $path_module 'hello_gte.c'
|
||||
$module_c = join-path $path_build 'hello_gte_c.o'
|
||||
|
||||
$compile_args = @()
|
||||
$compile_args += $f_debug
|
||||
# $compile_args += $f_optimize_none
|
||||
$compile_args += $f_optimize_none
|
||||
# $compile_args += $f_optimize_intrinsics
|
||||
$compile_args += $f_optimize_size
|
||||
# $compile_args += $f_optimize_size
|
||||
# $compile_args += $f_optimize_debug
|
||||
$compile_args += ($f_include + $path_code)
|
||||
compile-unit $src_c $module_c $includes $compile_args
|
||||
@@ -346,7 +392,113 @@ function build-gte_hello {
|
||||
$link_args = @()
|
||||
$link_args += $f_debug
|
||||
# $link_args += $f_optimize_size
|
||||
link-modules @($module_asm_crt, $module_asm, $module_c) $elf $link_args
|
||||
$link_modules = @(
|
||||
$module_asm_crt,
|
||||
$module_c
|
||||
)
|
||||
link-modules $link_modules $elf $link_args
|
||||
make-binary $elf $exe
|
||||
|
||||
# Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start).
|
||||
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
|
||||
|
||||
$dwarfLineBin = join-path $path_build_gen 'hello_gte.dwarf_line.bin'
|
||||
$dwarfArangesBin = join-path $path_build_gen 'hello_gte.dwarf_aranges.bin'
|
||||
$dwarfRnglistsBin = join-path $path_build_gen 'hello_gte.dwarf_rnglists.bin'
|
||||
$injectElf = join-path $path_build 'hello_gte.dwarf-injected.elf'
|
||||
if ((Test-Path $dwarfLineBin) -and (Test-Path $dwarfArangesBin) -and (Test-Path $dwarfRnglistsBin))
|
||||
{
|
||||
Write-Host "[build] DWARF-injecting $elf -> $injectElf"
|
||||
Copy-Item -LiteralPath $elf -Destination $injectElf -Force
|
||||
# Objcopy call: 3x --update-section for (line, aranges, rnglists).
|
||||
$f_args = @(
|
||||
"--update-section=.debug_line=$dwarfLineBin",
|
||||
"--update-section=.debug_aranges=$dwarfArangesBin",
|
||||
"--update-section=.debug_rnglists=$dwarfRnglistsBin"
|
||||
)
|
||||
& $Objcopy @f_args $injectElf 2>&1 | Out-Null
|
||||
if ($LASTEXITCODE -ne 0) {
|
||||
Write-Warning "[build] objcopy F' splice failed (exit $LASTEXITCODE); removing $injectElf"
|
||||
Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue
|
||||
return;
|
||||
}
|
||||
|
||||
$dwarfInfoBin = join-path $path_build_gen 'hello_gte.dwarf_info.bin'
|
||||
$dwarfAbbrevBin = join-path $path_build_gen 'hello_gte.dwarf_abbrev.bin'
|
||||
$dwarfStrBin = join-path $path_build_gen 'hello_gte.dwarf_str.bin'
|
||||
$dwarfLocBin = join-path $path_build_gen 'hello_gte.dwarf_loc.bin'
|
||||
$dwarfLoclistsBin = join-path $path_build_gen 'hello_gte.dwarf_loclists.bin'
|
||||
$g_args = @(
|
||||
"--update-section=.debug_info=$dwarfInfoBin",
|
||||
"--update-section=.debug_abbrev=$dwarfAbbrevBin",
|
||||
"--update-section=.debug_str=$dwarfStrBin",
|
||||
"--add-section=.debug_loc=$dwarfLocBin",
|
||||
"--add-section=.debug_loclists=$dwarfLoclistsBin"
|
||||
)
|
||||
& $Objcopy @g_args $injectElf 2>&1 | Out-Null
|
||||
if ($LASTEXITCODE -ne 0) {
|
||||
Write-Warning "[build] objcopy G' splice failed (exit $LASTEXITCODE); removing $injectElf"
|
||||
Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue
|
||||
return;
|
||||
}
|
||||
|
||||
# Baked atoms execute from RAM but are emitted as C data arrays, so their ELF sections lack SHF_EXECINSTR.
|
||||
# GDB discards line rows for non-code sections. Mark only the debug-copy sections executable.
|
||||
# The original ELF and PS-EXE remain byte/flag unchanged.
|
||||
& $Objcopy `
|
||||
--set-section-flags ".rodata=alloc,load,readonly,code,contents" `
|
||||
--set-section-flags ".data=alloc,load,data,code,contents" `
|
||||
$injectElf 2>&1 | Out-Null
|
||||
if ($LASTEXITCODE -ne 0) {
|
||||
Write-Warning "[build] atom-section flag update failed (exit $LASTEXITCODE); removing $injectElf"
|
||||
Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue
|
||||
}
|
||||
else {
|
||||
Write-Host "[build] DWARF-injected ELF: $injectElf"
|
||||
}
|
||||
}
|
||||
}
|
||||
build-gte_hello
|
||||
build-gte_hello
|
||||
|
||||
|
||||
# NO idea if this works yet...
|
||||
function Send-ToEmulator { param( [string]$exePath )
|
||||
$uri = "http://localhost:8080/api/v1/load-exec"
|
||||
|
||||
# Absolute path is safest for the emulator web server
|
||||
$absolutePath = [System.IO.Path]::GetFullPath($exePath)
|
||||
|
||||
# Create JSON payload pointing to your compiled .ps-exe
|
||||
$body = @{ filename = $absolutePath } | ConvertTo-Json
|
||||
|
||||
Write-Host "Pushing hot-reload to PCSX-Redux..." -ForegroundColor Magenta
|
||||
try {
|
||||
$response = Invoke-RestMethod -Uri $uri -Method Post -Body $body -ContentType "application/json"
|
||||
Write-Host "Hot-reload successful!" -ForegroundColor Green
|
||||
} catch {
|
||||
Write-Warning "Could not connect to PCSX-Redux web server. Ensure the emulator is running and Web Server is enabled."
|
||||
}
|
||||
}
|
||||
|
||||
# # Automatically hot-reloads it into the running emulator
|
||||
# Send-ToEmulator (join-path $path_build 'hello_gte.ps-exe')
|
||||
|
||||
# --- Hot Reload via PCSX-Redux Web Server ---
|
||||
# $exe_path = join-path $path_build 'hello_gte.ps-exe'
|
||||
# $absolute_path = [System.IO.Path]::GetFullPath($exe_path)
|
||||
|
||||
# PCSX-Redux expects the file location in the URL query string?
|
||||
# We URL-encode the path to ensure backslashes and spaces don't break the HTTP request?
|
||||
# $encoded_path = [uri]::EscapeDataString($absolute_path)
|
||||
# $uri = "http://localhost:8080/api/v1/load-exec?path=$encoded_path"
|
||||
|
||||
# Write-Host "Pushing hot-reload to PCSX-Redux..." -ForegroundColor Magenta
|
||||
# try {
|
||||
# # Send the request with the query string included
|
||||
# Invoke-RestMethod -Uri $uri -Method Post
|
||||
# Write-Host "Hot-reload successful!" -ForegroundColor Green
|
||||
# } catch {
|
||||
# Write-Host "Failed to hot-reload." -ForegroundColor Red
|
||||
# # This will print the *actual* HTTP error instead of our generic warning
|
||||
# Write-Host $_.Exception.Message -ForegroundColor Yellow
|
||||
# }
|
||||
|
||||
+2597
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,92 @@
|
||||
--- duffle_paths.lua — Single-line bootstrap helper for the tape-atom Lua scripts.
|
||||
---
|
||||
--- Each entry script (ps1_meta.lua + the 7 passes/*.lua files) starts with one of:
|
||||
--- ```lua
|
||||
--- -- Entry script (ps1_meta.lua — `arg[0]` is set):
|
||||
--- local duffle = dofile((arg[0]:match("(.*[/\\])") or "./") .. "duffle_paths.lua")
|
||||
---
|
||||
--- -- Pass module (debug.getinfo path resolution; works both standalone and when require'd):
|
||||
--- local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||
--- local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
--- ```
|
||||
---
|
||||
--- That small bootstrap: (a) locates this helper via `arg[0]` / `debug.getinfo`,
|
||||
--- (b) loads it (which sets `package.path` + `package.cpath`),
|
||||
--- (c) at the bottom calls `require("duffle")` (now resolvable since `package.path` was just set) and returns the duffle M.
|
||||
--- Net effect: the caller gets the duffle module in one statement; no separate `dofile(...)` + `require("duffle")` dance.
|
||||
---
|
||||
|
||||
local M = {}
|
||||
|
||||
-- Cache key for the repo root. Stored in `package.loaded` (process-global) so all 8 entry scripts + passes scripts share one resolution.
|
||||
local CACHE_KEY = "__duffle_repo_root__"
|
||||
|
||||
--- Resolve the repo root from this script's own path. Zero shell spawn.
|
||||
--- `duffle_paths.lua` always lives at `<repo>/scripts/duffle_paths.lua`, so the repo root is the
|
||||
--- parent of the directory containing this script. We derive it directly from `debug.getinfo(1, "S").source`
|
||||
--- (returns `@<path>` for the currently-running chunk).
|
||||
---
|
||||
--- If `debug.getinfo` can't parse this script's path (shouldn't happen — dofile always populates source),
|
||||
--- return nil and let `M.setup()` fail loud.
|
||||
--- @return string|nil
|
||||
local function find_repo_root()
|
||||
if package.loaded[CACHE_KEY] then return package.loaded[CACHE_KEY] end
|
||||
|
||||
local source = debug.getinfo(1, "S").source
|
||||
-- Strip the leading `@` (Lua's dofile marker) and the trailing `/duffle_paths.lua` filename.
|
||||
-- What remains is the directory containing this script, i.e. `<repo>/scripts/`.
|
||||
local scripts_dir = source and source:match("^@?(.*)[/\\]duffle_paths%.lua$")
|
||||
if not scripts_dir then return nil end
|
||||
|
||||
-- The repo root is the parent of `scripts/`. Strip the trailing `scripts/` (with or without trailing slash).
|
||||
local root = scripts_dir:gsub("scripts[\\/]?$", "")
|
||||
root = root:gsub("\\", "/")
|
||||
if root == "" then root = "./" end
|
||||
if not root:match("/$") then root = root .. "/" end
|
||||
package.loaded[CACHE_KEY] = root
|
||||
return root
|
||||
end
|
||||
|
||||
--- Set `package.path` (for `require("duffle")` + `require("passes.X")`) and
|
||||
--- `package.cpath` (for `lpeg.dll`).
|
||||
---
|
||||
--- This script does NOT touch the OS environment: no `os.setenv`, no `os.putenv`, no `$PATH` mods.
|
||||
--- It just sets `package.path` and `package.cpath` (the standard Lua way to register module search dirs).
|
||||
--- lpeg is built by `update_deps.ps1` to `toolchain/lpeg/`,
|
||||
--- which we wire into `package.cpath` here (so `require("lpeg")` from `duffle.lua` resolves without any global state).
|
||||
function M.setup()
|
||||
local repo_root = find_repo_root()
|
||||
if not repo_root then
|
||||
-- Unreachable in practice: find_repo_root() derives the repo root from this script's
|
||||
-- own source path via debug.getinfo(1, "S").source (no subprocess, no git CLI, <1ms).
|
||||
-- A nil return means the source path did not match the expected
|
||||
-- <repo>/scripts/duffle_paths.lua layout — a packaging bug, not a "missing git repo"
|
||||
-- condition. os.exit(2) is retained so a real failure surfaces loud rather than
|
||||
-- silently producing an unconfigured module table.
|
||||
os.exit(2)
|
||||
end
|
||||
|
||||
local scripts_dir = repo_root .. "scripts/"
|
||||
local passes_dir = repo_root .. "scripts/passes/"
|
||||
package.path = scripts_dir .. "?.lua;"
|
||||
.. scripts_dir .. "?/init.lua;"
|
||||
.. passes_dir .. "?.lua;"
|
||||
.. passes_dir .. "?/init.lua;"
|
||||
.. package.path
|
||||
|
||||
-- lpeg: built by `update_deps.ps1` to `toolchain/lpeg/lpeg.dll`.
|
||||
-- lfs: compiled from pcsx-redux's vendored luafilesystem source to `toolchain/lfs/lfs.dll`.
|
||||
-- Wire both directories into cpath so `require("lpeg")` and `require("lfs")` resolve.
|
||||
local lpeg_dir = repo_root .. "toolchain/lpeg/"
|
||||
local lfs_dir = repo_root .. "toolchain/lfs/"
|
||||
package.cpath = lpeg_dir .. "?.dll;"
|
||||
.. lfs_dir .. "?.dll;"
|
||||
.. package.cpath
|
||||
end
|
||||
|
||||
-- Run the setup as a side effect.
|
||||
M.setup()
|
||||
|
||||
-- Now that package.path includes scripts/, `require("duffle")` resolves. Return the duffle module
|
||||
-- so callers can do `local duffle = dofile(...duffle_paths.lua)` in one line.
|
||||
return require("duffle")
|
||||
@@ -0,0 +1,794 @@
|
||||
--- elf_dwarf.lua — ELF32 + DWARF + atoms source-map utilities.
|
||||
--- All ELF32 + DWARF-specific code lives here.
|
||||
---
|
||||
--- **What this module contains:**
|
||||
--- - **Format-constant tables** (the byte-offset / opcode / size encyclopedias for ELF32, DWARF4 aranges, DWARF5 rnglists, DWARF line-program, MIPS).
|
||||
--- Every constant carries a spec:` comment naming the spec section that defines it.
|
||||
--- - **I/O helpers**: little-endian byte read/write, ELF32 section walker, nm symbol reader, source-map parser, native directory glob.
|
||||
---
|
||||
--- **Conventions:** tabs (1/level), EmmyLua annotations, no regex,
|
||||
--- Lua 5.3 compatible.
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Native dependencies
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- lfs is wired into package.cpath by `duffle_paths.lua` (vendored under `toolchain/lfs/lfs.dll`).
|
||||
local lfs = require("lfs")
|
||||
|
||||
local M = {}
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- DWARF tag + form constants
|
||||
-- ════════════════════════════════════════════
|
||||
-- (DWARF5 §7.5.5 "Tag Encodings" + Table 7.1; gcc emits these exact values for the DWARF3-extension and DWARF5 line units.)
|
||||
|
||||
M.DW_TAG = {
|
||||
compile_unit = 0x11,
|
||||
subprogram = 0x2E,
|
||||
variable = 0x34,
|
||||
structure_type = 0x13,
|
||||
member = 0x0D,
|
||||
base_type = 0x24,
|
||||
typedef = 0x2A,
|
||||
pointer_type = 0x0F,
|
||||
const_type = 0x26,
|
||||
volatile_type = 0x27,
|
||||
inlined_subroutine = 0x1D,
|
||||
-- We index the canonical gcc-emitted tags. Anything else falls through.
|
||||
}
|
||||
|
||||
M.DW_AT = {
|
||||
name = 0x03,
|
||||
low_pc = 0x11,
|
||||
high_pc = 0x12,
|
||||
language = 0x13,
|
||||
location = 0x02,
|
||||
comp_dir = 0x1B,
|
||||
byte_size = 0x0B,
|
||||
encoding = 0x3E,
|
||||
data_member_location = 0x38,
|
||||
type = 0x49,
|
||||
linkage_name = 0x6E,
|
||||
external = 0x3F,
|
||||
abstract_origin = 0x31,
|
||||
call_file = 0x58,
|
||||
call_line = 0x59,
|
||||
inline = 0x20,
|
||||
decl_file = 0x3A,
|
||||
decl_line = 0x3B,
|
||||
}
|
||||
|
||||
M.DW_FORM = {
|
||||
addr = 0x01,
|
||||
data1 = 0x0B,
|
||||
data2 = 0x05,
|
||||
data4 = 0x06,
|
||||
string = 0x08,
|
||||
strp = 0x0E,
|
||||
exprloc = 0x18,
|
||||
ref4 = 0x13,
|
||||
udata = 0x0F,
|
||||
ref_sig8 = 0x20,
|
||||
implicit_const = 0x21,
|
||||
flag_present = 0x19,
|
||||
sec_offset = 0x17,
|
||||
}
|
||||
|
||||
M.DW_ATE = {
|
||||
address = 0x01,
|
||||
boolean = 0x02,
|
||||
complex_float = 0x03,
|
||||
float = 0x04,
|
||||
signed = 0x05,
|
||||
signed_char = 0x06,
|
||||
unsigned = 0x07,
|
||||
unsigned_char = 0x08,
|
||||
}
|
||||
|
||||
-- DWARF5 §7.5.6 DW_FORM_implicit_const
|
||||
local DW_FORM_implicit_const = 0x21
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Format-constant tables
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- ----------------------------------------------------------------------------
|
||||
-- MIPS sizes
|
||||
-- ----------------------------------------------------------------------------
|
||||
|
||||
--- spec: MIPS o32 ABI §"Register Usage" — 32-bit general-purpose registers
|
||||
M.MIPS_BYTES_PER_WORD = 0x04
|
||||
|
||||
|
||||
-- ----------------------------------------------------------------------------
|
||||
-- ELF32 (System V ABI gABI v1.2)
|
||||
-- ----------------------------------------------------------------------------
|
||||
--- **Wire-offset contract:** format offsets, fixed-width reader offsets, LEB/parser cursors,
|
||||
--- and section-relative values are zero-based wire offsets. Only Lua string APIs receive
|
||||
--- a `+ 1` conversion at their boundary (`byte`, `sub`, and `find`).
|
||||
---
|
||||
--- ELF/DWARF field offsets are expressed in hex so they map directly to the
|
||||
--- zero-based byte positions in the binary file.
|
||||
|
||||
|
||||
--- spec: System V ABI gABI v1.2 §"ELF Header" (Table 1) + §"Section Header Table"
|
||||
M.ELF32 = {
|
||||
magic_offset = 0x00, -- 4-byte magic "\127ELF" at file offset 0x00
|
||||
magic = "\127ELF",
|
||||
class_offset = 0x04, -- 1-byte; 1 = ELF32, 2 = ELF64
|
||||
class_elf32 = 1,
|
||||
endian_offset = 0x05, -- 1-byte; 1 = little-endian, 2 = big-endian
|
||||
endian_little = 1,
|
||||
header_bytes = 0x34, -- spec: gABI v1.2 §"ELF Header" — ELF32 header is 52 bytes total
|
||||
e_shoff_offset = 0x20, -- 4-byte LE; section-header table file offset
|
||||
e_shentsize_offset = 0x2E, -- 2-byte LE; section-header entry size in bytes
|
||||
e_shnum_offset = 0x30, -- 2-byte LE; number of section headers
|
||||
e_shstrndx_offset = 0x32, -- 2-byte LE; index of section-name string table
|
||||
sh_size_bytes = 0x28, -- spec: gABI v1.2 §"Section Header Table" — each entry is 40 bytes
|
||||
sh_name_offset = 0x00, -- 4-byte LE; offset into .shstrtab
|
||||
sh_type_offset = 0x04, -- 4-byte LE; section type (SHT_*)
|
||||
sh_offset_offset = 0x10, -- 4-byte LE; section's file offset
|
||||
sh_size_offset = 0x14, -- 4-byte LE; section's size in bytes
|
||||
dw_dwarf32_terminator = 0xFFFFFFFF, -- spec: DWARF4 spec §7.4 — 32-bit DWARF initial-length terminator
|
||||
}
|
||||
|
||||
-- ----------------------------------------------------------------------------
|
||||
-- DWARF4 .debug_aranges (per DWARF5 spec §7.4 — Address Range Table)
|
||||
-- ----------------------------------------------------------------------------
|
||||
-- All offsets are zero-based wire offsets.
|
||||
|
||||
--- spec: DWARF5 spec §7.4 (Address Range Table) — 32-bit DWARF form
|
||||
M.DWARF4_ARANGES = {
|
||||
unit_length_offset = 0x00, -- 4-byte LE; length of unit body (excludes these 4 bytes)
|
||||
version_offset = 0x04, -- 2-byte LE; expected = 2
|
||||
cu_offset_offset = 0x06, -- 4-byte LE; CU DIE offset in .debug_info
|
||||
addr_size_offset = 0x0A, -- 1-byte; expected = 4 (32-bit MIPS)
|
||||
seg_size_offset = 0x0B, -- 1-byte; expected = 0
|
||||
entry_size = 0x08, -- 4-byte addr + 4-byte length (per §7.4)
|
||||
terminator_size = 0x08, -- 8 zero bytes (per §7.4 end-of-list marker)
|
||||
version_expected = 2,
|
||||
addr_size_expected = 4,
|
||||
seg_size_expected = 0,
|
||||
}
|
||||
|
||||
-- ----------------------------------------------------------------------------
|
||||
-- DWARF5 .debug_rnglists (per DWARF5 spec §2.17 + §7.21)
|
||||
-- ----------------------------------------------------------------------------
|
||||
-- All offsets are zero-based wire offsets.
|
||||
|
||||
--- spec: DWARF5 spec §2.17 + §7.21 (Range List Table) — 32-bit DWARF form
|
||||
M.DWARF5_RNGLISTS = {
|
||||
unit_length_offset = 0x00, -- 4-byte LE
|
||||
version_offset = 0x04, -- 2-byte LE; expected = 5
|
||||
addr_size_offset = 0x06, -- 1-byte; expected = 4
|
||||
seg_size_offset = 0x07, -- 1-byte; expected = 0
|
||||
offset_count_offset = 0x08, -- 4-byte LE; expected = 0
|
||||
first_entry_offset = 0x0C,
|
||||
end_of_list = 0x00, -- spec: DWARF5 §7.7 — DW_RLE_end_of_list byte value
|
||||
start_length = 0x07, -- spec: DWARF5 §7.7 — DW_RLE_start_length byte value
|
||||
version_expected = 5,
|
||||
addr_size_expected = 4,
|
||||
seg_size_expected = 0,
|
||||
offset_count_expected = 0,
|
||||
}
|
||||
|
||||
-- ----------------------------------------------------------------------------
|
||||
-- DWARF line-program opcodes (per DWARF5 spec §6.2.5)
|
||||
-- ----------------------------------------------------------------------------
|
||||
-- Opcode VALUES stay in decimal — they're identifiers (DW_LNS_copy = 1), not binary positions.
|
||||
-- Compare to the *_offset fields above which are hex.
|
||||
|
||||
--- spec: DWARF5 spec §6.2.5 (Line Number Program Opcodes)
|
||||
M.DWARF_LINE_OPS = {
|
||||
-- Standard opcodes (§6.2.5.2)
|
||||
DW_LNS_extended = 0, -- spec: §6.2.5.2 — extended opcode marker byte
|
||||
DW_LNS_copy = 1,
|
||||
DW_LNS_advance_pc = 2,
|
||||
DW_LNS_advance_line = 3,
|
||||
DW_LNS_set_file = 4,
|
||||
DW_LNS_negate_stmt = 6, -- spec: §6.2.5.2 — toggle the line-state is_stmt register
|
||||
-- Extended sub-opcodes (§6.2.5.3)
|
||||
DW_LNE_end_sequence = 1, -- spec: §6.2.5.3
|
||||
DW_LNE_set_address = 2, -- spec: §6.2.5.3
|
||||
-- Standard opcode header (§6.2.5.1)
|
||||
-- opcode_base + line_range are 1-byte header fields; hex so they map
|
||||
-- directly to their position in the line-program header byte sequence.
|
||||
-- line_base stays signed decimal (=-5) since 0xFB obscures the spec semantics.
|
||||
opcode_base = 0x0D,
|
||||
line_base = -5,
|
||||
line_range = 0x0E,
|
||||
-- Extended opcode payload sizes (include the sub-opcode byte; §6.2.5.3)
|
||||
-- Hex so they match the byte positions in the line-program wire format.
|
||||
end_sequence_payload_size = 0x01, -- size = sub_opcode only
|
||||
set_address_payload_size = 0x05, -- size = sub_opcode(1) + addr(4)
|
||||
}
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- I/O helpers: little-endian byte read/write
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Read a 4-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
|
||||
--- Equivalent to `string.unpack("<I4", buf, off + 1)` but avoids the table-return shape + works under LuaJIT 2.1
|
||||
--- (which has partial `string.unpack` coverage).
|
||||
---
|
||||
--- **Convention:** `off` is a zero-based wire offset; `+ 1` is applied only at the `string.byte` boundary.
|
||||
---
|
||||
--- **Byte weights** are written as `0x100`, `0x10000`, `0x1000000` (i.e. 2^8, 2^16, 2^24) so the LE byte positions are visually explicit:
|
||||
--- byte 0 contributes its value directly; byte 1 is shifted left by 8
|
||||
--- (= 0x100); byte 2 by 16 (= 0x10000); byte 3 by 24 (= 0x1000000).
|
||||
--- @param buf string
|
||||
--- @param off integer -- zero-based wire offset
|
||||
--- @return integer
|
||||
function M.read_u32_le(buf, off)
|
||||
local byte_off = off + 1
|
||||
return buf:byte(byte_off)
|
||||
+ buf:byte(byte_off + 0x01) * 0x00000100
|
||||
+ buf:byte(byte_off + 0x02) * 0x00010000
|
||||
+ buf:byte(byte_off + 0x03) * 0x01000000
|
||||
end
|
||||
|
||||
--- Read a 2-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
|
||||
--- (`off` is zero-based; `+ 1` is applied only at the `string.byte` boundary.)
|
||||
--- @param buf string
|
||||
--- @param off integer -- zero-based wire offset
|
||||
--- @return integer
|
||||
function M.read_u16_le(buf, off)
|
||||
local byte_off = off + 1
|
||||
return buf:byte(byte_off) + buf:byte(byte_off + 0x01) * 0x00000100
|
||||
end
|
||||
|
||||
-- Pure-Lua 5.3 LEB128 readers (no `bit` library). `2^shift` arithmetic matches the existing parser.
|
||||
-- Offsets are 0-based; returns (value, next_pos).
|
||||
-- Promoted from `local function` to M.* exports so passes/dwarf_injection.lua
|
||||
-- can import them as file-scope locals per the 2nd-caller lift precedent
|
||||
-- (the uleb128 + sleb128 encoders were promoted the same way).
|
||||
function M.read_uleb128_at(buf, pos)
|
||||
local value, shift = 0, 0
|
||||
local len = #buf
|
||||
while pos < len do
|
||||
local b = buf:byte(pos + 1)
|
||||
value = value + (b % 0x80) * (2 ^ shift)
|
||||
shift = shift + 7
|
||||
pos = pos + 1
|
||||
if b < 0x80 then return value, pos end
|
||||
end
|
||||
return nil, pos
|
||||
end
|
||||
|
||||
function M.read_sleb128_at(buf, pos)
|
||||
local value, shift = 0, 0
|
||||
local len = #buf
|
||||
while pos < len do
|
||||
local b = buf:byte(pos + 1)
|
||||
value = value + (b % 0x80) * (2 ^ shift)
|
||||
shift = shift + 7
|
||||
pos = pos + 1
|
||||
if b < 0x80 then
|
||||
if b >= 0x40 then value = value - (2 ^ shift) end
|
||||
return value, pos
|
||||
end
|
||||
end
|
||||
return nil, pos
|
||||
end
|
||||
|
||||
-- Find the 0-based offset of the table-terminator byte (a single 0) for the abbrev table starting at `table_start`.
|
||||
-- Returns nil on truncated input. Walks declaration headers
|
||||
-- (code, tag, has_children, attr/form pairs, DW_FORM_implicit_const constant) until it finds a 0 byte that follows a complete declaration.
|
||||
function M.find_abbrev_table_end(table_bytes, table_start)
|
||||
local pos, len = table_start, #table_bytes
|
||||
if pos >= len or table_bytes:byte(pos + 1) == 0 then return pos end
|
||||
while pos < len do
|
||||
local _code, code_end = M.read_uleb128_at(table_bytes, pos)
|
||||
if not _code then return nil end
|
||||
pos = code_end
|
||||
local _tag, tag_end = M.read_uleb128_at(table_bytes, pos)
|
||||
if not _tag then return nil end
|
||||
pos = tag_end
|
||||
if pos >= len then return nil end
|
||||
pos = pos + 1 -- has_children byte
|
||||
while pos < len do
|
||||
local attr, attr_end = M.read_uleb128_at(table_bytes, pos)
|
||||
if not attr then return nil end
|
||||
pos = attr_end
|
||||
local form, form_end = M.read_uleb128_at(table_bytes, pos)
|
||||
if not form then return nil end
|
||||
pos = form_end
|
||||
if attr == 0 and form == 0 then break end
|
||||
if form == DW_FORM_implicit_const then
|
||||
local _c, ce = M.read_sleb128_at(table_bytes, pos)
|
||||
if not _c then return nil end
|
||||
pos = ce
|
||||
end
|
||||
end
|
||||
if pos >= len then return nil end
|
||||
if table_bytes:byte(pos + 1) == 0 then return pos end
|
||||
end
|
||||
return nil
|
||||
end
|
||||
|
||||
-- Read the null-terminated C string at 0-based offset `off` in `buf`.
|
||||
-- Stops at the first 0 byte or end of buffer.
|
||||
local function read_c_string_at(buf, off)
|
||||
local len = #buf
|
||||
local start = off
|
||||
while off < len and buf:byte(off + 1) ~= 0 do off = off + 1 end
|
||||
return buf:sub(start + 1, off)
|
||||
end
|
||||
|
||||
-- Walk the .debug_abbrev table starting at 0-based offset `table_start` and return a list of declarations:
|
||||
-- {code, tag, has_children, attrs={ {name, form}, ... }}.
|
||||
-- Stops at the table terminator.
|
||||
local function parse_abbrev_table(table_bytes, table_start)
|
||||
local table_end = M.find_abbrev_table_end(table_bytes, table_start)
|
||||
if not table_end then return nil, "no terminator" end
|
||||
local decls = {}
|
||||
local pos = table_start
|
||||
while pos < table_end do
|
||||
local code, code_end = M.read_uleb128_at(table_bytes, pos)
|
||||
if not code then return nil, "truncated code" end
|
||||
pos = code_end
|
||||
local tag, tag_end = M.read_uleb128_at(table_bytes, pos)
|
||||
if not tag then return nil, "truncated tag" end
|
||||
pos = tag_end
|
||||
local has_children = table_bytes:byte(pos + 1)
|
||||
pos = pos + 1
|
||||
local attrs = {}
|
||||
while true do
|
||||
local attr, attr_end = M.read_uleb128_at(table_bytes, pos)
|
||||
if not attr then return nil, "truncated attr" end
|
||||
pos = attr_end
|
||||
local form, form_end = M.read_uleb128_at(table_bytes, pos)
|
||||
if not form then return nil, "truncated form" end
|
||||
pos = form_end
|
||||
if attr == 0 and form == 0 then break end
|
||||
attrs[#attrs + 1] = { name = attr, form = form }
|
||||
if form == DW_FORM_implicit_const then
|
||||
local _c, ce = M.read_sleb128_at(table_bytes, pos)
|
||||
if not _c then return nil, "truncated const" end
|
||||
pos = ce
|
||||
end
|
||||
end
|
||||
decls[#decls + 1] = { code = code, tag = tag, has_children = has_children, attrs = attrs }
|
||||
end
|
||||
return decls
|
||||
end
|
||||
|
||||
-- Read a ULEB attribute value at 0-based offset `pos` for the given `form`.
|
||||
-- Returns (value, next_pos). For DW_FORM_string we return the inline string.
|
||||
-- For DW_FORM_strp we return the inline string resolved from `str_buf`.
|
||||
-- For DW_FORM_ref4 we return the absolute CU-relative offset.
|
||||
-- The caller decides whether to interpret that as a section offset.
|
||||
local function read_form_value(buf, str_buf, pos, form)
|
||||
if form == M.DW_FORM.addr then
|
||||
return M.read_u32_le(buf, pos), pos + 4
|
||||
elseif form == M.DW_FORM.string then
|
||||
local s = read_c_string_at(buf, pos)
|
||||
return s, pos + #s + 1
|
||||
elseif form == M.DW_FORM.strp then
|
||||
-- DW_FORM_strp: 4-byte offset into .debug_str.
|
||||
local strp_off = M.read_u32_le(buf, pos)
|
||||
return read_c_string_at(str_buf, strp_off), pos + 4
|
||||
elseif form == M.DW_FORM.udata then return M.read_uleb128_at(buf, pos)
|
||||
elseif form == M.DW_FORM.data1 then return buf:byte(pos + 1), pos + 1
|
||||
elseif form == M.DW_FORM.data2 then return M.read_u16_le(buf, pos), pos + 2
|
||||
elseif form == M.DW_FORM.data4 then return M.read_u32_le(buf, pos), pos + 4
|
||||
elseif form == M.DW_FORM.ref4 then return M.read_u32_le(buf, pos), pos + 4
|
||||
elseif form == M.DW_FORM.sec_offset then
|
||||
-- DW_FORM_sec_offset: 4-byte offset (size depends on DWARF version;
|
||||
-- on DWARF5 32-bit it's always 4 bytes).
|
||||
return M.read_u32_le(buf, pos), pos + 4
|
||||
elseif form == M.DW_FORM.flag_present then
|
||||
return 1, pos
|
||||
elseif form == M.DW_FORM.exprloc then
|
||||
-- DW_FORM_exprloc: ULEB byte count + that many bytes of DW_OP_*.
|
||||
local len, ne = M.read_uleb128_at(buf, pos)
|
||||
if not len then return nil, pos end
|
||||
return nil, ne + len
|
||||
elseif form == DW_FORM_implicit_const then
|
||||
-- The constant is declared in the abbrev; no value bytes in the DIE.
|
||||
return nil, pos
|
||||
elseif form == M.DW_FORM.ref_sig8 then
|
||||
-- DW_FORM_ref_sig8 (DWARF5 §7.4.2): an 8-byte value identifying a type
|
||||
-- by signature. The low 4 bytes (LE) are the type signature (content hash);
|
||||
-- the high 4 bytes (LE) are a CU-relative offset into the matching type unit.
|
||||
-- Consumers use the low 4 to look up the type unit (see M.find_type_unit_by_signature)
|
||||
-- then the high 4 to resolve the specific type within it.
|
||||
-- Return the low 4 as the primary value to preserve the (value, next_pos) shape;
|
||||
-- the high 4 is exposed via M.read_ref_sig8 (which returns both halves).
|
||||
local _, _, next_pos = M.read_ref_sig8(buf, pos)
|
||||
return M.read_u32_le(buf, pos), next_pos
|
||||
else
|
||||
return nil, pos
|
||||
end
|
||||
end
|
||||
|
||||
--- Read a `DW_FORM_ref_sig8` value at 0-based offset `pos` from `buf`.
|
||||
--- Returns the low 4 bytes (LE) as `low`, the high 4 bytes (LE) as `high`, and the cursor position after the 8-byte value as `next_pos`.
|
||||
--- Callers that need the full type-unit + type-offset pair (e.g. to resolve a type identifier embedded as a signature)
|
||||
--- should use this directly rather than going through `read_form_value`,
|
||||
--- which only exposes the low 4 bytes to preserve its existing (value, next_pos) return shape.
|
||||
--- @param buf string
|
||||
--- @param pos integer -- zero-based wire offset
|
||||
--- @return integer -- low 4 bytes (LE), the type signature
|
||||
--- @return integer -- high 4 bytes (LE), the offset within the matching type unit
|
||||
--- @return integer -- cursor after the 8-byte value
|
||||
function M.read_ref_sig8(buf, pos)
|
||||
return M.read_u32_le(buf, pos), M.read_u32_le(buf, pos + 4), pos + 8
|
||||
end
|
||||
|
||||
-- DWARF5 §7.5.6 (Type Entries).
|
||||
-- Walk all units in `info` and return the 0-based offset of the first unit
|
||||
-- whose `DW_AT_type_signature` (8-byte value at the end of the unit header) equals `target_sig`.
|
||||
-- The signature is interpreted as two 32-bit halves (low/high) per the read_ref_sig8 contract;
|
||||
-- we match both halves (i.e. the 8-byte value as a whole). Returns nil if no matching unit exists.
|
||||
--
|
||||
-- Unit header layout (from pos 0):
|
||||
-- unit_length(4) + version(2) + unit_type(1) + address_size(1) + debug_abbrev_offset(4)
|
||||
-- followed by type_unit_specific fields: type_signature(8) + type_offset(4)
|
||||
-- The type_signature is at byte offset 8 of the body (right after debug_abbrev_offset).
|
||||
-- @param info string -- the .debug_info section bytes
|
||||
-- @param target_sig_lo integer -- low 4 bytes (LE) of the desired signature
|
||||
-- @param target_sig_hi integer -- high 4 bytes (LE) of the desired signature
|
||||
-- @return integer|nil, integer|nil -- unit offset, type_offset within the unit
|
||||
function M.find_type_unit_by_signature(info, target_sig_lo, target_sig_hi)
|
||||
local pos = 0
|
||||
local section_len = #info
|
||||
while pos + 4 < section_len do
|
||||
local unit_length = M.read_u32_le(info, pos)
|
||||
if unit_length == 0xFFFFFFFF then
|
||||
return nil, nil -- DWARF64 not supported
|
||||
end
|
||||
-- unit_length is the body size, NOT including the 4-byte unit_length field itself.
|
||||
local body_start = pos + 4
|
||||
local body_end = body_start + unit_length
|
||||
if body_end > section_len then
|
||||
return nil, nil -- malformed
|
||||
end
|
||||
-- Per DWARF5 §7.5.6, the type_unit (DW_UT_type = 0x02) body layout is:
|
||||
-- 0: version (2)
|
||||
-- 2: unit_type (1) -- DW_UT_type = 0x02
|
||||
-- 3: address_size (1)
|
||||
-- 4: debug_abbrev_offset (4)
|
||||
-- 8: type_signature (8)
|
||||
-- 16: type_offset (4)
|
||||
-- 20: <children>
|
||||
if body_end - body_start >= 20 then
|
||||
-- read_ref_sig8 / write_u32_le / etc. are 1-indexed (string:byte);
|
||||
-- pos / body_start / body_end are 0-based wire offsets, so the
|
||||
-- 1-indexed byte at 0-based wire offset X is string:byte(X + 1).
|
||||
-- Per DWARF5 §7.5.6, the type_unit body is laid out as:
|
||||
-- byte 0-1: version (2)
|
||||
-- byte 2: unit_type (1) -- DW_UT_type = 0x02
|
||||
-- byte 3: address_size (1)
|
||||
-- byte 4-7: debug_abbrev_offset (4)
|
||||
-- byte 8-15: type_signature (8)
|
||||
-- byte 16-19: type_offset (4)
|
||||
local unit_type = info:byte(body_start + 2 + 1) -- 0-based +2 = unit_type in 1-indexed
|
||||
if unit_type == 0x02 then -- DW_UT_type
|
||||
local sig_lo, sig_hi, _ = M.read_ref_sig8(info, body_start + 8) -- 0-based +8 = type_signature in 1-indexed
|
||||
if sig_lo == target_sig_lo and sig_hi == target_sig_hi then
|
||||
local type_offset = M.read_u32_le(info, body_start + 16) -- 0-based +16 = type_offset in 1-indexed
|
||||
return pos, type_offset
|
||||
end
|
||||
end
|
||||
end
|
||||
-- Advance to the next unit (the 4-byte unit_length + the body).
|
||||
pos = body_end
|
||||
end
|
||||
return nil, nil
|
||||
end
|
||||
|
||||
--- Return a 4-byte little-endian byte string for `value`.
|
||||
--- Caller concatenates with `..` if composing multi-word blobs.
|
||||
--- **Byte weights** written as `0x100` etc. (see `M.read_u32_le` for rationale).
|
||||
--- @param value integer -- 0 ≤ value ≤ 0xFFFFFFFF
|
||||
--- @return string
|
||||
function M.write_u32_le(value)
|
||||
return string.char(
|
||||
value % 0x00000100,
|
||||
math.floor(value / 0x00000100) % 0x00000100,
|
||||
math.floor(value / 0x00010000) % 0x00000100,
|
||||
math.floor(value / 0x01000000) % 0x00000100)
|
||||
end
|
||||
|
||||
--- Return a 2-byte little-endian byte string for `value`.
|
||||
--- @param value integer -- 0 ≤ value ≤ 0xFFFF
|
||||
--- @return string
|
||||
function M.write_u16_le(value)
|
||||
return string.char(value % 0x00000100, math.floor(value / 0x00000100) % 0x00000100)
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- I/O helpers: ELF32 / DWARF / symbols
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Read the named sections from a post-link ELF32 by walking the ELF32 section-header table directly
|
||||
--- (no subprocess; lfs only for the existence check). Returns `{[name] = bytes_or_empty_string, ...}`.
|
||||
---
|
||||
--- **Convention:** ELF/DWARF offsets are zero-based wire offsets. Direct Lua string APIs add `+ 1` at the boundary.
|
||||
--- Every requested name has an entry in the returned dict;
|
||||
--- missing sections have an empty string (NOT nil) so callers can do `sections[".debug_x"] or ""` for the missing case.
|
||||
---
|
||||
--- **Cost:** one file open + one `f:seek` + one `f:read` per section header
|
||||
--- (we walk all `e_shnum` headers regardless of how many names are requested, to find the .shstrtab first).
|
||||
--- For frequent callers, pass the union of all needed sections in one call.
|
||||
-- Can add `.debug_info` + `.debug_loc` + `.debug_str_offsets` to the list without writing a 2nd ELF walker.
|
||||
--- @param elf_path Path
|
||||
--- @param section_names string[] -- list of section names to read
|
||||
--- @return table<string, string>
|
||||
function M.read_elf_sections(elf_path, section_names)
|
||||
-- Initialize result with all requested names set to "" so callers can do `sections[X]
|
||||
-- or ""` for missing sections without nil-checks.
|
||||
local result = {}
|
||||
for _, name in ipairs(section_names) do result[name] = "" end
|
||||
|
||||
-- O(1) lookup set.
|
||||
local wanted = {}
|
||||
for _, name in ipairs(section_names) do wanted[name] = true end
|
||||
|
||||
-- Existence check (lfs.attributes avoids an io.open-vs-fail race).
|
||||
if lfs.attributes(elf_path, "mode") ~= "file" then
|
||||
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] ELF not found: %s\n", elf_path))
|
||||
return result
|
||||
end
|
||||
|
||||
local f = io.open(elf_path, "rb")
|
||||
if not f then
|
||||
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] io.open failed: %s\n", elf_path))
|
||||
return result
|
||||
end
|
||||
|
||||
-- Read the ELF32 header.
|
||||
local header = f:read(M.ELF32.header_bytes)
|
||||
if not header or #header < M.ELF32.header_bytes then
|
||||
io.stderr:write("[elf_dwarf.read_elf_sections] ELF too small for ELF32 header\n")
|
||||
f:close()
|
||||
return result
|
||||
end
|
||||
|
||||
-- Sanity-check magic + class + endianness.
|
||||
if header:sub(M.ELF32.magic_offset + 1, M.ELF32.magic_offset + 0x04) ~= M.ELF32.magic then
|
||||
io.stderr:write("[elf_dwarf.read_elf_sections] not an ELF file\n")
|
||||
f:close()
|
||||
return result
|
||||
end
|
||||
if header:byte(M.ELF32.class_offset + 1) ~= M.ELF32.class_elf32 then
|
||||
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] not ELF32 (class=%d)\n", header:byte(M.ELF32.class_offset + 1)))
|
||||
f:close()
|
||||
return result
|
||||
end
|
||||
if header:byte(M.ELF32.endian_offset + 1) ~= M.ELF32.endian_little then
|
||||
io.stderr:write("[elf_dwarf.read_elf_sections] not little-endian; unsupported\n")
|
||||
f:close()
|
||||
return result
|
||||
end
|
||||
|
||||
-- Parse section-header table location + dimensions from the header.
|
||||
local e_shoff = M.read_u32_le(header, M.ELF32.e_shoff_offset)
|
||||
local e_shentsize = M.read_u16_le(header, M.ELF32.e_shentsize_offset)
|
||||
local e_shnum = M.read_u16_le(header, M.ELF32.e_shnum_offset)
|
||||
local e_shstrndx = M.read_u16_le(header, M.ELF32.e_shstrndx_offset)
|
||||
|
||||
-- Read the section-header string table (.shstrtab) so we can resolve section names from their `sh_name` offsets.
|
||||
f:seek("set", e_shoff + e_shstrndx * e_shentsize)
|
||||
local strtab_hdr = f:read(e_shentsize)
|
||||
if not strtab_hdr or #strtab_hdr < e_shentsize then
|
||||
io.stderr:write("[elf_dwarf.read_elf_sections] could not read .shstrtab header\n")
|
||||
f:close()
|
||||
return result
|
||||
end
|
||||
local strtab_offset = M.read_u32_le(strtab_hdr, M.ELF32.sh_offset_offset)
|
||||
local strtab_size = M.read_u32_le(strtab_hdr, M.ELF32.sh_size_offset)
|
||||
f:seek("set", strtab_offset)
|
||||
local strtab = f:read(strtab_size) or ""
|
||||
|
||||
-- Walk all section headers; collect (offset, size) for the wanted names.
|
||||
local function read_section_bytes(sh_offset, sh_size)
|
||||
f:seek("set", sh_offset)
|
||||
return f:read(sh_size) or ""
|
||||
end
|
||||
|
||||
for sh_idx = 0, e_shnum - 1 do
|
||||
f:seek("set", e_shoff + sh_idx * e_shentsize)
|
||||
local sh = f:read(e_shentsize)
|
||||
if not sh or #sh < e_shentsize then break end
|
||||
local sh_name = M.read_u32_le(sh, M.ELF32.sh_name_offset)
|
||||
local sh_offset = M.read_u32_le(sh, M.ELF32.sh_offset_offset)
|
||||
local sh_size = M.read_u32_le(sh, M.ELF32.sh_size_offset)
|
||||
|
||||
-- Extract the name (null-terminated C string in strtab).
|
||||
local name_end = strtab:find("\0", sh_name + 1, true) or (sh_name + 1)
|
||||
local name = strtab:sub(sh_name + 1, name_end - 1)
|
||||
if wanted[name] then
|
||||
result[name] = read_section_bytes(sh_offset, sh_size)
|
||||
end
|
||||
end
|
||||
|
||||
f:close()
|
||||
return result
|
||||
end
|
||||
|
||||
--- Read ELF symbol addresses by walking the `.symtab` + `.strtab` sections directly (no `nm` subprocess).
|
||||
--- Returns a map `{name -> {addr, size_bytes}}` for every `code_<name>` symbol.
|
||||
---
|
||||
--- **Conventions:**
|
||||
--- - ELF32 symtab entry = 16 bytes (`st_name:4 + st_value:4 + st_size:4 + st_info:1 + st_other:1 + st_shndx:2`); offsets within each entry are zero-based wire offsets.
|
||||
--- - Direct Lua `string.byte`/`string.sub`/`string.find` boundaries receive `+ 1`.
|
||||
--- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded.
|
||||
--- - The `code_` prefix is stripped (MipsAtom_ macros emit bare atom names, no `code_` prefix).
|
||||
--- - `st_size > 0` filter excludes undefined/imported symbols.
|
||||
--- @param elf_path Path
|
||||
--- @return table<string, {integer, integer}>
|
||||
function M.read_nm(elf_path)
|
||||
local addrs = {}
|
||||
|
||||
-- Read .symtab + .strtab via the existing ELF walker (no subprocess).
|
||||
local sections = M.read_elf_sections(elf_path, {".symtab", ".strtab"})
|
||||
local symtab = sections[".symtab"]
|
||||
local strtab = sections[".strtab"]
|
||||
if not symtab or not strtab or #symtab == 0 or #strtab == 0 then
|
||||
-- No symbol table (e.g. stripped ELF). Return empty.
|
||||
return addrs
|
||||
end
|
||||
|
||||
-- Iterate the 16-byte ELF32 symtab entries.
|
||||
-- Each entry (zero-based): st_name at 0, st_value at 4, st_size at 8, st_info at 12, st_other at 13, st_shndx at 14.
|
||||
local SYM_ENTRY_BYTES = 0x10
|
||||
local SYM_ST_NAME = 0x00
|
||||
local SYM_ST_VALUE = 0x04
|
||||
local SYM_ST_SIZE = 0x08
|
||||
local SYM_ST_INFO = 0x0C
|
||||
local n_syms = #symtab / SYM_ENTRY_BYTES
|
||||
for i = 0, n_syms - 1 do
|
||||
local entry_off = i * SYM_ENTRY_BYTES
|
||||
local st_info = symtab:byte(entry_off + SYM_ST_INFO + 1)
|
||||
-- High nibble = binding (STB_LOCAL=0, STB_GLOBAL=1, STB_WEAK=2).
|
||||
-- Use math.floor(/16) instead of bit.rshift for LuaJIT 2.1 compat
|
||||
-- (LuaJIT's `>>` is 5.3+, but math.floor(x/16) works on all versions).
|
||||
local binding = math.floor(st_info / 16)
|
||||
if binding == 0 or binding == 1 then -- STB_LOCAL or STB_GLOBAL
|
||||
local st_size = M.read_u32_le(symtab, entry_off + SYM_ST_SIZE)
|
||||
if st_size > 0 then
|
||||
local st_name_off = M.read_u32_le(symtab, entry_off + SYM_ST_NAME)
|
||||
-- Extract the name from .strtab (null-terminated C string).
|
||||
local name_end = strtab:find("\0", st_name_off + 1, true) or (st_name_off + 1)
|
||||
local name = strtab:sub(st_name_off + 1, name_end - 1)
|
||||
-- Filter: keep all symbol-table symbols (atoms emit their name as the bare `<name>` — MipsAtom_ macros strip the `code_` prefix).
|
||||
-- The atoms_source_map pass already filters out non-atom symbols via the source-map.txt cross-ref.
|
||||
if name and #name > 0 then
|
||||
local st_value = M.read_u32_le(symtab, entry_off + SYM_ST_VALUE)
|
||||
addrs[name] = { st_value, st_size }
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
return addrs
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- LEB128 encoders (Unsigned + Signed Little-Endian Base 128)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
--
|
||||
-- DWARF uses LEB128 to encode variable-length integers in its wire format (line-program opcodes, DW_AT values, etc.).
|
||||
-- Both encoders pack 7 bits of data per byte + 1 bit of "more bytes follow" signaling.
|
||||
--
|
||||
-- Per-byte layout:
|
||||
-- bit: 7 6 5 4 3 2 1 0
|
||||
-- │ └───── 7-bit data ─────┘
|
||||
-- └─ continuation flag (LEB_CONT_BIT = 0x80)
|
||||
--
|
||||
-- For SLEB128 (signed), bit 6 of the 7-bit data is the sign bit that the
|
||||
-- decoder uses for sign extension:
|
||||
-- bit 6 = 0 → value is positive (or zero); zero-extend on decode
|
||||
-- bit 6 = 1 → value is negative; one-extend on decode
|
||||
--
|
||||
-- The signed encoder must emit the MINIMUM number of bytes whose final 7-bit payload already has the correct sign bit set
|
||||
-- (otherwise the decoder would round-trip to a different value).
|
||||
--
|
||||
-- Spec: DWARF5 §7.6 "Variable-Length Data" / Appendix C.
|
||||
|
||||
-- Top bit of each LEB128 byte. Set if more bytes follow in the encoding.
|
||||
local LEB_CONT_BIT = 0x80
|
||||
|
||||
-- Low 7 bits of each LEB128 byte. The actual data payload.
|
||||
local LEB_DATA_MASK = 0x7F
|
||||
|
||||
-- Bit 6 of the 7-bit data (i.e. 0x40). For SLEB128: the sign-bit position used by the decoder for sign extension.
|
||||
-- Encoders MUST stop when the next byte would be redundant AND the sign bit in the last byte matches the value's sign.
|
||||
local SLEB_SIGN_BIT = 0x40
|
||||
|
||||
--- ULEB128 (Unsigned Little-Endian Base 128) encoder. Returns the byte string for the non-negative integer `n`.
|
||||
--- Algorithm:
|
||||
--- - Extract the low 7 bits of `n` (LEB_DATA_MASK = 0x7F).
|
||||
--- - Shift `n` right by 7 bits.
|
||||
--- - If more bytes remain, OR in the continuation flag (LEB_CONT_BIT).
|
||||
--- - Repeat until `n` is fully consumed.
|
||||
--- @param n integer -- non-negative
|
||||
--- @return string
|
||||
function M.uleb128(n)
|
||||
if n == nil or type(n) ~= "number" then
|
||||
io.stderr:write("[elf_dwarf.uleb128] got " .. type(n) .. ": " .. tostring(n) .. "\n")
|
||||
io.stderr:write(debug.traceback() .. "\n")
|
||||
error("uleb128 requires non-negative number")
|
||||
end
|
||||
assert(n >= 0, "uleb128 requires non-negative input")
|
||||
local bytes = {}
|
||||
repeat
|
||||
local b = n % (LEB_DATA_MASK + 1) -- extract low 7 bits
|
||||
n = (n - b) / (LEB_DATA_MASK + 1) -- shift right by 7 bits
|
||||
if n > 0 then b = b + LEB_CONT_BIT end -- set continuation bit if more bytes follow
|
||||
bytes[#bytes + 1] = string.char(b)
|
||||
until n == 0
|
||||
return table.concat(bytes)
|
||||
end
|
||||
|
||||
--- SLEB128 (Signed Little-Endian Base 128) encoder. Returns the byte string for the integer `n` (may be negative).
|
||||
--- Algorithm differs from ULEB128 by the termination condition:
|
||||
--- stop when the remaining bits can be inferred from the sign bit in the last byte's 7-bit data payload.
|
||||
--- - If `n == 0` (no more value bits) AND bit 6 of the data = 0 → positive terminator (sign bit says "zero-extend").
|
||||
--- - If `n == -1` (sign-extended all-1s) AND bit 6 of the data = 1 → negative terminator (sign bit says "one-extend").
|
||||
---
|
||||
--- Without these checks, the decoder would round-trip to a different value
|
||||
--- (e.g. encoding `0` as `0x80 0x00` decodes to `0` correctly but is 2 bytes long; the termination check picks the 1-byte `0x00` form).
|
||||
--- @param n integer -- any integer (negative allowed)
|
||||
--- @return string
|
||||
function M.sleb128(n)
|
||||
local bytes = {}
|
||||
local more = true
|
||||
while more do
|
||||
local b = n % (LEB_DATA_MASK + 1) -- extract low 7 bits
|
||||
n = (n - b) / (LEB_DATA_MASK + 1) -- arithmetic shift right by 7
|
||||
-- Termination: remaining value bits fit in the sign bit of the last byte.
|
||||
if n == 0 and b < SLEB_SIGN_BIT then more = false end -- positive terminator
|
||||
if n == -1 and b >= SLEB_SIGN_BIT then more = false end -- negative terminator
|
||||
if more then b = b + LEB_CONT_BIT end
|
||||
bytes[#bytes + 1] = string.char(b)
|
||||
end
|
||||
return table.concat(bytes)
|
||||
end
|
||||
|
||||
--- ULEB128 byte-length: number of bytes the encoder M.uleb128 would produce for `n`.
|
||||
--- Used by callers that need to size a buffer before encoding (e.g. compute_loclists_offsets
|
||||
--- needs the encoded length of an `uleb128(4)` for a `DW_OP_piece + uleb128(U4_BYTE_SIZE)` tail).
|
||||
--- @param n integer -- non-negative
|
||||
--- @return integer -- 1..5 for n in [0, 2^32)
|
||||
function M.uleb128_size(n)
|
||||
assert(n >= 0, "uleb128_size requires non-negative input")
|
||||
if n == 0 then return 1 end
|
||||
local bytes = 1
|
||||
while n >= 0x80 do
|
||||
n = (n - (n % (LEB_DATA_MASK + 1))) / (LEB_DATA_MASK + 1) -- arithmetic shift right by 7
|
||||
bytes = bytes + 1
|
||||
end
|
||||
return bytes
|
||||
end
|
||||
|
||||
--- SLEB128 byte-length: number of bytes the encoder M.sleb128 would produce for `n`.
|
||||
--- Used by callers that need to size a buffer before encoding.
|
||||
--- (e.g. compute_loclists_offsets needs the encoded length of an `sleb128(field.offset)` in a tape piece).
|
||||
--- Handles the signed DWARF5 termination: positive terminator if (n == 0) and bit 6 of last byte is unset;
|
||||
--- negative terminator if (n == -1) and bit 6 of last byte is set.
|
||||
--- @param n integer -- any integer (negative allowed)
|
||||
--- @return integer
|
||||
function M.sleb128_size(n)
|
||||
local more = true
|
||||
local bytes = 0
|
||||
local v = n
|
||||
while more do
|
||||
local b = v % (LEB_DATA_MASK + 1) -- extract low 7 bits
|
||||
v = (v - b) / (LEB_DATA_MASK + 1) -- arithmetic shift right by 7
|
||||
if v == 0 and b < SLEB_SIGN_BIT then more = false end -- positive terminator
|
||||
if v == -1 and b >= SLEB_SIGN_BIT then more = false end -- negative terminator
|
||||
if more then b = b + LEB_CONT_BIT end
|
||||
bytes = bytes + 1
|
||||
end
|
||||
return bytes
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- I/O helpers: atoms source-map + native directory glob
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
return M
|
||||
@@ -0,0 +1,105 @@
|
||||
# scripts/gdb/gdb_tape_atoms.gdb
|
||||
#
|
||||
# Wrapper for the tape-atom step-debug helpers.
|
||||
# The 9 user commands are defined here as STUBS (degraded-state messages).
|
||||
# The real implementations + the per-atom data tables are emitted by `passes/atoms_source_map.lua`
|
||||
# (post-link invocation: `ps1_meta.lua --atoms-source-map --gdb-runtime --elf <elf>`) into `build/gen/gdb_tape_atoms_runtime.gdb`.
|
||||
# Sourcing that file RE-DEFINES the commands with real implementations.
|
||||
#
|
||||
# If `build/gen/gdb_tape_atoms_runtime.gdb` is missing or stale, the stubs remain (E1: no source map).
|
||||
# The user just needs to re-run `build_psyq.ps1` to regenerate.
|
||||
|
||||
# ── Stub commands (defined here so they're always present, even if the runtime file is missing). The runtime file overrides these if sourced. ──
|
||||
|
||||
define tape_atoms
|
||||
echo "[gdb_tape_atoms] STUB: runtime file build/gen/gdb_tape_atoms_runtime.gdb not found."
|
||||
echo "[gdb_tape_atoms] STUB: run .\\build_psyq.ps1 to regenerate, then re-source this file."
|
||||
end
|
||||
document tape_atoms
|
||||
List every tape atom symbol in the loaded ELF (code_<name>) with its .rodata address and word count.
|
||||
STUB state: runtime file not sourced. Run build_psyq.ps1 to regenerate.
|
||||
end
|
||||
|
||||
define break_atom
|
||||
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
end
|
||||
document break_atom
|
||||
Set a breakpoint at the start of tape atom <name>. STUB state.
|
||||
end
|
||||
|
||||
define step_atom
|
||||
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
end
|
||||
document step_atom
|
||||
Resume execution until the next atom boundary. STUB state.
|
||||
end
|
||||
|
||||
define next_atom
|
||||
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
end
|
||||
document next_atom
|
||||
Alias for step_atom. STUB state.
|
||||
end
|
||||
|
||||
define where_in_atom
|
||||
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
end
|
||||
document where_in_atom
|
||||
Report current atom name, .rodata addr, word offset, and source line (if known). STUB state.
|
||||
end
|
||||
|
||||
define stepi_inside_atom
|
||||
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
end
|
||||
document stepi_inside_atom
|
||||
One MIPS-instruction step, then where_in_atom. STUB state.
|
||||
end
|
||||
|
||||
define show_c2
|
||||
printf "C2[ 0] 0x%08x\n", $c2_data[0]
|
||||
printf "C2[ 7] 0x%08x [otz]\n", $c2_data[7]
|
||||
printf "C2[12] 0x%08x [sxy0]\n", $c2_data[12]
|
||||
printf "C2[13] 0x%08x [sxy1]\n", $c2_data[13]
|
||||
printf "C2[14] 0x%08x [sxy2]\n", $c2_data[14]
|
||||
printf "C2[24] 0x%08x [mac0]\n", $c2_data[24]
|
||||
printf "...\n"
|
||||
echo "(STUB state: only 7 representative regs shown. Run build_psyq.ps1 for full dump.)"
|
||||
end
|
||||
document show_c2
|
||||
Pretty-print all 32 C2 data registers as hex + named alias. STUB state (7 reg subset).
|
||||
end
|
||||
|
||||
define show_c2ctl
|
||||
printf "C2CTL[ 0] 0x%08x\n", $c2_control[0]
|
||||
printf "...\n"
|
||||
echo "(STUB state: only 1 reg shown. Run build_psyq.ps1 for full dump.)"
|
||||
end
|
||||
document show_c2ctl
|
||||
Pretty-print all 32 C2 control registers. STUB state (1 reg subset).
|
||||
end
|
||||
|
||||
define wave_ctx
|
||||
printf "$t4 = R_FaceCursor 0x%08x\n", $t4
|
||||
printf "$t5 = R_VertBase 0x%08x\n", $t5
|
||||
printf "$t6 = R_OtBase 0x%08x\n", $t6
|
||||
printf "$t7 = R_PrimCursor 0x%08x\n", $t7
|
||||
end
|
||||
document wave_ctx
|
||||
Pretty-print the 4 wave-context GPRs ($t4..$t7). (wave_ctx works in stub state too.)
|
||||
end
|
||||
|
||||
|
||||
# ── Source the runtime file (re-defines commands with real impls + data). ──
|
||||
|
||||
# Try to source from project-root-relative path first (the typical case).
|
||||
# If the user is in a different CWD, the source will fail and stubs remain.
|
||||
# The runtime file path is computed relative to the ELF's source map convention (build/gen/gdb_tape_atoms_runtime.gdb).
|
||||
echo [gdb_tape_atoms] Wrapper loaded. Sourcing runtime file...
|
||||
# Suppress the "Redefine command" prompts that would otherwise appear when the runtime file overrides the 9 stub commands defined above.
|
||||
# The runtime's `define` blocks are intended to overwrite — there's no ambiguity to confirm.
|
||||
set confirm off
|
||||
|
||||
# Source the runtime file (re-defines commands with real impls + data).
|
||||
source build/gen/gdb_tape_atoms_runtime.gdb
|
||||
set confirm on
|
||||
echo [gdb_tape_atoms] Runtime sourced successfully (9 commands now have real implementations).
|
||||
@@ -0,0 +1,94 @@
|
||||
# scripts/launch_pcsx_debug.ps1
|
||||
#
|
||||
# One-shot launcher for debug sessions: starts pcsx-redux with the .ps-exe
|
||||
# loaded, the gdb stub enabled, AND the pcsx_debug_helper Lua plugin loaded
|
||||
# so external CLI tools (gdb's `shell` command, etc.)
|
||||
# can read GTE state via http://localhost:8080/api/v1/lua/gte
|
||||
# (the gdb stub doesn't expose COP2 at all).
|
||||
#
|
||||
# usage:
|
||||
# .\scripts\launch_pcsx_debug.ps1
|
||||
# .\scripts\launch_pcsx_debug.ps1 -ExePath build\hello_gte.ps-exe
|
||||
# .\scripts\launch_pcsx_debug.ps1 -HelperZip scripts\pcsx_debug_helper.zip
|
||||
#
|
||||
# After launch:
|
||||
# - gdb: target remote localhost:3333
|
||||
# - web: curl http://localhost:8080/api/v1/lua/gte
|
||||
#
|
||||
# Companion: scripts/debug_psyq.ps1 (bare launch — no .ps-exe, no helper).
|
||||
|
||||
[CmdletBinding()]
|
||||
param(
|
||||
[string]$PcsxPath = (Join-Path $PSScriptRoot '..\toolchain\pcsx-redux\vsprojects\x64\Release\pcsx-redux.exe'),
|
||||
[string]$ExePath = (Join-Path $PSScriptRoot '..\build\hello_gte.ps-exe'),
|
||||
[string]$HelperZip = (Join-Path $PSScriptRoot 'pcsx_debug_helper.zip'),
|
||||
[int] $GdbPort = 3333,
|
||||
[int] $WebPort = 8080
|
||||
)
|
||||
|
||||
$ErrorActionPreference = 'Stop'
|
||||
|
||||
# ── Pre-checks ──
|
||||
foreach ($p in @($PcsxPath, $ExePath, $HelperZip)) {
|
||||
if (-not (Test-Path $p)) {
|
||||
Write-Error "Missing: $p"
|
||||
exit 1
|
||||
}
|
||||
}
|
||||
|
||||
# Kill any existing pcsx-redux so the archive file isn't locked.
|
||||
Get-Process pcsx-redux -ErrorAction SilentlyContinue | Stop-Process -Force
|
||||
Start-Sleep -Seconds 2
|
||||
|
||||
# ── Launch ──
|
||||
$absExe = [System.IO.Path]::GetFullPath($ExePath)
|
||||
$absZip = [System.IO.Path]::GetFullPath($HelperZip)
|
||||
|
||||
$args = @(
|
||||
'-gdb', '-run'
|
||||
'-loadexe', "`"$absExe`""
|
||||
'-archive', "`"$absZip`""
|
||||
)
|
||||
|
||||
Write-Host "Launching pcsx-redux..." -ForegroundColor Cyan
|
||||
Write-Host " ps-exe : $absExe"
|
||||
Write-Host " helper zip: $absZip"
|
||||
Write-Host " gdb : localhost:$GdbPort"
|
||||
Write-Host " web : localhost:$WebPort/api/v1/lua/gte"
|
||||
Write-Host ""
|
||||
|
||||
Start-Process -FilePath $PcsxPath -ArgumentList $args | Out-Null
|
||||
|
||||
# ── Wait for both endpoints to come up ──
|
||||
$deadline = (Get-Date).AddSeconds(15)
|
||||
while ((Get-Date) -lt $deadline) {
|
||||
$gdbUp = $false
|
||||
$webUp = $false
|
||||
try {
|
||||
$tcp = New-Object System.Net.Sockets.TcpClient
|
||||
$tcp.BeginConnect('localhost', $GdbPort, $null, $null) | Out-Null
|
||||
Start-Sleep -Milliseconds 100
|
||||
$gdbUp = $tcp.Connected
|
||||
$tcp.Close()
|
||||
} catch { $gdbUp = $false }
|
||||
try {
|
||||
$r = Invoke-WebRequest -Uri "http://localhost:$WebPort/" -UseBasicParsing -TimeoutSec 1 -ErrorAction SilentlyContinue
|
||||
$webUp = $r.StatusCode -ne 0
|
||||
} catch { $webUp = $false }
|
||||
if ($gdbUp -and $webUp) { break }
|
||||
Start-Sleep -Milliseconds 500
|
||||
}
|
||||
|
||||
# ── Smoke-test the gte handler ──
|
||||
try {
|
||||
$r = Invoke-WebRequest -Uri "http://localhost:$WebPort/api/v1/lua/gte" -UseBasicParsing -TimeoutSec 5
|
||||
$firstLine = ([System.Text.Encoding]::UTF8.GetString($r.Content) -split "`n")[0]
|
||||
Write-Host "GTE handler OK: $firstLine" -ForegroundColor Green
|
||||
} catch {
|
||||
Write-Warning "GTE handler NOT responding: $_"
|
||||
Write-Host "Check the pcsx-redux Lua Console for debug cli messages." -ForegroundColor Yellow
|
||||
}
|
||||
|
||||
Write-Host ""
|
||||
Write-Host "pcsx-redux running. PIDs:" -ForegroundColor Cyan
|
||||
Get-Process pcsx-redux | Select-Object Id, ProcessName | Format-Table
|
||||
+1
-1
@@ -2,6 +2,6 @@
|
||||
"configureme": true,
|
||||
"grey": false,
|
||||
"mask": 0.5,
|
||||
"masktype": 2,
|
||||
"masktype": 1,
|
||||
"warp": 0
|
||||
}
|
||||
|
||||
@@ -0,0 +1,696 @@
|
||||
--- passes/annotation.lua — Atom-annotation DSL validator.
|
||||
---
|
||||
--- Validates `MipsAtom_(name) atom_info(atom_bind(Binds_X), atom_reads(...), atom_writes(...)) { ... }` declarations in source files.
|
||||
--- Also reads `Binds_*` struct declarations (`typedef Struct_(Binds_X) { ... };`).
|
||||
---
|
||||
--- `duffle.scan_source()` scans each source once upstream; `ps1_meta.lua` stores that result in `src.scan`.
|
||||
---
|
||||
--- Ownership: the canonical `ctx.shared.corpus` supplies cross-source registries, while each `src.scan` supplies its source's declarations and bodies.
|
||||
--- A context without `ctx.shared.corpus` is rejected with an explicit canonical-corpus message.
|
||||
---
|
||||
--- Writes `<ctx.out_root>/<dir_basename>.errors.h` once per module, with `#error` directives for findings that the C compile surfaces.
|
||||
--- `passes/report.lua` renders annotations.txt from `corpus.sources_by_dir`, re-validating each source through `M.validate()`.
|
||||
---
|
||||
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex, Lua 5.3 compatible.
|
||||
|
||||
-- Bootstrap follows the entry scripts; `scripts/duffle_paths.lua` sets package.path and package.cpath. See `ps1_meta.lua` for the rationale.
|
||||
-- `debug.getinfo(1, "S").source` locates this file for standalone and orchestrated runs, then `duffle_paths.lua` returns the loaded `duffle` module.
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
local write_file = duffle.write_file
|
||||
local ensure_dir = duffle.ensure_dir
|
||||
|
||||
-- The annotation pass reads the source-derived registries from scan_source:
|
||||
-- * pipe_ctx.register_alias_registry — for atom_dbg_reg_default(R_X, ...) and atom_reg_types(R_X, ...) member-identity checks
|
||||
-- * pipe_ctx.type_name_registry — for atom_dbg_reg_default(<T>, ...) and atom_reg_types(<T>, ...) type-identity checks
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Type declarations
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- @class SourceFile
|
||||
--- @field path string -- absolute path to the source file
|
||||
--- @field text string -- the full source text
|
||||
--- @field dir string -- the directory containing the source
|
||||
--- @field basename string -- filename without extension
|
||||
--- @field scan table -- pre-scanned SourceScan payload (from duffle.scan_source)
|
||||
|
||||
--- @class PassCtx
|
||||
--- @field sources SourceFile[]
|
||||
--- @field metadata_path string
|
||||
--- @field shared table
|
||||
--- @field shared.word_counts table<string, integer>
|
||||
--- @field out_root string
|
||||
--- @field project_root string
|
||||
--- @field upstream table<string, table>
|
||||
--- @field flags table
|
||||
--- @field verbose boolean
|
||||
|
||||
--- @class PassResult
|
||||
--- @field outputs table[]
|
||||
--- @field errors table[]
|
||||
--- @field warnings table[]
|
||||
|
||||
--- @class AtomAnnotation
|
||||
--- @field line integer -- source line of the atom_info call
|
||||
--- @field macro string -- the macro name (always "atom_info" in the new shape)
|
||||
--- @field name string -- the atom name
|
||||
--- @field kind string -- always "info"
|
||||
--- @field binds string|nil -- Binds_X name if any
|
||||
--- @field reads string[] -- R_* names (read targets)
|
||||
--- @field writes string[] -- R_* names (write targets)
|
||||
--- @field errors string[]|nil -- parse-time errors from scan_source (atom_info body malformed)
|
||||
|
||||
--- @class DebugSkipMarker -- sub-shape of scan_source.lua's @class DebugSkipMarker
|
||||
--- @field marker_kind string -- exact marker ident read from source. Only "atom_dbg_skip" (bare) is positive.
|
||||
--- @field marker_line integer
|
||||
--- @field args string|nil -- trimmed text inside the parens (nil when has_parens is false)
|
||||
--- @field has_parens boolean
|
||||
--- @field is_bare boolean -- true iff marker_kind == "atom_dbg_skip" AND has_parens == false (the only positive form)
|
||||
--- @field pending boolean -- true while awaiting the following declaration
|
||||
--- @field superseded_by_marker_line integer|nil -- set on a marker that was bumped out of the pending slot
|
||||
--- @field target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed
|
||||
|
||||
--- @class Finding
|
||||
--- @field line integer -- source line (or 0 for pass-level)
|
||||
--- @field msg string -- finding message
|
||||
|
||||
--- @class Findings
|
||||
--- @field errors Finding[]
|
||||
--- @field warnings Finding[]
|
||||
--- @field info Finding[]
|
||||
|
||||
--- @class PipeCtx
|
||||
--- @field atom_index table<string, AtomAnnotation> -- name -> AtomAnnotation (only kind=="atom")
|
||||
--- @field binds_index table<string, BindsStruct> -- name -> BindsStruct
|
||||
--- @field annot_counts table<string, integer> -- name -> annotation count (for unique_annotation check)
|
||||
--- @field types table<string, RegTypeDefault> -- from scan_source
|
||||
--- @field atom_views table<string, AtomViewEntry> -- from scan_source
|
||||
--- @field seen_defaults table<string, integer> -- duplicate atom_dbg_reg_default detection
|
||||
--- @field seen_field table<string, integer> -- Binds_* -> count of fields (set/checked by check_binds_no_duplicate_fields)
|
||||
--- @field _scan SourceScan -- full scan payload (typed-view sub-calls live here)
|
||||
|
||||
--- @class AnnotatedResult
|
||||
--- @field atoms AtomEntry[]
|
||||
--- @field annots AtomAnnotation[]
|
||||
--- @field macros MacroEntry[]
|
||||
--- @field binds BindsEntry[]
|
||||
--- @field errors Finding[]
|
||||
--- @field warnings Finding[]
|
||||
--- @field info Finding[]
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Per-check functions (the CHECK_RULES table's payload)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
--
|
||||
--- The dispatcher in `validate()` routes each result by convention: existence checks write errors[] and shape checks write warnings[].
|
||||
--- `macro_word_drift` writes errors[] for missing or mismatched metadata and info[] for a match.
|
||||
|
||||
--- Check: every annotated atom must have a matching MipsAtom_(name) declaration.
|
||||
--- @param a AtomAnnotation
|
||||
--- @param pipe_ctx PipeCtx
|
||||
--- @param findings Findings
|
||||
local function check_atom_decl_exists(a, pipe_ctx, findings)
|
||||
if not pipe_ctx.atom_index[a.name] then
|
||||
findings.errors[#findings.errors + 1] = {
|
||||
line = a.line,
|
||||
msg = string.format("annotation for '%s' has no matching MipsAtom_(%s) { ... }", a.name, a.name),
|
||||
}
|
||||
end
|
||||
end
|
||||
|
||||
--- Check: every atom may have AT MOST ONE annotation.
|
||||
--- Post-loop: needs full-corpus `annot_counts` from pipe_ctx.
|
||||
--- @param pipe_ctx PipeCtx
|
||||
--- @param findings Findings
|
||||
local function check_unique_annotation(pipe_ctx, findings)
|
||||
for name, n in pairs(pipe_ctx.annot_counts) do
|
||||
if n > 1 then
|
||||
findings.errors[#findings.errors + 1] = {
|
||||
line = pipe_ctx.atom_index[name] and pipe_ctx.atom_index[name].line or 0,
|
||||
msg = string.format("MipsAtom_(%s) has %d annotations (expected at most 1)", name, n),
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
--- Check: BIND atoms must reference a real Binds_* struct.
|
||||
--- I keep this as a warning so the annotation pass can report the common test-fixture case; `check_abi_handoff` in static analysis supplies the build-stopping error.
|
||||
--- @param a AtomAnnotation
|
||||
--- @param pipe_ctx PipeCtx
|
||||
--- @param findings Findings
|
||||
local function check_binds_struct_exists(a, pipe_ctx, findings)
|
||||
if not a.binds then return end
|
||||
if pipe_ctx.binds_index[a.binds] then return end
|
||||
findings.warnings[#findings.warnings + 1] = {
|
||||
line = a.line,
|
||||
msg = string.format("'%s' binds '%s' but no Struct_(%s) { ... } "
|
||||
.. "declaration found (also flagged as an error by check_abi_handoff in the static-analysis pass)"
|
||||
, a.name, a.binds, a.binds),
|
||||
}
|
||||
end
|
||||
|
||||
--- Check: TAPE_WORDS(mac_X, N) ↔ WORD_COUNT(mac_X, N) drift.
|
||||
--- Three outcomes: missing (error), mismatch (error), match (info).
|
||||
--- @param m MacroEntry
|
||||
--- @param wc table<string, integer> -- the shared word-count table (from ctx.shared.word_counts)
|
||||
--- @param findings Findings
|
||||
local function check_macro_word_drift(m, wc, findings)
|
||||
local declared = wc[m.name]
|
||||
if not declared then
|
||||
findings.errors[#findings.errors + 1] = {
|
||||
line = m.line,
|
||||
msg = string.format("TAPE_WORDS(%s, %d) but '%s' is not in metadata.h", m.name, m.words, m.name),
|
||||
}
|
||||
return
|
||||
end
|
||||
if declared ~= m.words then
|
||||
findings.errors[#findings.errors + 1] = {
|
||||
line = m.line,
|
||||
msg = string.format("DRIFT: TAPE_WORDS(%s, %d) but metadata.h declares WORD_COUNT(%s, %d)", m.name, m.words, m.name, declared),
|
||||
}
|
||||
return
|
||||
end
|
||||
findings.info[#findings.info + 1] = {
|
||||
line = m.line,
|
||||
msg = string.format("OK: %s = %d words", m.name, m.words),
|
||||
}
|
||||
end
|
||||
|
||||
--- Check: atom_dbg_reg_default(R_X, <type>) targets an alias in `pipe_ctx.register_alias_registry` and a type in `pipe_ctx.type_name_registry`.
|
||||
--- Pointer depth remains bounded to 0 or 1, and duplicate defaults remain errors.
|
||||
--- @param _src SourceFile -- unused (kept for the per_source shape)
|
||||
--- @param pipe_ctx PipeCtx
|
||||
--- @param findings Findings
|
||||
local function check_semantic_reg_defaults(_src, pipe_ctx, findings)
|
||||
-- Detect duplicate defaults using the ordered occurrence list (the out.types hash only retains the last declaration).
|
||||
local seen_first_line = {}
|
||||
for _, occ in ipairs(pipe_ctx.type_occurrences or {}) do
|
||||
if seen_first_line[occ.reg] == nil then
|
||||
seen_first_line[occ.reg] = occ.source_line
|
||||
else
|
||||
findings.errors[#findings.errors + 1] = {
|
||||
line = occ.source_line,
|
||||
msg = string.format(
|
||||
"duplicate atom_dbg_reg_default for %q at line %d (first declared at line %d); one default per register",
|
||||
occ.reg, occ.source_line, seen_first_line[occ.reg]),
|
||||
}
|
||||
end
|
||||
end
|
||||
local reg_registry = pipe_ctx.register_alias_registry or {}
|
||||
local type_registry = pipe_ctx.type_name_registry or {}
|
||||
for reg, def in pairs(pipe_ctx.types or {}) do
|
||||
if not reg_registry[reg] then
|
||||
findings.errors[#findings.errors + 1] = {
|
||||
line = def.source_line,
|
||||
msg = string.format(
|
||||
"atom_dbg_reg_default at line %d references unknown register %q (not in register_alias_registry)",
|
||||
def.source_line, reg),
|
||||
}
|
||||
end
|
||||
if def.pointer_depth == nil or def.pointer_depth < 0 or def.pointer_depth > 1 then
|
||||
findings.errors[#findings.errors + 1] = {
|
||||
line = def.source_line,
|
||||
msg = string.format(
|
||||
"atom_dbg_reg_default at line %d for %q has unsupported pointer depth %d (expected 0 or 1)",
|
||||
def.source_line, reg, def.pointer_depth or -1),
|
||||
}
|
||||
end
|
||||
if not def.type_name or not type_registry[def.type_name] then
|
||||
findings.errors[#findings.errors + 1] = {
|
||||
line = def.source_line,
|
||||
msg = string.format(
|
||||
"atom_dbg_reg_default at line %d for %q uses unknown type %q (not in type_name_registry)",
|
||||
def.source_line, reg, tostring(def.type_name)),
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
--- Check: atom_reg_types(R_X, <type>) entries target an alias in `pipe_ctx.register_alias_registry` and a type in `pipe_ctx.type_name_registry`.
|
||||
--- A bare `atom_reg` marker opts the `R_<n>` alias into GPR identity; references to R_T0..R_T3 require the same explicit marker.
|
||||
--- @param _src SourceFile
|
||||
--- @param pipe_ctx PipeCtx
|
||||
--- @param findings Findings
|
||||
local function check_atom_reg_types(_src, pipe_ctx, findings)
|
||||
local reg_registry = pipe_ctx.register_alias_registry or {}
|
||||
local type_registry = pipe_ctx.type_name_registry or {}
|
||||
for _, ai in ipairs(pipe_ctx.atom_infos_list or {}) do
|
||||
if ai.reg_type_overrides then
|
||||
for reg, ov in pairs(ai.reg_type_overrides) do
|
||||
if not reg_registry[reg] then
|
||||
findings.errors[#findings.errors + 1] = {
|
||||
line = ai.info_line,
|
||||
msg = string.format(
|
||||
"atom '%s' has atom_reg_types for %q; compute-register types are restricted to opt-in aliases (%q not in register_alias_registry)",
|
||||
ai.atom_name, reg, reg),
|
||||
}
|
||||
end
|
||||
if not ov.type_name or not type_registry[ov.type_name] then
|
||||
findings.errors[#findings.errors + 1] = {
|
||||
line = ai.info_line,
|
||||
msg = string.format(
|
||||
"atom '%s' atom_reg_types for %q uses unknown compute type %q (not in type_name_registry)",
|
||||
ai.atom_name, reg, tostring(ov.type_name)),
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
--- Check: atom_view(Binds_X) entries reference a Binds_* struct with at least one field.
|
||||
--- @param _src SourceFile
|
||||
--- @param pipe_ctx PipeCtx
|
||||
--- @param findings Findings
|
||||
local function check_atom_view_layout(_src, pipe_ctx, findings)
|
||||
for atom_name, view in pairs(pipe_ctx.atom_views or {}) do
|
||||
if not view.binds_name then
|
||||
-- The atom had atom_reg_types but no atom_view; no layout check needed.
|
||||
else
|
||||
local bs = pipe_ctx.binds_index[view.binds_name]
|
||||
if not bs then
|
||||
findings.errors[#findings.errors + 1] = {
|
||||
line = view.info_line,
|
||||
msg = string.format(
|
||||
"atom '%s' has atom_view(%s) but no Struct_(%s) { ... } declaration was found",
|
||||
atom_name, view.binds_name, view.binds_name),
|
||||
}
|
||||
elseif not bs.fields or #bs.fields == 0 then
|
||||
findings.errors[#findings.errors + 1] = {
|
||||
line = bs.line,
|
||||
msg = string.format(
|
||||
"atom '%s' has atom_view(%s) but that struct declares zero typed fields",
|
||||
atom_name, view.binds_name),
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
--- Check: Binds_* structs require unique field names because atom_view uses those names for typed-field lookup in gdb.
|
||||
--- @param _src SourceFile
|
||||
--- @param pipe_ctx PipeCtx
|
||||
--- @param findings Findings
|
||||
local function check_binds_no_duplicate_fields(_src, pipe_ctx, findings)
|
||||
for _, bs in ipairs(pipe_ctx.binds_list or {}) do
|
||||
local seen = {}
|
||||
for _, f in ipairs(bs.fields or {}) do
|
||||
seen[f.name] = (seen[f.name] or 0) + 1
|
||||
end
|
||||
for name, count in pairs(seen) do
|
||||
if count > 1 then
|
||||
findings.errors[#findings.errors + 1] = {
|
||||
line = bs.line,
|
||||
msg = string.format(
|
||||
"%s has duplicate field name %q (count %d); the typed-view contract requires unique field names",
|
||||
bs.name, name, count),
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
-- Check: debug-skip markers must satisfy shape + placement constraints.
|
||||
--- Walks the priority list once; each marker produces at most one error, so one source defect yields one finding.
|
||||
--- Priority order (first defect wins):
|
||||
--- 1. marker_kind ~= "atom_dbg_skip" -> legacy/renamed spelling (use `atom_dbg_skip`)
|
||||
--- 2. marker_kind == "atom_dbg_skip" AND has_parens -> parenthesized form (the marker is bare-only)
|
||||
--- 3. args ~= "" -> takes no arguments
|
||||
--- 4. superseded_by_marker_line -> duplicate marker (cite superseding line)
|
||||
--- 5. pending + no target_kind -> dangling (no following declaration)
|
||||
--- 6. unsupported target_kind -> marker precedes an unrelated declaration
|
||||
--- Valid markers stamp `debug_skip` on whole-atom, bare-component, and proc-component declaration records in scan_source.lua.
|
||||
--- @param marker DebugSkipMarker
|
||||
--- @param _pipe_ctx PipeCtx -- unused today; kept for plex-shape consistency with per_annot
|
||||
--- @param findings Findings
|
||||
local function check_skip_marker(marker, _pipe_ctx, findings)
|
||||
local kind = marker.marker_kind
|
||||
local line = marker.marker_line
|
||||
|
||||
-- Left `scan.debug_skip_markers` with production records for `atom_dbg_skip` only; other identifiers take the walker's unrelated branch.
|
||||
|
||||
if marker.has_parens then
|
||||
findings.errors[#findings.errors + 1] = {
|
||||
line = line,
|
||||
msg = string.format("%s marker at line %d must be bare; the parenthesized form is no longer accepted (use `atom_dbg_skip MipsAtom_(name) { ... }`)",
|
||||
kind, line),
|
||||
}
|
||||
return
|
||||
end
|
||||
|
||||
if marker.args ~= nil and marker.args ~= "" then
|
||||
findings.errors[#findings.errors + 1] = {
|
||||
line = line,
|
||||
msg = string.format("%s marker at line %d takes no arguments; found %q", kind, line, marker.args),
|
||||
}
|
||||
return
|
||||
end
|
||||
|
||||
if marker.superseded_by_marker_line then
|
||||
findings.errors[#findings.errors + 1] = {
|
||||
line = line,
|
||||
msg = string.format("duplicate %s marker at line %d; superseded by another %s marker at line %d"
|
||||
, kind, line, kind, marker.superseded_by_marker_line),
|
||||
}
|
||||
return
|
||||
end
|
||||
|
||||
if marker.pending and not marker.target_kind then
|
||||
findings.errors[#findings.errors + 1] = {
|
||||
line = line,
|
||||
msg = string.format("dangling %s marker at line %d: no following MipsAtom_/MipsAtomComp_/MipsAtomComp_Proc_ declaration"
|
||||
, kind, line),
|
||||
}
|
||||
return
|
||||
end
|
||||
|
||||
if marker.target_kind
|
||||
and marker.target_kind ~= "atom"
|
||||
and marker.target_kind ~= "comp_bare"
|
||||
and marker.target_kind ~= "comp_proc" then
|
||||
findings.errors[#findings.errors + 1] = {
|
||||
line = line,
|
||||
msg = string.format("%s marker at line %d must precede MipsAtom_, MipsAtomComp_, or MipsAtomComp_Proc_; found an unrelated declaration"
|
||||
, kind, line),
|
||||
}
|
||||
end
|
||||
end
|
||||
|
||||
--- Warn when a source references an unregistered alias.
|
||||
---
|
||||
--- R_TapePtr, R_AtomJmp, R_PrimCursor, R_FaceCursor, R_VertBase, and R_OtBase opt in through `#define atom_reg` in lottes_tape.h.
|
||||
--- When a source uses an unregistered R_X, this check emits one pass-level info entry for that source and directs C-ABI register names to explicit alias registration.
|
||||
--- @param _src SourceFile
|
||||
--- @param pipe_ctx PipeCtx
|
||||
--- @param findings Findings
|
||||
local function check_wave_context_migration(_src, pipe_ctx, findings)
|
||||
if not (pipe_ctx.types and next(pipe_ctx.types)) then return end
|
||||
if not (pipe_ctx.atom_infos_list) then return end
|
||||
local reg_registry = pipe_ctx.register_alias_registry or {}
|
||||
for _, ai in ipairs(pipe_ctx.atom_infos_list) do
|
||||
if ai.reg_type_overrides then
|
||||
for reg, _ in pairs(ai.reg_type_overrides) do
|
||||
if not reg_registry[reg] then
|
||||
findings.warnings[#findings.warnings + 1] = {
|
||||
line = 0,
|
||||
msg = "wave-context removed; opt in via #define atom_reg in mips.h "
|
||||
.. "(every R_<alias> that should be visible to the annotation pass "
|
||||
.. "must be enum-declared with the bare atom_reg marker)",
|
||||
}
|
||||
return
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- CHECK_RULES — data-driven check dispatch (the plex pattern)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
--
|
||||
-- Each rule entry picks one of four "shapes" of dispatch:
|
||||
-- per_annot(annot, pipe_ctx, findings) -- runs once per AtomAnnotation
|
||||
-- post(pipe_ctx, findings) -- runs once after all per_annot calls complete (full-corpus aggregation)
|
||||
-- per_macro(macro, wc, findings) -- runs once per TAPE_WORDS / _Pragma macro declaration
|
||||
-- per_skip_marker(marker, pipe_ctx, findings) -- runs once per src.scan.debug_skip_markers entry
|
||||
--
|
||||
-- Adding a new check = 1 row here + 1 function above. The `validate()` dispatch loop never needs editing.
|
||||
|
||||
local CHECK_RULES = {
|
||||
{ name = "atom_decl_exists", per_annot = check_atom_decl_exists },
|
||||
{ name = "binds_struct_exists", per_annot = check_binds_struct_exists },
|
||||
{ name = "unique_annotation", post = check_unique_annotation },
|
||||
{ name = "macro_word_drift", per_macro = check_macro_word_drift },
|
||||
{ name = "skip_marker_validation", per_skip_marker = check_skip_marker },
|
||||
{ name = "semantic_reg_defaults", per_source = check_semantic_reg_defaults },
|
||||
{ name = "atom_reg_types", per_source = check_atom_reg_types },
|
||||
{ name = "atom_view_layout", per_source = check_atom_view_layout },
|
||||
{ name = "binds_no_duplicate_fields", per_source = check_binds_no_duplicate_fields },
|
||||
{ name = "wave_context_migration", per_source = check_wave_context_migration },
|
||||
}
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Validation
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
--
|
||||
-- Pure check: read from src.scan, run validations, emit findings. The scan was done once upstream.
|
||||
|
||||
--- Builds one pass-wide pipe_ctx from the merged `corpus.*` registries and source-ordered `corpus.atom_infos`; per-source declarations and bodies remain in `src.scan`.
|
||||
--- The module ownership contract above requires callers to construct `ctx.shared.corpus` through `build_ctx`; the error message below enforces that gate.
|
||||
--- @param ctx PassCtx
|
||||
--- @return PipeCtx
|
||||
local function build_corpus_pipe_ctx(ctx)
|
||||
local corpus = ctx.shared and ctx.shared.corpus
|
||||
if not corpus then
|
||||
error("annotation requires ctx.shared.corpus "
|
||||
.. "(the canonical corpus is the source of truth; "
|
||||
.. "no per-source fallback is supported)", 0)
|
||||
end
|
||||
|
||||
-- `corpus.atom_infos` preserves source order and duplicates; I precompute counts here for `check_unique_annotation` and the per-source checks.
|
||||
local annot_counts = {}
|
||||
for _, info in ipairs(corpus.atom_infos or {}) do
|
||||
if info and info.atom_name then
|
||||
annot_counts[info.atom_name] = (annot_counts[info.atom_name] or 0) + 1
|
||||
end
|
||||
end
|
||||
|
||||
-- Every consumer of these fields observes mutations via the canonical corpus without independently mutable registry construction.
|
||||
return {
|
||||
-- Cross-source lookup tables from corpus.
|
||||
register_alias_registry = corpus.register_alias_registry or {},
|
||||
type_name_registry = corpus.type_name_registry or {},
|
||||
atom_views = corpus.atom_views or {},
|
||||
atom_ctxs = corpus.atom_ctxs or {},
|
||||
atom_phases = corpus.atom_phases or {},
|
||||
binds_by_name = corpus.binds_by_name or {},
|
||||
atoms_by_name = corpus.atoms_by_name or {},
|
||||
-- Corpus-wide ordered list of atom_info records (source-order + duplicates).
|
||||
atom_infos_list = corpus.atom_infos or {},
|
||||
-- Corpus-wide annotation count aggregation (post-rule consumes this).
|
||||
annot_counts = annot_counts,
|
||||
-- Corpus-wide collisions (recorded by scan_source.merge_corpus_registries).
|
||||
collisions = corpus.collisions or {},
|
||||
-- `check_macro_word_drift` reads `corpus.word_counts`, populated by word_count_eval.run.
|
||||
word_counts = corpus.word_counts or {},
|
||||
}
|
||||
end
|
||||
|
||||
--- Validate one source against its pre-scanned SourceScan payload + the corpus-wide pipe_ctx.
|
||||
--- @param ctx PassCtx
|
||||
--- @param src SourceFile
|
||||
--- @param corpus_pipe_ctx PipeCtx|nil -- built once per pass from corpus registries; nil builds the same projection here.
|
||||
--- @return AnnotatedResult
|
||||
local function validate(ctx, src, corpus_pipe_ctx)
|
||||
corpus_pipe_ctx = corpus_pipe_ctx or build_corpus_pipe_ctx(ctx)
|
||||
local scan = src.scan
|
||||
|
||||
-- Project the pre-scanned atoms to the AtomEntry shape this pass needs.
|
||||
local atoms = {}
|
||||
for _, a in ipairs(scan.atoms) do
|
||||
if a.kind == "atom" then
|
||||
atoms[#atoms + 1] = { line = a.line, name = a.raw_name }
|
||||
end
|
||||
end
|
||||
|
||||
-- Project the pre-scanned atom_infos to AtomAnnotation shape.
|
||||
local annots = {}
|
||||
for _, info in ipairs(scan.atom_infos) do
|
||||
annots[#annots + 1] = {
|
||||
line = info.info_line,
|
||||
macro = "atom_info",
|
||||
name = info.atom_name,
|
||||
kind = "info",
|
||||
binds = info.binds,
|
||||
reads = info.reads or {},
|
||||
writes = info.writes or {},
|
||||
errors = info.errors,
|
||||
}
|
||||
end
|
||||
|
||||
-- Build a per-source pipe_ctx: shared lookups come from `corpus_pipe_ctx`, while declarations, bodies, types, views, defaults, and occurrences come from `src.scan`.
|
||||
local seen_defaults = {}
|
||||
for reg, _ in pairs(scan.types or {}) do
|
||||
seen_defaults[reg] = (seen_defaults[reg] or 0) + 1
|
||||
end
|
||||
local atom_infos_list = {}
|
||||
for _, ai in ipairs(scan.atom_infos or {}) do
|
||||
atom_infos_list[#atom_infos_list + 1] = ai
|
||||
end
|
||||
|
||||
local pipe_ctx = {
|
||||
atom_index = {},
|
||||
binds_index = {},
|
||||
annot_counts = corpus_pipe_ctx.annot_counts,
|
||||
types = scan.types or {},
|
||||
type_occurrences = scan.type_occurrences or {},
|
||||
atom_views = scan.atom_views or {},
|
||||
seen_defaults = seen_defaults,
|
||||
atom_infos_list = atom_infos_list,
|
||||
binds_list = scan.binds or {},
|
||||
-- See the module ownership contract; these shared lookup tables come from corpus_pipe_ctx.
|
||||
register_alias_registry = corpus_pipe_ctx.register_alias_registry,
|
||||
type_name_registry = corpus_pipe_ctx.type_name_registry,
|
||||
}
|
||||
for _, a in ipairs(atoms) do pipe_ctx.atom_index [a.name] = a end
|
||||
for _, b in ipairs(scan.binds) do pipe_ctx.binds_index[b.name] = b end
|
||||
|
||||
-- Findings live in a single struct with three lists (errors / warnings / info).
|
||||
-- Each check writes to the list appropriate for its severity.
|
||||
local findings = { errors = {}, warnings = {}, info = {} }
|
||||
|
||||
-- Lift parse-time errors already recorded in scan_source's atom_info payload into this pass's findings list.
|
||||
for _, a in ipairs(annots) do
|
||||
if a.errors then
|
||||
for _, msg in ipairs(a.errors) do
|
||||
findings.errors[#findings.errors + 1] = {
|
||||
line = a.line,
|
||||
msg = string.format("'%s': %s", a.name, msg),
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
-- THE per-annotation pipeline. ONE loop. CHECK_RULES dispatches per_annot rules.
|
||||
for _, a in ipairs(annots) do
|
||||
for _, rule in ipairs(CHECK_RULES) do
|
||||
if rule.per_annot then rule.per_annot(a, pipe_ctx, findings) end
|
||||
end
|
||||
end
|
||||
|
||||
-- Post-loop rules (one-shot checks that need full-corpus aggregation in pipe_ctx).
|
||||
for _, rule in ipairs(CHECK_RULES) do
|
||||
if rule.post then rule.post(pipe_ctx, findings) end
|
||||
end
|
||||
|
||||
-- scan_source records each marker in scan.debug_skip_markers; this loop validates each record independently and emits at most one error per marker.
|
||||
-- Valid markers stamp `debug_skip = true` on the following atom or component declaration, which downstream consumers read directly.
|
||||
local skip_markers = scan.debug_skip_markers or {}
|
||||
for _, marker in ipairs(skip_markers) do
|
||||
for _, rule in ipairs(CHECK_RULES) do
|
||||
if rule.per_skip_marker then rule.per_skip_marker(marker, pipe_ctx, findings) end
|
||||
end
|
||||
end
|
||||
|
||||
-- Per-macro rules (TAPE_WORDS vs WORD_COUNT drift).
|
||||
local wc = corpus_pipe_ctx.word_counts
|
||||
for _, m in ipairs(scan.macros) do
|
||||
for _, rule in ipairs(CHECK_RULES) do
|
||||
if rule.per_macro then rule.per_macro(m, wc, findings) end
|
||||
end
|
||||
end
|
||||
|
||||
-- Per-source rules (reg defaults, atom_view layout, compute-register type overrides, Binds_* field uniqueness).
|
||||
-- Each per_source rule sees the full scan payload via pipe_ctx.
|
||||
for _, rule in ipairs(CHECK_RULES) do
|
||||
if rule.per_source then rule.per_source(src, pipe_ctx, findings) end
|
||||
end
|
||||
|
||||
-- Information summary (always emitted).
|
||||
findings.info[#findings.info + 1] = {
|
||||
line = 0,
|
||||
msg = string.format("scanned: %d atom(s), %d annotation(s), %d macro-word-decl(s), %d binds struct(s)"
|
||||
, #atoms, #annots, #scan.macros, #scan.binds),
|
||||
}
|
||||
|
||||
return {
|
||||
atoms = atoms,
|
||||
annots = annots,
|
||||
macros = scan.macros,
|
||||
binds = scan.binds,
|
||||
errors = findings.errors,
|
||||
warnings = findings.warnings,
|
||||
info = findings.info,
|
||||
}
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Per-DIRECTORY (per-module) output: errors.h + annotations.txt
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Render `<dir_basename>.errors.h` with `#error` directives for every error found across all sources in the directory.
|
||||
--- Empty directories (no errors, no atoms) produce no file.
|
||||
local function emit_module_errors_h(ctx, dir_basename, atoms_count, errors, sources)
|
||||
if atoms_count == 0 and #errors == 0 then
|
||||
return nil
|
||||
end
|
||||
local out_path = ctx.out_root .. "/" .. dir_basename .. ".errors.h"
|
||||
local lines = {
|
||||
"// Auto-generated by ps1_meta.lua (passes/annotation.lua) — DO NOT EDIT",
|
||||
string.format("// Module: %s Sources: %d", dir_basename, #sources),
|
||||
"#pragma once",
|
||||
"",
|
||||
}
|
||||
if #errors == 0 then
|
||||
lines[#lines + 1] = "// annotation pass OK"
|
||||
else
|
||||
for _, e in ipairs(errors) do
|
||||
local src_tag = ""
|
||||
if e.source then
|
||||
local src_name = e.source:match("([^/\\]+)$") or e.source
|
||||
src_tag = src_name .. ": "
|
||||
end
|
||||
lines[#lines + 1] = string.format('#error "%s%s (line %d)"', src_tag, e.msg, e.line)
|
||||
end
|
||||
end
|
||||
ensure_dir(ctx.out_root)
|
||||
write_file(out_path, table.concat(lines, "\n") .. "\n")
|
||||
return out_path
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- M.run — orchestrator entry
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- @class M
|
||||
|
||||
local M = {}
|
||||
|
||||
-- Expose `validate` for downstream passes (e.g. report.lua) that need to re-render the per-source results into a per-MODULE report.
|
||||
M.validate = validate
|
||||
|
||||
--- @param ctx PassCtx
|
||||
--- @return PassResult
|
||||
function M.run(ctx)
|
||||
local outputs = {}
|
||||
local errors = {}
|
||||
local warnings = {}
|
||||
|
||||
-- Build the shared pipe_ctx once for this run; every validate() call sees the same cross-source registries.
|
||||
-- The corpus owns the canonical cross-source registries; per-source scans retain body / declaration ownership.
|
||||
local corpus_pipe_ctx = build_corpus_pipe_ctx(ctx)
|
||||
local corpus = ctx.shared.corpus
|
||||
|
||||
-- Group `corpus.sources_by_dir` by module, validate every source in each bucket, and emit one errors.h per directory.
|
||||
local by_dir = (corpus and corpus.sources_by_dir) or {}
|
||||
|
||||
for dir, dir_sources in pairs(by_dir) do
|
||||
local dir_basename = dir:match("([^/\\]+)$") or dir
|
||||
local dir_atoms = 0
|
||||
local dir_errors = {}
|
||||
local dir_warnings = {}
|
||||
for _, src in ipairs(dir_sources) do
|
||||
local result = validate(ctx, src, corpus_pipe_ctx)
|
||||
result.source = src.path -- tag for downstream rendering
|
||||
dir_atoms = dir_atoms + #result.atoms
|
||||
for _, e in ipairs(result.errors) do
|
||||
dir_errors[#dir_errors + 1] = { line = e.line, msg = e.msg, source = src.path }
|
||||
errors [#errors + 1] = { line = e.line, msg = e.msg }
|
||||
end
|
||||
for _, w in ipairs(result.warnings) do
|
||||
dir_warnings[#dir_warnings + 1] = { line = w.line, msg = w.msg }
|
||||
warnings [#warnings + 1] = { line = w.line, msg = w.msg }
|
||||
end
|
||||
end
|
||||
|
||||
local err_path = emit_module_errors_h(ctx, dir_basename, dir_atoms, dir_errors, dir_sources)
|
||||
if err_path then
|
||||
table.insert(outputs, { errors_h = err_path })
|
||||
end
|
||||
end
|
||||
|
||||
return { outputs = outputs, errors = errors, warnings = warnings }
|
||||
end
|
||||
|
||||
return M
|
||||
@@ -0,0 +1,535 @@
|
||||
--- passes/atoms_source_map.lua — Per-.word source-line map emitter for tape atoms.
|
||||
---
|
||||
--- Writer: this pass, given `atom.paths` (the per-atom mutable surface owned by `emission_model`). Readers:
|
||||
--- `passes/dwarf_injection.lua` (synthesizes DW_TAG_inlined_subroutine + per-word line program rows) and the gdb-runtime
|
||||
--- wrapper at `scripts/gdb/gdb_tape_atoms.gdb` (loads the source map via `source <path>`).
|
||||
---
|
||||
--- Inputs from `atom.paths`: the ordered `items` stream, dense `word_events`, `invocations` views. Outputs: one
|
||||
--- `WORD N LINE L TEXT T` line per emitted `.word`, plus the per-word provenance form that DWARF synthesis consumes.
|
||||
---
|
||||
--- **Two output forms** (per the workspace's per-emission-form pattern from `guide_metaprogram_ssdl.md`):
|
||||
--- 1. **Sourcemap.txt form** — `<out_root>/<basename>.atoms.sourcemap.txt`. Format-version-tagged for forward-compat.
|
||||
--- Lives in `<out_root>/` (build/gen). Mirrors the convention used by `annotation.lua`
|
||||
--- (`<out_root>/<basename>.errors.h`) and `static_analysis.lua` (`<out_root>/<basename>.static_analysis.txt`).
|
||||
--- Compile artifacts (`*.macs.h`, `*.offsets.h`) stay in `<source_dir>/gen/`.
|
||||
--- 2. **gdb-runtime form** — `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`. A pure gdb command script — addresses come
|
||||
--- from `nm`, the 9 user commands are static `define ... end` blocks. Emitted when `ctx.flags.gdb_runtime` is true
|
||||
--- AND `ctx.flags.elf_path` points to an existing ELF. Useful for `gdb-multiarch --without-python` users
|
||||
--- (the common case on Windows MinGW builds) — `source <path>` loads it with no Python / Tcl / Guile required.
|
||||
---
|
||||
--- **Output format** (sourcemap.txt form):
|
||||
--- ```
|
||||
--- # FORMAT_VERSION 1
|
||||
--- # auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT
|
||||
--- ATOM <name> "<abs-source-path>" <total_words>
|
||||
--- WORD 0 LINE 49 TEXT load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||
--- WORD 1 LINE 49 TEXT load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||
--- ... (one WORD line per .word emitted by the atom body) ...
|
||||
--- ENDATOM
|
||||
--- ATOM <next-name> "<abs-source-path>" <total_words>
|
||||
--- ...
|
||||
--- ENDATOM
|
||||
--- ```
|
||||
---
|
||||
--- Marker records are zero-width in `atom.paths.items`, so they emit no WORD rows in the dense word view.
|
||||
---
|
||||
--- **Conventions:** tabs (1/level), EmmyLua annotations, no regex, Lua 5.3 compatible.
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Module-scope requires + package.path setup
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source`
|
||||
-- (works both standalone + when require'd). `duffle_paths.lua` sets package.path then returns `require("duffle")`
|
||||
-- at the bottom, so the dofile value IS the duffle module.
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
local elf_dwarf = require("elf_dwarf")
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Constants
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Format version emitted as the first line. Bump + add a migration test if the format changes;
|
||||
-- the gdb runtime loader rejects mismatches (E2).
|
||||
local FORMAT_VERSION = 1
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Type declarations
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- @class AtomSourceMapCtx
|
||||
--- @field shared table -- `ctx.shared`
|
||||
--- @field shared.corpus table -- source-order registry; single writer is build_ctx
|
||||
--- @field shared.word_counts table
|
||||
--- @field out_root string -- output root (e.g. "build/gen")
|
||||
--- @field flags table -- `ctx.flags`; reads `flags.gdb_runtime` + `flags.elf_path`
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Atom-path renderers
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Join word boundaries (from `items`) to per-word call text + source lines (from `word_events`).
|
||||
--- @param atom table
|
||||
--- @return table[], integer
|
||||
local function canonical_word_entries(atom)
|
||||
local paths = atom.paths or {}
|
||||
local events = paths.word_events or {}
|
||||
local word_items = {}
|
||||
for _, item in ipairs(paths.items or {}) do
|
||||
if item.kind == "word" then word_items[#word_items + 1] = item end
|
||||
end
|
||||
|
||||
local entries = {}
|
||||
for index, event in ipairs(events) do
|
||||
local item = word_items[index] or {}
|
||||
entries[#entries + 1] = {
|
||||
pos = event.i or (index - 1),
|
||||
line = event.call_line or item.line or 0,
|
||||
text = event.call_text or item.call_text or "",
|
||||
body_line = event.body_line or item.body_line or item.line or 0,
|
||||
invocation = (event.outermost_invocation_id
|
||||
and paths.invocations
|
||||
and paths.invocations[event.outermost_invocation_id]) or nil,
|
||||
}
|
||||
end
|
||||
return entries, #events
|
||||
end
|
||||
|
||||
--- Render one atom's provenance stanza. Format 1 line shapes:
|
||||
--- `WORD N CALL <src-path>:<src-line> MACRO <name> "<def-path>:<def-line>" BODY <line>` (component invocation)
|
||||
--- `WORD N CALL <src-path>:<src-line> RAW` (raw `.word` outside any mac_* component)
|
||||
--- Component identity comes from the outermost invocation record; the count-table lookup confirms the component was
|
||||
--- declared in `corpus.word_counts` (populated by word_count_eval + components passes).
|
||||
--- @param src table
|
||||
--- @param atom table
|
||||
--- @param wc table -- identity alias of corpus.word_counts
|
||||
--- @return string[], integer
|
||||
local function emit_provenance_stanza(src, atom, wc)
|
||||
local lines = {}
|
||||
local rel_path = src.path:gsub("\\\\", "/")
|
||||
local entries, total = canonical_word_entries(atom)
|
||||
|
||||
lines[#lines + 1] = string.format('ATOM %s "%s" 0', atom.raw_name or atom.name, rel_path)
|
||||
|
||||
for _, entry in ipairs(entries) do
|
||||
local inv = entry.invocation
|
||||
local macro_count = inv and wc["mac_" .. inv.component_name]
|
||||
if inv and macro_count ~= nil then
|
||||
lines[#lines + 1] = string.format(
|
||||
'WORD %d CALL %s:%d MACRO %s "%s:%d" BODY %d',
|
||||
entry.pos, rel_path, entry.line, inv.component_name,
|
||||
inv.def_path or "", inv.def_line or 0, entry.body_line)
|
||||
else
|
||||
lines[#lines + 1] = string.format(
|
||||
"WORD %d CALL %s:%d RAW", entry.pos, rel_path, entry.line)
|
||||
end
|
||||
end
|
||||
|
||||
lines[1] = lines[1]:gsub(" 0$", " " .. tostring(total))
|
||||
lines[#lines + 1] = "ENDATOM"
|
||||
return lines, total
|
||||
end
|
||||
|
||||
--- Render the full provenance file content for one source.
|
||||
--- @param src table
|
||||
--- @param wc table
|
||||
--- @return string
|
||||
local function render_provenance(src, wc)
|
||||
local lines = {}
|
||||
lines[#lines + 1] = "# FORMAT_VERSION 1"
|
||||
lines[#lines + 1] = "# auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT"
|
||||
lines[#lines + 1] = "# Per-.word provenance: maps each emitted .word to its call site (atom body"
|
||||
lines[#lines + 1] = "# file:line) and, when the word was emitted by a `mac_X(...)` component invocation,"
|
||||
lines[#lines + 1] = "# the component's definition file:line + the per-word BODY line. Used by"
|
||||
lines[#lines + 1] = "# dwarf_injection to synthesize DW_TAG_inlined_subroutine instances + per-word"
|
||||
lines[#lines + 1] = "# line program rows for native source-level step into component bodies."
|
||||
|
||||
local function append(atom)
|
||||
local stanza = emit_provenance_stanza(src, atom, wc)
|
||||
for _, line in ipairs(stanza) do lines[#lines + 1] = line end
|
||||
end
|
||||
for _, atom in ipairs(src.scan.atoms or {}) do
|
||||
if atom.paths then append(atom) end
|
||||
end
|
||||
for _, atom in ipairs(src.scan.raw_atoms or {}) do
|
||||
if atom.paths then append(atom) end
|
||||
end
|
||||
|
||||
return table.concat(lines, "\n") .. "\n"
|
||||
end
|
||||
|
||||
--- Render one atom's stanza for the sourcemap.txt form (ATOM header line, N WORD lines, ENDATOM marker).
|
||||
--- Returns (lines, total_words).
|
||||
--- @param src table
|
||||
--- @param atom table
|
||||
--- @param wc table
|
||||
--- @return string[], integer
|
||||
local function emit_atom_stanza(src, atom)
|
||||
local lines = {}
|
||||
local rel_path = src.path:gsub("\\\\", "/")
|
||||
local entries, total = canonical_word_entries(atom)
|
||||
|
||||
lines[#lines + 1] = string.format('ATOM %s "%s" 0', atom.raw_name or atom.name, rel_path)
|
||||
for _, entry in ipairs(entries) do
|
||||
lines[#lines + 1] = string.format("WORD %d LINE %d TEXT %s",
|
||||
entry.pos, entry.line, entry.text)
|
||||
end
|
||||
lines[1] = lines[1]:gsub(" 0$", " " .. tostring(total))
|
||||
lines[#lines + 1] = "ENDATOM"
|
||||
return lines, total
|
||||
end
|
||||
|
||||
--- Render the full source map file content for one source (one .atoms.sourcemap.txt per source). Mirrors offsets.lua's
|
||||
--- `project_atoms` shape: scan.atoms + scan.raw_atoms, no kind filter.
|
||||
--- @param src table
|
||||
--- @param wc table
|
||||
--- @return string
|
||||
local function render_source_map(src)
|
||||
local lines = {}
|
||||
lines[#lines + 1] = "# FORMAT_VERSION " .. FORMAT_VERSION
|
||||
lines[#lines + 1] = "# auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT"
|
||||
|
||||
local function append(atom)
|
||||
local stanza = emit_atom_stanza(src, atom)
|
||||
for _, line in ipairs(stanza) do lines[#lines + 1] = line end
|
||||
end
|
||||
for _, atom in ipairs(src.scan.atoms or {}) do
|
||||
if atom.paths then append(atom) end
|
||||
end
|
||||
for _, atom in ipairs(src.scan.raw_atoms or {}) do
|
||||
if atom.paths then append(atom) end
|
||||
end
|
||||
|
||||
return table.concat(lines, "\n") .. "\n"
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- gdb-runtime emission (post-link, addresses via nm)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Escape a string for embedding in a gdb `set $var = "..."` literal.
|
||||
--- gdb uses C-style escaping; we escape `\` and `"` (newlines were flattened earlier).
|
||||
--- @param s string
|
||||
--- @return string
|
||||
local function gdb_escape(s)
|
||||
return (s:gsub("\\", "\\\\"):gsub('"', '\\"'))
|
||||
end
|
||||
|
||||
--- Build the list of atoms with addresses + word entries. Shared helper for the gdb-runtime file emission.
|
||||
--- @param ctx PassCtx
|
||||
--- @return table[] -- list of {idx, name, src_path, file_base, addr, size_bytes, words, entries}
|
||||
local function build_atom_table(ctx)
|
||||
local addrs = elf_dwarf.read_nm(ctx.flags.elf_path)
|
||||
local corpus = ctx.shared and ctx.shared.corpus
|
||||
local matched = {}
|
||||
|
||||
for _, src in ipairs(corpus.source_order or {}) do
|
||||
local file_base = src.path:match("([^/\\\\]+)$") or src.path
|
||||
local function append(atom)
|
||||
if not atom.paths then return end
|
||||
local name = atom.raw_name or atom.name
|
||||
local info = addrs[name]
|
||||
if not info then return end
|
||||
local entries, total = canonical_word_entries(atom)
|
||||
matched[#matched + 1] = {
|
||||
name = name,
|
||||
src_path = src.path,
|
||||
file_base = file_base,
|
||||
addr = info[1],
|
||||
size_bytes = info[2],
|
||||
words = total,
|
||||
entries = entries,
|
||||
}
|
||||
end
|
||||
for _, atom in ipairs((src.scan or {}).atoms or {}) do append(atom) end
|
||||
for _, atom in ipairs((src.scan or {}).raw_atoms or {}) do append(atom) end
|
||||
end
|
||||
|
||||
-- Deterministic order: sort by address (matches `nm` output ordering).
|
||||
table.sort(matched, function(a, b) return a.addr < b.addr end)
|
||||
for i, a in ipairs(matched) do a.idx = i - 1 end
|
||||
return matched
|
||||
end
|
||||
|
||||
--- Append the 9 gdb command definitions to `lines`. Pure gdb scripting — addresses come from `nm`, the convenience
|
||||
--- vars set in `emit_gdb_runtime` provide printf args, and each command is a static sequence of `printf` / `tbreak` /
|
||||
--- `if ... end` blocks. The Lua pass emits N atoms' worth of lines; runtime iteration is gdb's job.
|
||||
---
|
||||
--- Why hardcoded per-atom: gdb's `$` substitution doesn't concat inside var names — `$__atom_name_$__i` in a `while`
|
||||
--- loop resolves to one literal identifier, not `name_i`. Compile-time emission is the only path.
|
||||
--- @param lines table -- output line buffer (mutated in place)
|
||||
--- @param matched table -- list of atom records from `build_atom_table`
|
||||
local function append_gdb_commands(lines, matched)
|
||||
-- ── tape_atoms ──
|
||||
-- Hardcoded one printf per atom. No loop.
|
||||
lines[#lines + 1] = "define tape_atoms"
|
||||
for _, a in ipairs(matched) do
|
||||
-- gdb 12.1 quirk: literals in printf args require an attached target.
|
||||
-- Use the per-atom convenience vars set above as printf args.
|
||||
lines[#lines + 1] = string.format(' printf " code_%%-32s @ 0x%%08x %%4d words\\n", $__atom_name_%d, $__atom_addr_%d, $__atom_words_%d',
|
||||
a.idx, a.idx, a.idx)
|
||||
end
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = "document tape_atoms"
|
||||
lines[#lines + 1] = " List every tape atom symbol in the loaded ELF (code_<name>) with .rodata addr + word count."
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
-- ── break_atom (generic) + per-atom break_atom_X ──
|
||||
lines[#lines + 1] = "define break_atom"
|
||||
lines[#lines + 1] = ' echo "Usage: break_atom_<exact_name> (pick from the list below)"'
|
||||
for _, a in ipairs(matched) do
|
||||
lines[#lines + 1] = string.format(' printf " break_atom_%%-32s\\n", $__atom_name_%d', a.idx)
|
||||
end
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = "document break_atom"
|
||||
lines[#lines + 1] = " Generic help: lists the per-atom break_atom_<name> commands."
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
for _, a in ipairs(matched) do
|
||||
lines[#lines + 1] = string.format("define break_atom_%s", a.name)
|
||||
lines[#lines + 1] = string.format(" break *$__atom_addr_%d", a.idx)
|
||||
lines[#lines + 1] = string.format(' printf " Breakpoint set at code_%s (0x%%08x)\\n", $__atom_addr_%d', a.name, a.idx)
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = string.format("document break_atom_%s", a.name)
|
||||
lines[#lines + 1] = string.format(" Set a breakpoint at code_%s.", a.name)
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = ""
|
||||
end
|
||||
|
||||
-- ── step_atom / next_atom ──
|
||||
-- Hardcoded one tbreak per atom. No loop.
|
||||
lines[#lines + 1] = "define step_atom"
|
||||
for _, a in ipairs(matched) do
|
||||
lines[#lines + 1] = string.format(" tbreak *$__atom_addr_%d", a.idx)
|
||||
end
|
||||
lines[#lines + 1] = " continue"
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = "document step_atom"
|
||||
lines[#lines + 1] = " Set one-shot BPs at every atom + continue. Stops at the next atom boundary."
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
lines[#lines + 1] = "define next_atom"
|
||||
lines[#lines + 1] = " step_atom"
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = "document next_atom"
|
||||
lines[#lines + 1] = " Alias for step_atom."
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
-- ── where_in_atom ──
|
||||
-- Hardcoded one outer-if per atom; inside, one inner-if per WORD entry.
|
||||
lines[#lines + 1] = "define where_in_atom"
|
||||
lines[#lines + 1] = " set $__pc = (unsigned int)$pc"
|
||||
lines[#lines + 1] = " set $__matched = 0"
|
||||
for _, a in ipairs(matched) do
|
||||
-- Precompute end_addr (gdb 12.1's expression evaluator chokes on `addr + words*4`).
|
||||
lines[#lines + 1] = string.format(" set $__end_%d = $__atom_addr_%d + $__atom_words_%d * 4", a.idx, a.idx, a.idx)
|
||||
lines[#lines + 1] = string.format(" if $__pc >= $__atom_addr_%d && $__pc < $__end_%d", a.idx, a.idx)
|
||||
lines[#lines + 1] = string.format(' printf "atom: code_%%s\\n", $__atom_name_%d', a.idx)
|
||||
lines[#lines + 1] = ' printf "addr: 0x%08x\\n", $__pc'
|
||||
lines[#lines + 1] = string.format(" set $__word = ($__pc - $__atom_addr_%d) / 4", a.idx)
|
||||
lines[#lines + 1] = string.format(' printf "word: %%d/%%d\\n", $__word, $__atom_words_%d', a.idx)
|
||||
-- One inner-if per WORD entry. Each word's line + text hardcoded.
|
||||
for _, we in ipairs(a.entries) do
|
||||
lines[#lines + 1] = string.format(" if $__word == %d", we.pos)
|
||||
-- Escape TEXT for printf format string.
|
||||
local escaped_text = we.text:gsub("%%", "%%%%"):gsub('"', '\\"')
|
||||
lines[#lines + 1] = string.format(' printf "source: %%s:%%d %%s\\n", $__atom_file_%d, %d, "%s"', a.idx, we.line, escaped_text)
|
||||
lines[#lines + 1] = " end"
|
||||
end
|
||||
-- Fallback for words beyond the source map (shouldn't happen if nm matches).
|
||||
local max_word = 0
|
||||
if #a.entries > 0 then max_word = a.entries[#a.entries].pos end
|
||||
lines[#lines + 1] = string.format(' if $__word > %d', max_word)
|
||||
lines[#lines + 1] = ' printf "source: (no source-map entry for word %%d; map may be stale)\\n", $__word'
|
||||
lines[#lines + 1] = " end"
|
||||
lines[#lines + 1] = " set $__matched = 1"
|
||||
lines[#lines + 1] = " end"
|
||||
end
|
||||
lines[#lines + 1] = " if !$__matched"
|
||||
lines[#lines + 1] = ' echo PC is not inside any known atom (in .text or unmapped region).'
|
||||
lines[#lines + 1] = " end"
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = "document where_in_atom"
|
||||
lines[#lines + 1] = " Report current atom name, .rodata addr, word offset, and source line."
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
-- ── stepi_inside_atom ──
|
||||
-- Hardcoded one if-containment-check per atom (no loop).
|
||||
-- Precompute end_addr in Lua so we don't ask gdb to evaluate `addr + words*4` inside the if condition
|
||||
-- (gdb 12.1's expression evaluator chokes on the `*` and emits a misleading 'function malloc' error in some gdb builds).
|
||||
lines[#lines + 1] = "define stepi_inside_atom"
|
||||
lines[#lines + 1] = " set $__in_atom = 0"
|
||||
lines[#lines + 1] = " set $__did_step = 0"
|
||||
lines[#lines + 1] = " set $__pc = (unsigned int)$pc"
|
||||
for _, a in ipairs(matched) do
|
||||
-- Precompute end_addr in the convenience var (single expression gdb handles).
|
||||
lines[#lines + 1] = string.format(" set $__end_%d = $__atom_addr_%d + $__atom_words_%d * 4", a.idx, a.idx, a.idx)
|
||||
lines[#lines + 1] = string.format(" if $__pc >= $__atom_addr_%d && $__pc < $__end_%d", a.idx, a.idx)
|
||||
lines[#lines + 1] = " set $__in_atom = 1"
|
||||
lines[#lines + 1] = " stepi"
|
||||
lines[#lines + 1] = " set $__did_step = 1"
|
||||
lines[#lines + 1] = " end"
|
||||
end
|
||||
lines[#lines + 1] = " if !$__did_step"
|
||||
lines[#lines + 1] = ' echo [gdb_tape_atoms] stepi_inside_atom: PC is not inside any atom; refusing to step.'
|
||||
lines[#lines + 1] = " end"
|
||||
lines[#lines + 1] = " where_in_atom"
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = "document stepi_inside_atom"
|
||||
lines[#lines + 1] = " One MIPS-instruction step, then where_in_atom. The step-and-see-source-line workflow."
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
-- ── wave_ctx ──
|
||||
lines[#lines + 1] = "define wave_ctx"
|
||||
lines[#lines + 1] = ' printf "$t4 = R_FaceCursor 0x%08x\\n", $t4'
|
||||
lines[#lines + 1] = ' printf "$t5 = R_VertBase 0x%08x\\n", $t5'
|
||||
lines[#lines + 1] = ' printf "$t6 = R_OtBase 0x%08x\\n", $t6'
|
||||
lines[#lines + 1] = ' printf "$t7 = R_PrimCursor 0x%08x\\n", $t7'
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = "document wave_ctx"
|
||||
lines[#lines + 1] = " Pretty-print the 4 wave-context GPRs ($t4=R_FaceCursor, $t5=R_VertBase, $t6=R_OtBase, $t7=R_PrimCursor). Requires target attached."
|
||||
lines[#lines + 1] = "end"
|
||||
end
|
||||
|
||||
--- Emit the gdb-runtime file (post-link). Pure gdb scripting — addresses come from `mipsel-none-elf-nm -S`, get embedded
|
||||
--- in `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`, and load via `set $var = ...` + `define ... end` blocks at gdb
|
||||
--- source-time.
|
||||
--- @param ctx PassCtx
|
||||
local function emit_gdb_runtime(ctx)
|
||||
if not (ctx.flags and ctx.flags.gdb_runtime) then return end
|
||||
local elf_path = ctx.flags.elf_path
|
||||
if not elf_path or elf_path == "" then
|
||||
io.stderr:write("[atoms_source_map] --gdb-runtime requires --elf <elf>\n")
|
||||
return
|
||||
end
|
||||
if lfs.attributes(elf_path, "mode") ~= "file" then
|
||||
io.stderr:write(string.format(
|
||||
"[atoms_source_map] --gdb-runtime: ELF not found at %s\n", elf_path))
|
||||
return
|
||||
end
|
||||
|
||||
local matched = build_atom_table(ctx)
|
||||
if #matched == 0 then
|
||||
io.stderr:write("[atoms_source_map] --gdb-runtime: no atoms matched against nm symbols (stale scan?).\n")
|
||||
return
|
||||
end
|
||||
|
||||
local lines = {}
|
||||
lines[#lines + 1] = "# Auto-generated by ps1_meta.lua (passes/atoms_source_map.lua)"
|
||||
lines[#lines + 1] = "# DO NOT EDIT — re-run ps1_meta.lua --atoms-source-map --gdb-runtime to regenerate"
|
||||
lines[#lines + 1] = "# Sourced by scripts/gdb/gdb_tape_atoms.gdb (the wrapper)."
|
||||
lines[#lines + 1] = "# Pure gdb scripting — no Python, no Tcl, no Guile required."
|
||||
lines[#lines + 1] = "# Commands are FULLY HARDCODED per-atom because gdb doesn't do nested"
|
||||
lines[#lines + 1] = "# `$` substitution in var names (`$foo_$i` is one literal identifier)."
|
||||
lines[#lines + 1] = "# Per-atom convenience vars ($__atom_name_<i> etc.) are set so gdb's"
|
||||
lines[#lines + 1] = "# `printf` has valid expression args (gdb 12.1 quirks: literals in"
|
||||
lines[#lines + 1] = "# printf args require an attached target; convenience-var args do not)."
|
||||
lines[#lines + 1] = string.format("# %d atoms from ELF: %s", #matched, elf_path)
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
-- Format version + count + ELF path (the latter is referenced by the load-line).
|
||||
lines[#lines + 1] = "set $__atom_format_version = " .. FORMAT_VERSION
|
||||
lines[#lines + 1] = string.format("set $__atom_count = %d", #matched)
|
||||
lines[#lines + 1] = string.format('set $__elf_path = "%s"', gdb_escape(elf_path))
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
-- Per-atom convenience vars (used as printf args; literals aren't accepted
|
||||
-- without an attached target on gdb 12.1).
|
||||
for _, a in ipairs(matched) do
|
||||
lines[#lines + 1] = string.format('set $__atom_name_%d = "%s"', a.idx, gdb_escape(a.name))
|
||||
lines[#lines + 1] = string.format("set $__atom_addr_%d = 0x%x", a.idx, a.addr)
|
||||
lines[#lines + 1] = string.format("set $__atom_words_%d = %d", a.idx, a.words)
|
||||
lines[#lines + 1] = string.format('set $__atom_file_%d = "%s"', a.idx, gdb_escape(a.file_base))
|
||||
end
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
-- The 9 commands (each `define ... end` overrides the wrapper's stub).
|
||||
lines[#lines + 1] = "# ── 9 user commands (overrides wrapper stubs) ──"
|
||||
append_gdb_commands(lines, matched)
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
-- Confirmation line for the source operator.
|
||||
lines[#lines + 1] = 'printf "[gdb_tape_atoms] runtime loaded %d atoms from %s\\n", $__atom_count, $__elf_path'
|
||||
|
||||
local out_path = ctx.out_root .. "/gdb_tape_atoms_runtime.gdb"
|
||||
duffle.ensure_dir(duffle.dirname(out_path))
|
||||
duffle.write_file_lf(out_path, table.concat(lines, "\n") .. "\n")
|
||||
-- io.stderr:write(string.format("[atoms_source_map] wrote %s (%d atoms)\n", out_path, #matched))
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- M — module exports
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
local M = {}
|
||||
|
||||
--- Pass entry. For each source that declares at least one `MipsAtom_(name)` / `MipsCode code_<name>`, emit two files
|
||||
--- in `<out_root>/`: `<basename>.atoms.sourcemap.txt` (per-word call-site map) and `<basename>.atoms.provenance.txt`
|
||||
--- (per-word definition + body line, resolved via the outermost `mac_X(...)` invocation). When `ctx.flags.gdb_runtime`
|
||||
--- is true and `ctx.flags.elf_path` exists, also emit the post-link gdb script `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`.
|
||||
--- @param ctx PassCtx
|
||||
--- @return PassResult
|
||||
function M.run(ctx)
|
||||
local outputs = {}
|
||||
local errors = {}
|
||||
local warnings = {}
|
||||
|
||||
local corpus = ctx.shared and ctx.shared.corpus
|
||||
if type(corpus) ~= "table" or type(corpus.source_order) ~= "table" then
|
||||
error("atoms_source_map.run requires ctx.shared.corpus.source_order (canonical corpus).", 0)
|
||||
end
|
||||
|
||||
-- Word counts come from `corpus.word_counts` (populated by word_count_eval + components passes).
|
||||
local wc = corpus.word_counts or {}
|
||||
if not next(wc) then
|
||||
warnings[#warnings + 1] = {
|
||||
line = 0,
|
||||
msg = "atoms_source_map: corpus.word_counts is empty; the word-counts + components passes may not have populated it. Check the PASSES dep edges.",
|
||||
}
|
||||
end
|
||||
|
||||
-- Always emit the text form (per-source).
|
||||
for _, src in ipairs(corpus.source_order) do
|
||||
local has_projection = false
|
||||
for _, atom in ipairs((src.scan or {}).atoms or {}) do
|
||||
if (atom.kind == "atom" or atom.kind == "raw_atom") and atom.paths then
|
||||
has_projection = true; break
|
||||
end
|
||||
end
|
||||
if not has_projection then
|
||||
for _, atom in ipairs((src.scan or {}).raw_atoms or {}) do
|
||||
if atom.paths then has_projection = true; break end
|
||||
end
|
||||
end
|
||||
if has_projection then
|
||||
local basename = duffle.basename_no_ext(src.path)
|
||||
-- (1) atoms.sourcemap.txt — format-1 per-word call-site map.
|
||||
local sourcemap_path = ctx.out_root .. "/" .. basename .. ".atoms.sourcemap.txt"
|
||||
local sourcemap_body = render_source_map(src)
|
||||
-- (2) atoms.provenance.txt — format-1 per-word definition/body map.
|
||||
local prov_path = ctx.out_root .. "/" .. basename .. ".atoms.provenance.txt"
|
||||
local prov_body = render_provenance(src, wc)
|
||||
duffle.ensure_dir(duffle.dirname(sourcemap_path))
|
||||
duffle.write_file_lf(sourcemap_path, sourcemap_body)
|
||||
duffle.write_file_lf(prov_path, prov_body)
|
||||
outputs[#outputs + 1] = { kind = "report", path = sourcemap_path }
|
||||
outputs[#outputs + 1] = { kind = "report", path = prov_path }
|
||||
end
|
||||
end
|
||||
|
||||
-- Optionally emit the gdb-runtime form (post-link, one file per build).
|
||||
if ctx.flags and ctx.flags.gdb_runtime then
|
||||
emit_gdb_runtime(ctx)
|
||||
end
|
||||
|
||||
return { outputs = outputs, errors = errors, warnings = warnings }
|
||||
end
|
||||
|
||||
return M
|
||||
@@ -0,0 +1,649 @@
|
||||
--- passes/components.lua — Component-macro header generator.
|
||||
---
|
||||
--- Ownership: `corpus.word_counts`, `corpus.components`, and `corpus.component_body_index`.
|
||||
--- Scanner owns `declaration_comment` and `debug_skip` on each declaration record; this pass projects both forward.
|
||||
---
|
||||
--- Reads the pre-scanned SourceScan payload from `duffle.scan_source` for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations,
|
||||
--- then resolves the function-args string from the preceding `FI_ MipsAtom ac_X(...)` declaration via a backward walk.
|
||||
---
|
||||
--- Emits one `<dir_basename>.macs.h` per source with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
|
||||
---
|
||||
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
|
||||
--- Lua 5.3 compatible.
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Module-scope requires + package.path setup
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Bootstrap: same as entry scripts. See `ps1_meta.lua` for the rationale.
|
||||
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
||||
-- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator.
|
||||
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
|
||||
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Constants
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Atom component declaration identifiers.
|
||||
local ATOM_COMP_PROC = "MipsAtomComp_Proc_"
|
||||
local MIPS_ATOM = "MipsAtom" -- prefix on the function declaration that wraps an AtomComp_Proc_
|
||||
|
||||
-- Component-name prefixes.
|
||||
local AC_PREFIX = "ac_" -- arg to MipsAtomComp_(ac_X); the X is the atom name
|
||||
local AC_PREFIX_LEN = 3
|
||||
local MAC_PREFIX = "mac_" -- prefix on generated macros; the rest is the atom name
|
||||
local MAC_PREFIX_LEN = 4
|
||||
|
||||
-- ASCII byte values used in tokenization.
|
||||
local BYTE_NEWLINE = 10
|
||||
local BYTE_SLASH = 47
|
||||
|
||||
-- Source dir basename used as the output `.macs.h` filename.
|
||||
local GEN_SUBDIR = "gen"
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Type declarations
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- @class SourceFile
|
||||
--- @field path string -- absolute path to the source file
|
||||
--- @field text string -- the full source text
|
||||
--- @field dir string -- the directory containing the source
|
||||
--- @field basename string -- filename without extension
|
||||
--- @field scan table -- pre-scanned SourceScan payload (from duffle.scan_source)
|
||||
|
||||
--- @class PassCtx
|
||||
--- @field sources SourceFile[] -- all source files in the build
|
||||
--- @field metadata_path string -- path to word_count.metadata.h
|
||||
--- @field shared table -- cross-pass shared state
|
||||
--- @field out_root string -- output root (e.g. "build/gen")
|
||||
--- @field project_root string -- project root (e.g. "code/")
|
||||
--- @field upstream table<string, table> -- per-pass upstream outputs
|
||||
--- @field flags table -- CLI flags
|
||||
--- @field verbose boolean -- log diagnostic info
|
||||
|
||||
--- @class PassResult
|
||||
--- @field outputs table[] -- {kind=, path=} entries describing emit files
|
||||
--- @field errors table[] -- {line=, msg=} entries; build-stops
|
||||
--- @field warnings table[] -- {line=, msg=} entries; build-succeeds
|
||||
|
||||
--- @class Component
|
||||
--- @field name string -- atom name (without `ac_` prefix)
|
||||
--- @field body string -- brace-delimited body (without the braces)
|
||||
--- @field args string|nil -- function-args string (function form only)
|
||||
--- @field line integer -- source line of the declaration
|
||||
--- @field comment string|nil -- scanner-owned `declaration_comment`; the components pass reads it from the scanner record
|
||||
--- @field kind string -- "comp_bare" | "comp_proc"
|
||||
--- @field debug_skip boolean -- mirror of `a.debug_skip` (scanner-owned); true iff a bare `atom_dbg_skip` marker immediately preceded the declaration
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Local helpers (file I/O + path normalization)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
local M = {}
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Back-walk helpers (composed into the entry point below: find_function_args_for)
|
||||
--
|
||||
-- Only the function-args lookup for proc components occurs here.
|
||||
-- The preceding-comment walk occur in `scan_source.lua` — `a.declaration_comment` carries the resolved comment,
|
||||
-- so this file reads it forward rather than re-walking the source.
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Find the args of the function declaration that immediately precedes a `MipsAtomComp_Proc_` invocation of the given name.
|
||||
--- Returns the args string (e.g., `"U4 off, U4 code, U1 r, U1 g, U1 b"`) or nil if no function declaration is found.
|
||||
---
|
||||
--- Convention: function form is
|
||||
--- `FI_ MipsAtom ac_X(args) MipsAtomComp_Proc_(ac_X, { body })`
|
||||
--- We find the LAST occurrence of `"ac_X("` before `before_pos` and extract the args from inside the parens.
|
||||
--- We then verify the preceding context ends with `MipsAtom`
|
||||
--- (the function-decl keyword with possible qualifiers between).
|
||||
---
|
||||
--- @param source string
|
||||
--- @param name string
|
||||
--- @param before_pos integer
|
||||
--- @return string|nil
|
||||
local function find_function_args_for(source, name, before_pos)
|
||||
-- Find the LAST occurrence of `name + "("` in `source[1..before_pos]`.
|
||||
local name_open = name .. "("
|
||||
local last_idx = nil
|
||||
local scan_pos = 1
|
||||
while true do
|
||||
-- Pass `before_pos + 1` so string.find only returns positions < before_pos + 1
|
||||
-- (string.find's 4th arg `plain` is true; we use the 3rd arg `init` for the upper bound).
|
||||
local found = source:find(name_open, scan_pos, true)
|
||||
if not found or found >= before_pos then break end
|
||||
last_idx = found
|
||||
scan_pos = found + #name_open
|
||||
end
|
||||
if not last_idx then return nil end
|
||||
|
||||
-- Verify the preceding context ends with "MipsAtom" (with possible qualifiers between).
|
||||
local before = source:sub(1, last_idx - 1)
|
||||
local trimmed = duffle.trim(before)
|
||||
if trimmed:sub(-#MIPS_ATOM) ~= MIPS_ATOM then
|
||||
-- Preceding context is not a function declaration.
|
||||
return nil
|
||||
end
|
||||
|
||||
local open_paren = last_idx + #name -- position of "("
|
||||
-- scan: MipsAtom ac_X(
|
||||
local inner = duffle.read_parens(source, open_paren)
|
||||
-- scan: MipsAtom ac_X(<args>)
|
||||
if not inner then return nil end
|
||||
return inner
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Argument-name extraction
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Extract just the parameter NAMES from a function-args string (stripping type annotations). E.g.,
|
||||
--- `"U4 off, U4 code, U1 r, U1 g, U1 b"` -> `{"off", "code", "r", "g", "b"}`
|
||||
--- `"U4 *ptr"` -> `{"ptr"}`
|
||||
--- `""` -> nil
|
||||
--- @param args_str string|nil
|
||||
--- @return string[]|nil
|
||||
local function extract_arg_names(args_str)
|
||||
if not args_str or args_str == "" then return nil end
|
||||
local names = {}
|
||||
local tokens = duffle.split_top_level_commas(args_str)
|
||||
for _, tok in ipairs(tokens) do
|
||||
local trimmed = duffle.trim(tok)
|
||||
if trimmed ~= "" then
|
||||
-- Find the identifier at the end: walk back over trailers (whitespace + `*` + `[]`),
|
||||
-- then walk back over the identifier chars (alnum + `_`).
|
||||
local ident_end = #trimmed
|
||||
while ident_end > 0 do
|
||||
local ch = trimmed:sub(ident_end, ident_end)
|
||||
if ch == " " or ch == "\t" or ch == "*" or ch == "]" or ch == "[" then
|
||||
ident_end = ident_end - 1
|
||||
else
|
||||
break
|
||||
end
|
||||
end
|
||||
local ident_start = ident_end
|
||||
while ident_start > 0 do
|
||||
local ch = trimmed:sub(ident_start, ident_start)
|
||||
if duffle.is_alnum_byte(string.byte(ch)) or ch == "_" then
|
||||
ident_start = ident_start - 1
|
||||
else
|
||||
break
|
||||
end
|
||||
end
|
||||
ident_start = ident_start + 1
|
||||
local name = trimmed:sub(ident_start, ident_end)
|
||||
if name ~= "" then names[#names + 1] = name end
|
||||
end
|
||||
end
|
||||
if #names == 0 then return nil end
|
||||
return names
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Component projection (read from pre-scanned SourceScan)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Project pre-scanned MipsAtomComp_ / MipsAtomComp_Proc_ entries into Component shape.
|
||||
--- Reads the scanner-owned `declaration_comment` (resolved by scan_source.lua, skipping backward across an associated bare `atom_dbg_skip` marker when present).
|
||||
--- Per-source backward lookups remain in place only for the function `args` of proc components.
|
||||
--- That lookup is unique to components.lua and stays separate from the declaration-comment walk.
|
||||
--- Carries `body_tokens` forward from scan-source so word_count_rec reads from the precomputed table instead of calling duffle.tokenize_body again.
|
||||
--- Carries the scanner-owned `debug_skip` flag forward so the generated projection can emit `/* atom_dbg_skip */`
|
||||
--- before the authored comment and so `update_canonical_components` can mirror the same field onto `corpus.components[name]`.
|
||||
--- @param source string -- the full source text (needed for backward lookups)
|
||||
--- @param scan table -- SourceScan from duffle.scan_source
|
||||
--- @return Component[]
|
||||
local function project_components(source, scan)
|
||||
local out = {}
|
||||
for _, a in ipairs(scan.atoms) do
|
||||
if a.kind == "comp_bare" or a.kind == "comp_proc" then
|
||||
local args = find_function_args_for(source, a.raw_name, a.ident_pos)
|
||||
-- Comment ownership: scan_source.lua stamps `declaration_comment` on the record by walking backward past any associated bare marker.
|
||||
-- The pass reads `declaration_comment` directly.
|
||||
local comment = a.declaration_comment or ""
|
||||
out[#out + 1] = {
|
||||
line = a.line,
|
||||
name = a.name,
|
||||
body = a.body,
|
||||
body_off = a.body_off,
|
||||
body_tokens = a.body_tokens,
|
||||
args = args,
|
||||
comment = comment,
|
||||
kind = a.kind, -- "comp_bare" | "comp_proc"; provenance emitter reads this.
|
||||
debug_skip = a.debug_skip == true,
|
||||
}
|
||||
end
|
||||
end
|
||||
return out
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Line-comment → block-comment conversion
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Convert `//` line comments to `/* */` block comments in a token.
|
||||
-- C macros use `\` line-continuations; a `//` comment before `\` would consume the continuation,
|
||||
-- breaking the macro. We convert `//` to `/* */` so the multi-line macro structure is preserved.
|
||||
--
|
||||
-- Skips `//` sequences that are inside string or character literals
|
||||
-- (a rough heuristic — sufficient for component bodies which don't have those constructs).
|
||||
--
|
||||
--- @param s string
|
||||
--- @return string
|
||||
local function convert_line_comments_to_block(s)
|
||||
local result = s
|
||||
local pos = 1
|
||||
local len = #result
|
||||
while pos <= len do
|
||||
local is_double_slash = result:byte(pos) == BYTE_SLASH
|
||||
and pos + 1 <= len and result:byte(pos + 1) == BYTE_SLASH
|
||||
if not is_double_slash then
|
||||
pos = pos + 1
|
||||
else
|
||||
-- Find end of line.
|
||||
local eol = pos
|
||||
while eol <= len and result:byte(eol) ~= BYTE_NEWLINE do
|
||||
eol = eol + 1
|
||||
end
|
||||
local before = result:sub(1, pos - 1)
|
||||
local comment = result:sub(pos + 2, eol - 1) -- skip the `//`
|
||||
local after
|
||||
if eol <= len and result:byte(eol) == BYTE_NEWLINE then
|
||||
after = " */" .. result:sub(eol) -- keep the newline
|
||||
else
|
||||
after = " */"
|
||||
end
|
||||
result = before .. "/*" .. comment .. after
|
||||
pos = #before + 2 + #comment + 3 -- skip past converted comment
|
||||
end
|
||||
end
|
||||
return result
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Word-count computation (memoized recursive lookup)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Strip the `mac_` prefix from a component-call ident so we can look it up against the components-by-name table.
|
||||
--- Returns the ident unchanged if it doesn't start with the prefix
|
||||
--- (so a non-component ident like `mask_upper` falls through to the wc-table branch).
|
||||
--- @param ident string|nil
|
||||
--- @return string|nil
|
||||
local function strip_mac_prefix(ident)
|
||||
if not ident then return nil end
|
||||
if ident:sub(1, MAC_PREFIX_LEN) == MAC_PREFIX then
|
||||
return ident:sub(MAC_PREFIX_LEN + 1)
|
||||
end
|
||||
return ident
|
||||
end
|
||||
|
||||
--- (internal) Recursive word-count lookup. `cache` is the memoization table shared across all components
|
||||
--- in a single source's `count_all_components` pass; the in-progress -1 sentinel detects cycles (A -> B -> A).
|
||||
--- @param name string -- the component name (without `mac_`)
|
||||
--- @param comp_by_name table<string, Component>
|
||||
--- @param wc table<string, integer>
|
||||
--- @param cache table<string, integer>
|
||||
--- @return integer
|
||||
local function word_count_rec(name, comp_by_name, wc, cache)
|
||||
if cache[name] ~= nil then return cache[name] end
|
||||
cache[name] = -1 -- mark in-progress (cycle detection)
|
||||
local cc = comp_by_name[name]
|
||||
local n
|
||||
if cc then
|
||||
n = 0
|
||||
local tokens = cc.body_tokens
|
||||
for _, t in ipairs(tokens) do
|
||||
local trimmed = t.tok
|
||||
if trimmed ~= "" then
|
||||
local lookup = strip_mac_prefix(duffle.read_ident(trimmed, 1))
|
||||
if lookup and comp_by_name[lookup] then
|
||||
-- It's a `mac_X(...)` call. Recurse.
|
||||
n = n + word_count_rec(lookup, comp_by_name, wc, cache)
|
||||
elseif lookup and wc and wc[lookup] then
|
||||
-- Encoding macro or pseudo-instruction (e.g. mask_upper = 2, nop2 = 2).
|
||||
n = n + wc[lookup]
|
||||
else
|
||||
-- Unrecognized token. Fall back to 1 word.
|
||||
n = n + 1
|
||||
end
|
||||
end
|
||||
end
|
||||
else
|
||||
-- Not a known component: assume 1 word (regular instruction).
|
||||
n = 1
|
||||
end
|
||||
cache[name] = n
|
||||
return n
|
||||
end
|
||||
|
||||
--- Compute word counts for every component in `components` in a single pass.
|
||||
--- The name-lookup table + memoization cache are built ONCE (per source) instead of per-component,
|
||||
--- so the cache survives across siblings and a component's recursive `mac_Y(...)`
|
||||
--- references hit memoized values instead of re-walking the body.
|
||||
--- Cycle detection (A -> B -> A) is preserved via the in-progress `-1` sentinel in `cache`.
|
||||
--- @param components Component[]
|
||||
--- @param wc table<string, integer>
|
||||
--- @return table<string, integer> -- map of component name (without `mac_`) -> word count
|
||||
local function count_all_components(components, wc)
|
||||
local comp_by_name = {}
|
||||
for _, cc in ipairs(components) do comp_by_name[cc.name] = cc end
|
||||
local cache = {}
|
||||
local counts = {}
|
||||
for _, c in ipairs(components) do
|
||||
counts[c.name] = word_count_rec(c.name, comp_by_name, wc, cache)
|
||||
end
|
||||
return counts
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Per-component emit logic
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Split a (possibly multi-line) comment into per-line entries.
|
||||
--- Hand-rolled (no regex patterns used).
|
||||
--- @param s string
|
||||
--- @return string[]
|
||||
local function split_comment_lines(s)
|
||||
local out = {}
|
||||
local pos = 1
|
||||
local s_len = #s
|
||||
while pos <= s_len do
|
||||
local nl = s:find("\n", pos, true)
|
||||
if not nl then
|
||||
out[#out + 1] = s:sub(pos)
|
||||
break
|
||||
end
|
||||
out[#out + 1] = s:sub(pos, nl - 1)
|
||||
pos = nl + 1
|
||||
end
|
||||
return out
|
||||
end
|
||||
|
||||
--- Determine the macro signature: function-args list (function form) or variadic-ignored (bare form).
|
||||
--- @param args_str string|nil
|
||||
--- @return string
|
||||
local function signature_from_args(args_str)
|
||||
local arg_names = extract_arg_names(args_str)
|
||||
if arg_names and #arg_names > 0 then
|
||||
return table.concat(arg_names, ", ")
|
||||
end
|
||||
return "..."
|
||||
end
|
||||
|
||||
--- Strip the trailing `" \"` (space + backslash) line continuation from the last body line.
|
||||
--- The last 2 chars are always that pair.
|
||||
local function strip_trailing_continuation(lines)
|
||||
local last = lines[#lines]
|
||||
if last:sub(-2) == " \\" then
|
||||
lines[#lines] = last:sub(1, -3)
|
||||
end
|
||||
end
|
||||
|
||||
--- Emit the `#define mac_X(sig) \<newline>\t<tok1> \<newline>,\t<tok2> ...` block.
|
||||
--- Converts `//` line comments to `/* */` block comments in each token so they don't break the C macro `\` line continuations.
|
||||
local function emit_macro_body(lines, c, sig, tokens)
|
||||
for tok_idx = 1, #tokens do
|
||||
tokens[tok_idx] = convert_line_comments_to_block(tokens[tok_idx])
|
||||
end
|
||||
lines[#lines + 1] = "#define mac_" .. c.name .. "(" .. sig .. ") \\"
|
||||
lines[#lines + 1] = "\t" .. tokens[1] .. " \\"
|
||||
for tok_idx = 2, #tokens do
|
||||
lines[#lines + 1] = ",\t" .. tokens[tok_idx] .. " \\"
|
||||
end
|
||||
strip_trailing_continuation(lines)
|
||||
end
|
||||
|
||||
--- Build the list of lines for one component
|
||||
--- (signature comment, `#define mac_X(...)` line with backslash-continued tokens, then `WORD_COUNT(mac_X, N)` entry).
|
||||
--- For skipped components, a `/* atom_dbg_skip */` marker comment is emitted immediately before the authored comment block.
|
||||
--- The marker is a single line, the comment comes next, and the `#define` line follows. The `debug_skip` stamp is scanner-owned
|
||||
--- (`a.debug_skip == true` on the declaration record); the components pass projects it directly.
|
||||
--- @param c Component
|
||||
--- @param components Component[]
|
||||
--- @param wc table<string, integer>
|
||||
--- @return string[] -- list of lines for this component
|
||||
local function build_component_lines(c, counts)
|
||||
local lines = {}
|
||||
|
||||
-- Marker comment: emitted once for every skipped component.
|
||||
-- The marker is scanner-owned (declared by `atom_dbg_skip` immediately before the declaration in the source);
|
||||
-- the components pass projects `c.debug_skip` and emits the marker as a generated comment.
|
||||
if c.debug_skip then
|
||||
lines[#lines + 1] = "/* atom_dbg_skip */"
|
||||
end
|
||||
|
||||
if c.comment and c.comment ~= "" then
|
||||
for _, line in ipairs(split_comment_lines(c.comment)) do
|
||||
lines[#lines + 1] = line
|
||||
end
|
||||
end
|
||||
|
||||
local tokens = duffle.split_top_level_commas(c.body)
|
||||
for i = 1, #tokens do tokens[i] = duffle.trim(tokens[i]) end
|
||||
local sig = signature_from_args(c.args)
|
||||
-- Direct lookup against the per-source precomputed `counts` table (built once by count_all_components).
|
||||
local n = counts[c.name]
|
||||
|
||||
if n > 0 then
|
||||
emit_macro_body(lines, c, sig, tokens)
|
||||
end
|
||||
|
||||
-- Emit the WORD_COUNT(mac_<X>, N) entry.
|
||||
lines[#lines + 1] = "WORD_COUNT(mac_" .. c.name .. ", " .. n .. ")"
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
return lines
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Per-source emit logic
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Build the boilerplate header lines (the `#ifdef INTELLISENSE_DIRECTIVES` block,
|
||||
--- the `// Auto-generated` comment, the `// Source:` line, and the self-contained `WORD_COUNT` macro definition).
|
||||
--- @param src SourceFile
|
||||
--- @return string[]
|
||||
local function header_boilerplate(src)
|
||||
return {
|
||||
-- #pragma once wrapped in #ifdef INTELLISENSE_DIRECTIVES, matching the convention in lottes_tape.h.
|
||||
-- The build does manual unity includes (the user controls include order), so the pragma is only active for IDE/tooling.
|
||||
"#ifdef INTELLISENSE_DIRECTIVES",
|
||||
"#pragma once",
|
||||
"#endif",
|
||||
"// Auto-generated by ps1_meta.lua — DO NOT EDIT",
|
||||
"// Source: " .. duffle.to_absolute_path(src.path),
|
||||
"// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)",
|
||||
"",
|
||||
-- Self-contained: define WORD_COUNT if not already defined.
|
||||
-- We use the same definition here so the auto-generated entries below expand
|
||||
-- to compile-time constants whether the metadata file is included first or not.
|
||||
"#ifndef WORD_COUNT",
|
||||
"#define WORD_COUNT(name, count) enum { words_##name = (count) };",
|
||||
"#endif",
|
||||
"",
|
||||
}
|
||||
end
|
||||
|
||||
--- Compute the output path for one source's `.macs.h` file.
|
||||
--- The pre-rework convention uses the *directory* basename (not the source file basename)
|
||||
--- e.g. `code/duffle/lottes_tape.h` produces `code/duffle/gen/duffle.macs.h`.
|
||||
--- This matches what the C codebase #includes.
|
||||
--- @param src SourceFile
|
||||
--- @return string -- the output directory
|
||||
--- @return string -- the full output path
|
||||
local function compute_macs_h_path(src)
|
||||
local out_dir = src.dir .. "/" .. GEN_SUBDIR
|
||||
local out_path = out_dir .. "/" .. duffle.basename_no_ext(src.dir) .. ".macs.h"
|
||||
return out_dir, out_path
|
||||
end
|
||||
|
||||
--- Emit a per-source `.macs.h` header with the `mac_X` macros + `WORD_COUNT` entries.
|
||||
--- Writes in BINARY mode so LF line endings are preserved (the git blob is LF; Windows text-mode would emit CRLF and break the byte-identical diff).
|
||||
--- @param ctx PassCtx
|
||||
--- @param src SourceFile
|
||||
--- @param components Component[]
|
||||
--- @param counts table<string, integer> -- precomputed word counts (from count_all_components)
|
||||
--- @return string|nil -- path to the written file (nil if no components)
|
||||
local function emit_component_macros_h(ctx, src, components, counts)
|
||||
if #components == 0 then return nil end
|
||||
local out_dir, out_path = compute_macs_h_path(src)
|
||||
local lines = header_boilerplate(src)
|
||||
|
||||
for _, c in ipairs(components) do
|
||||
for _, l in ipairs(build_component_lines(c, counts)) do
|
||||
lines[#lines + 1] = l
|
||||
end
|
||||
end
|
||||
|
||||
local content = table.concat(lines, "\n") .. "\n"
|
||||
duffle.ensure_dir(out_dir)
|
||||
duffle.write_file_lf(out_path, content)
|
||||
print(string.format(" -> %s", out_path))
|
||||
return out_path
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Pass entry
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- (internal) Extend `corpus.word_counts` with this source's component macros so offsets sees them without re-reading the file.
|
||||
--- First declaration wins: a later caller's count is dropped (the existing entry from the first source is preserved).
|
||||
--- @param corpus table -- the corpus
|
||||
--- @param components Component[]
|
||||
--- @param counts table<string, integer> -- precomputed word counts (from count_all_components)
|
||||
local function update_canonical_word_counts(corpus, components, counts)
|
||||
local wc = corpus.word_counts
|
||||
for _, c in ipairs(components) do
|
||||
local key = "mac_" .. c.name
|
||||
if wc[key] == nil then
|
||||
wc[key] = counts[c.name]
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
--- @class ComponentDef
|
||||
--- @field name string -- bare name (without ac_/mac_ prefix)
|
||||
--- @field line integer -- definition source line (line of `MipsAtomComp_(ac_X)` / `MipsAtomComp_Proc_(ac_X, ...)`)
|
||||
--- @field path string -- absolute source path of the definition
|
||||
--- @field kind string -- "comp_bare" | "comp_proc"
|
||||
--- @field debug_skip boolean -- mirror of the scanner-owned `a.debug_skip`; consumers read this directly
|
||||
|
||||
--- (internal) Populate `corpus.components` with this source's components-by-name map.
|
||||
--- First declaration wins; later declarations of the same bare name are dropped and recorded as a collision via `corpus.collisions` (kind = "component").
|
||||
--- The pass does NOT write to `ctx.shared.components` (ownership follows the canonical contract).
|
||||
--- The `debug_skip` field mirrors the scanner-owned declaration record (`c.debug_skip`).
|
||||
--- No parallel skip map is built here; consumers that need the per-component skip state read `corpus.components[name].debug_skip` directly.
|
||||
--- @param corpus table -- the corpus
|
||||
--- @param src SourceFile
|
||||
--- @param components Component[]
|
||||
local function update_canonical_components(corpus, src, components)
|
||||
local rel_path = src.path:gsub("\\", "/")
|
||||
for _, c in ipairs(components) do
|
||||
-- Keyed by bare name (e.g. `yield`, `load_tri_indices`).
|
||||
-- The atoms_source_map pass looks up components by bare name from the corpus;
|
||||
-- `mac_` prefix lives at the call-site identifier and is stripped before lookup.
|
||||
if corpus.components[c.name] == nil then
|
||||
corpus.components[c.name] = {
|
||||
name = c.name,
|
||||
line = c.line,
|
||||
path = rel_path,
|
||||
kind = c.kind or "comp_bare",
|
||||
debug_skip = c.debug_skip == true,
|
||||
}
|
||||
else
|
||||
-- A second declaration of the same bare name: record a typed collision so static-analysis + the report can surface it.
|
||||
-- Identical-shape declarations (same path + line) reuse the first-wins entry without a collision record.
|
||||
local existing = corpus.components[c.name]
|
||||
if existing.path ~= rel_path or existing.line ~= c.line then
|
||||
local kind = c.kind or "comp_bare"
|
||||
local first_kind = existing.kind or "comp_bare"
|
||||
corpus.collisions[#corpus.collisions + 1] = {
|
||||
kind = "component",
|
||||
name = c.name,
|
||||
first_site = { path = existing.path, line = existing.line },
|
||||
conflicting_site = { path = rel_path, line = c.line },
|
||||
first_shape = "kind=" .. first_kind,
|
||||
conflicting_shape = "kind=" .. kind,
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
--- (internal) Populate `corpus.component_body_index` with this source's body index entries.
|
||||
--- First declaration wins; later declarations are dropped (no separate collision record: the components collision is already surfaced by `update_canonical_components`).
|
||||
--- The pass writes to `corpus.component_body_index` only (the corpus owns this projection).
|
||||
--- @param corpus table -- the corpus
|
||||
--- @param src SourceFile
|
||||
--- @param components Component[]
|
||||
--- @param scan table -- the SourceScan payload (for line_of)
|
||||
local function update_canonical_component_body_index(corpus, src, components, scan)
|
||||
local line_of = scan and scan.line_of
|
||||
for _, c in ipairs(components) do
|
||||
if corpus.component_body_index[c.name] == nil then
|
||||
corpus.component_body_index[c.name] = {
|
||||
body_tokens = c.body_tokens,
|
||||
body_off = c.body_off,
|
||||
line_of = line_of,
|
||||
source = src.path,
|
||||
declaration = c.line,
|
||||
kind = c.kind,
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
--- @param ctx PassCtx
|
||||
--- @return PassResult
|
||||
function M.run(ctx)
|
||||
local outputs = {}
|
||||
local errors = {}
|
||||
local warnings = {}
|
||||
|
||||
-- Corpus ownership gate.
|
||||
local corpus = ctx.shared and ctx.shared.corpus
|
||||
if type(corpus) ~= "table" then
|
||||
error("components.run requires ctx.shared.corpus.", 0)
|
||||
end
|
||||
if type(corpus.source_order) ~= "table" then
|
||||
error("components.run requires ctx.shared.corpus.source_order.", 0)
|
||||
end
|
||||
if type(corpus.word_counts) ~= "table" then
|
||||
error("components.run requires ctx.shared.corpus.word_counts; "
|
||||
.. "word_count_eval.run must run before components.run "
|
||||
.. "(see PASSES deps).", 0)
|
||||
end
|
||||
|
||||
-- Projection ownership:
|
||||
-- * `corpus.word_counts["mac_"..name]` — current component count
|
||||
-- * `corpus.components[name]` — bare-name component definition
|
||||
-- * `corpus.component_body_index[name]` — body / line_of / source index
|
||||
-- The pass writes to the corpus only; consumers read from the corpus directly.
|
||||
|
||||
for _, src in ipairs(corpus.source_order) do
|
||||
-- project_components reads from src.scan + does backward lookups on src.text
|
||||
local components = project_components(src.text, src.scan)
|
||||
if #components > 0 then
|
||||
-- Compute all component word counts once per source.
|
||||
-- Use `corpus.word_counts` so the recursive lookup sees both authored-metadata entries
|
||||
-- (loaded by word_count_eval.run) AND same-source component entries (populated earlier in this loop by `update_canonical_word_counts`).
|
||||
local counts = count_all_components(components, corpus.word_counts)
|
||||
local macs_path = emit_component_macros_h(ctx, src, components, counts)
|
||||
if macs_path then
|
||||
outputs[#outputs + 1] = { macs_h = macs_path }
|
||||
-- Populate the projections AFTER disk emission (so the byte-identical `.macs.h` contract is preserved before any current-count mutation).
|
||||
update_canonical_word_counts(corpus, components, counts)
|
||||
update_canonical_components(corpus, src, components)
|
||||
update_canonical_component_body_index(corpus, src, components, src.scan)
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
return { outputs = outputs, errors = errors, warnings = warnings }
|
||||
end
|
||||
|
||||
return M
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,239 @@
|
||||
--- passes/emission_model.lua: Per-atom emission projection.
|
||||
---
|
||||
--- The `emission-model` pass owns `atom.paths`, the canonical per-atom mutable surface for atoms and raw atoms with bodies in `ctx.shared.corpus.source_order`.
|
||||
--- For each atom, the pass invokes `duffle.project_emission(body_text, component_index, word_counts, components)`.
|
||||
--- It stores the ordered `items` stream plus the dense `word_events` / `markers` / `invocations` views on `atom.paths`.
|
||||
---
|
||||
--- Public boundary:
|
||||
--- * `M.run(ctx)` is the only entry point.
|
||||
--- * The pass returns `{outputs = {}, errors = ..., warnings = ...}`.
|
||||
--- Pass kind = `validation` → `PASS_KIND_STOP_ON_ERROR.validation` preserves the existing build-stopping policy.
|
||||
---
|
||||
--- Source-order discipline:
|
||||
--- * `corpus.source_order` sets the source-record order.
|
||||
--- * Within each source, the pass visits `src.scan.atoms` and `src.scan.raw_atoms` in declaration order.
|
||||
---
|
||||
--- Per-atom projection fields on `atom.paths`:
|
||||
--- `tokens`, `line_in_body`, `items`, `word_events`, `markers`, `invocations`, `errors`, `warnings`.
|
||||
--- The construction walk appends `items` and derives each dense view from that ordered stream.
|
||||
---
|
||||
--- Component expansion and construction validation:
|
||||
--- * known `mac_X(...)` calls recursively expand component bodies;
|
||||
--- * invocation records retain monotonic IDs, parent IDs, immediate call text, and the immutable outermost root call text;
|
||||
--- * invocation construction stamps `debug_skip` from `corpus.components[name].debug_skip` at the construction site (no second pass, no source parse, no parallel lookup);
|
||||
--- * component cycles close balanced invocation boundaries and emit a `cycle` construction error at the recursive edge;
|
||||
--- * declared-vs-measured component word counts emit `count_mismatch` construction errors; opaque uncounted macros emit warnings.
|
||||
---
|
||||
--- `passes.scan_source` strips its private `_code_macros` / `_code_macro_bodies` tables before this pass runs.
|
||||
|
||||
local M = {}
|
||||
|
||||
-- ─────────────────────────────────────────────────────────────────────────
|
||||
-- Bootstrap: load `duffle_paths.lua` via debug.getinfo so the module works standalone (run as `luajit passes/emission_model.lua`) and when require'd from the orchestrator.
|
||||
-- ─────────────────────────────────────────────────────────────────────────
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
|
||||
-- ─────────────────────────────────────────────────────────────────────────
|
||||
-- Helpers
|
||||
-- ─────────────────────────────────────────────────────────────────────────
|
||||
|
||||
-- Convert the recursive walk's body-relative line numbers into physical source lines once.
|
||||
-- The walker builds `line_of` from `body_text` and stamps body-relative line numbers (1..N) into `item.line` and `invocation.call_line`.
|
||||
-- This function converts those values to physical source lines at the close site with the forwarded source `line_of` closure.
|
||||
--
|
||||
-- `call_line` discipline:
|
||||
-- * ROOT invocations (`inv.parent_id == 0`) receive body-relative `call_line` values directly from `M.LineIndex(body_text)` in the walker.
|
||||
-- The source `line_of` closure supplies physical lines at the close site, so this function converts each root value exactly once.
|
||||
-- * INNER invocations (`inv.parent_id ~= 0`) receive physical `call_line` values directly from the COMPONENT's `line_of` in the walker.
|
||||
-- Recursive descent forwards that closure through `corpus.component_body_index[name].line_of`; those values arrive physical and remain unchanged.
|
||||
--
|
||||
-- After this function, every `inv.call_line` is physical. DWARF and provenance output read it directly.
|
||||
-- The word-event loop forwards the already-physical `outer_inv.call_line` into `we.call_line` for words inside an invocation.
|
||||
local function stamp_root_provenance(projection, atom_record, src, corpus)
|
||||
local root_line_of = src.scan and src.scan.line_of
|
||||
assert(type(root_line_of) == "function"
|
||||
, "emission_model: src.scan.line_of is required (canonical LineIndex closure over the source text) to stamp physical provenance")
|
||||
assert(type(atom_record.body_off) == "number"
|
||||
, "emission_model: atom_record.body_off (byte offset of the body's first byte in source) is required to derive `root_body_line`. The scanner must populate body_off for every atom record.")
|
||||
-- `root_body_line` is the physical source line of the ATOM HEADER byte containing the opening `{`; that byte is one byte BEFORE `atom_record.body_off`.
|
||||
-- The walker assigns line 2 to the body's first content line because line 1 is the trailing `\n` after `{`. Body-text line k therefore maps to `root_body_line + (k - 1)`.
|
||||
-- `body_off - 1` points at the opening `{`, whose line index identifies the header line. `body_off` points after `{` and would shift every word row forward by one line.
|
||||
local root_body_line = root_line_of(atom_record.body_off - 1)
|
||||
or atom_record.line or 0
|
||||
local component_index = corpus.component_body_index or {}
|
||||
local word_items = {}
|
||||
|
||||
for _, item in ipairs(projection.items) do
|
||||
if item.kind == "word" then word_items[#word_items + 1] = item end
|
||||
end
|
||||
|
||||
-- Resolve one word's physical body line, where the byte containing that word appears in source.
|
||||
-- * Component expansions carry `invocation_ids`; the component's full-file `line_of` leaves `item.line` physical.
|
||||
-- * Raw tokens in the root atom body carry an empty `invocation_ids` list and a body-relative `item.line`; convert them here.
|
||||
local function body_line_for(event, item)
|
||||
local ids = event.invocation_ids or {}
|
||||
-- The innermost open invocation identifies which line index the walker used.
|
||||
-- A component `line_of` makes `item.line` physical; the atom's `body_text` line index makes it body-relative.
|
||||
if ids and #ids > 0 then
|
||||
local inner_id = ids[#ids]
|
||||
local inner_inv = inner_id and projection.invocations[inner_id]
|
||||
if inner_inv then
|
||||
local component = component_index[inner_inv.component_name]
|
||||
if component and component.line_of then
|
||||
-- Walker used `comp.line_of`, which is the source's physical LineIndex. item.line is already physical.
|
||||
return item.line or 0
|
||||
end
|
||||
end
|
||||
end
|
||||
-- RAW root-body word: item.line is body-text's 1-based line number (the first content line is line 2 because line 1 is the trailing `\n` after `{`).
|
||||
-- Convert body-text-relative → physical using `root_body_line + (item.line - 1)`.
|
||||
return (root_body_line or 0) + (item.line or 1) - 1
|
||||
end
|
||||
|
||||
-- Stamp the root source path onto invocation records whose `call_path` the walker left empty.
|
||||
-- The walker passes `body_entry.source` to `emit_invoke_begin`; `M.project_emission` creates the root `body_entry` with source `""`, leaving its `call_path` empty.
|
||||
-- This stamp gives every invocation a physical `call_path` matching `passes/atoms_source_map.lua`'s in-memory provenance projection.
|
||||
local root_path = src.path or ""
|
||||
for _, inv in ipairs(projection.invocations) do
|
||||
if inv.call_path == nil or inv.call_path == "" then
|
||||
inv.call_path = root_path
|
||||
end
|
||||
end
|
||||
|
||||
-- Normalize `inv.call_line` to a physical source line.
|
||||
-- * ROOT invocations (`parent_id == 0`) carry body-relative `call_line` values from `M.LineIndex(body_text)`; convert them once with `root_body_line`.
|
||||
-- * INNER invocations (`parent_id ~= 0`) carry physical `call_line` values from the component's `line_of`; retain them unchanged.
|
||||
for _, inv in ipairs(projection.invocations) do
|
||||
if inv.parent_id == 0 then
|
||||
inv.call_line = (root_body_line or 0) + (inv.call_line or 1) - 1
|
||||
end
|
||||
end
|
||||
|
||||
-- Build `body_lines` for each invocation.
|
||||
-- `atoms_source_map` and `dwarf_injection` read `inv.body_lines[k]` directly from the invocation record created here.
|
||||
-- Component words already carry physical `item.line` values from the walker's COMPONENT line index, so `body_line_for` returns them unchanged.
|
||||
for _, inv in ipairs(projection.invocations) do
|
||||
local sw = inv.start_word
|
||||
local ew = inv.end_word
|
||||
local bls = {}
|
||||
for i = sw, ew do
|
||||
local it = projection.items and projection.items[i]
|
||||
if it and it.kind == "word" then
|
||||
local fake_event = { invocation_ids = { inv.id } }
|
||||
bls[#bls + 1] = body_line_for(fake_event, it) or 0
|
||||
end
|
||||
end
|
||||
inv.body_lines = bls
|
||||
end
|
||||
|
||||
-- Resolve each `word_event`'s physical `body_line` and `call_line`.
|
||||
-- For words inside an invocation, `we.call_line` identifies the OUTER atom source line containing the `mac_X(...)` token that triggered expansion.
|
||||
-- The root-invocation conversion above makes every `inv.call_line` physical; forward it directly and use each raw word's `body_line` as the fallback.
|
||||
for index, we in ipairs(projection.word_events) do
|
||||
local item = word_items[index] or {}
|
||||
local body_line = body_line_for(we, item)
|
||||
item.line = body_line
|
||||
we.body_line = body_line
|
||||
|
||||
local call_line = body_line
|
||||
local outer_id = we.outermost_invocation_id or 0
|
||||
local outer_inv = projection.invocations[outer_id]
|
||||
if outer_inv then
|
||||
-- `outer_inv.call_line` is physical after the conversion loop above, so use it directly.
|
||||
call_line = outer_inv.call_line
|
||||
end
|
||||
we.call_line = call_line
|
||||
|
||||
if we.def_path == nil or we.def_path == "" then we.def_path = src.path or "" end
|
||||
if we.def_line == nil or we.def_line == 0 then we.def_line = atom_record.line or 0 end
|
||||
if we.call_path == nil or we.call_path == "" then we.call_path = src.path or "" end
|
||||
end
|
||||
end
|
||||
|
||||
-- Project one atom record into `atom.paths`.
|
||||
-- Mutates the atom record in-place and returns the projection (for pass-level error/warning accumulation).
|
||||
local function project_atom(atom_record, src, corpus)
|
||||
local body = atom_record.body or ""
|
||||
local wc = corpus.word_counts or {}
|
||||
local cbi = corpus.component_body_index or {}
|
||||
-- That construction site stamps `invocation.debug_skip` while appending each record to `proj.invocations`.
|
||||
local proj = duffle.project_emission(body, cbi, wc, corpus.components)
|
||||
local paths = {
|
||||
tokens = atom_record.body_tokens or {},
|
||||
line_in_body = duffle.build_body_line_index(body),
|
||||
items = proj.items,
|
||||
word_events = proj.word_events,
|
||||
markers = proj.markers,
|
||||
invocations = proj.invocations,
|
||||
errors = proj.errors,
|
||||
warnings = proj.warnings,
|
||||
}
|
||||
stamp_root_provenance(proj, atom_record, src, corpus)
|
||||
atom_record.paths = paths
|
||||
return proj
|
||||
end
|
||||
|
||||
-- ─────────────────────────────────────────────────────────────────────────
|
||||
-- Run the emission-model pass.
|
||||
-- ─────────────────────────────────────────────────────────────────────────
|
||||
|
||||
--- @param ctx PassCtx -- { shared = { corpus = ... }, out_root, ... }
|
||||
--- @return PassResult
|
||||
function M.run(ctx)
|
||||
local outputs = {}
|
||||
local errors = {}
|
||||
local warnings = {}
|
||||
|
||||
local corpus = ctx and ctx.shared and ctx.shared.corpus
|
||||
if type(corpus) ~= "table" then error("emission_model: ctx.shared.corpus is required (canonical projection)", 0) end
|
||||
if type(corpus.source_order) ~= "table" then error("emission_model: ctx.shared.corpus.source_order is required", 0) end
|
||||
|
||||
-- Project once, collect errors + warnings for one atom.
|
||||
-- Kind must be one of: atom | raw_atom | comp_bare | comp_proc.
|
||||
local function process_atom(atom, src)
|
||||
if not (atom and atom.body) then return end
|
||||
local kind = atom.kind
|
||||
if kind ~= "atom" and kind ~= "raw_atom" and kind ~= "comp_bare" and kind ~= "comp_proc" then
|
||||
return
|
||||
end
|
||||
local proj = project_atom(atom, src, corpus)
|
||||
for _, e in ipairs(proj.errors) do
|
||||
-- Preserve `kind` (cycle / count_mismatch / unbalanced) so readers dispatch on the diagnostic class and leave the message string as display text.
|
||||
errors[#errors + 1] = {
|
||||
kind = e.kind,
|
||||
line = e.line,
|
||||
msg = e.msg,
|
||||
source = e.source or src.path,
|
||||
}
|
||||
end
|
||||
for _, w in ipairs(proj.warnings) do
|
||||
warnings[#warnings + 1] = {
|
||||
kind = w.kind,
|
||||
line = w.line,
|
||||
msg = w.msg,
|
||||
}
|
||||
end
|
||||
end
|
||||
|
||||
-- Walk `corpus.source_order`; within each source, visit atoms followed by raw_atoms.
|
||||
-- Recognized kinds (atom | raw_atom | comp_bare | comp_proc) each receive the atom.paths projection via duffle.project_emission.
|
||||
-- Components are macros inlined into atom bodies; focused tests and isolated component analyses consume atom.paths directly.
|
||||
for _, src in ipairs(corpus.source_order) do
|
||||
local scan = src.scan or {}
|
||||
for _, atom in ipairs(scan.atoms or {}) do
|
||||
process_atom(atom, src)
|
||||
end
|
||||
for _, atom in ipairs(scan.raw_atoms or {}) do
|
||||
process_atom(atom, src)
|
||||
end
|
||||
end
|
||||
|
||||
return {
|
||||
outputs = outputs,
|
||||
errors = errors,
|
||||
warnings = warnings,
|
||||
}
|
||||
end
|
||||
|
||||
return M
|
||||
@@ -0,0 +1,252 @@
|
||||
--- passes/offsets.lua — Branch-offset generator.
|
||||
---
|
||||
--- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`)
|
||||
--- for `MipsAtom_(name)` and `MipsCode code_<name>` declarations, computes the word offset
|
||||
--- from each `atom_offset(F, T)` marker to its target `atom_label(T)` declaration, and emits
|
||||
--- `<dir_basename>.offsets.h` with one `#define _atom_offset_F_T = N` per branch.
|
||||
---
|
||||
--- The offset is `target_word - branch_word - 1` (the standard MIPS branch-immediate encoding: branch_offset = relative_pc_in_words - 1).
|
||||
---
|
||||
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
|
||||
--- Lua 5.3 compatible.
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Module-scope requires + package.path setup
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Bootstrap: same as entry scripts. See `ps1_meta.lua` for the rationale.
|
||||
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
||||
-- Uses `debug.getinfo` to find this file's own directory, so it works
|
||||
-- both standalone and when require'd from the orchestrator.
|
||||
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
|
||||
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Constants
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Offset macro/enum naming prefixes (the emitted header uses these).
|
||||
local OFFSET_MACRO_PREFIX = "_atom_offset_"
|
||||
local OFFSET_ENUM_PREFIX = "atom_offset_"
|
||||
|
||||
-- Column width for the `#define _atom_offset_F_T = N` alignment.
|
||||
local OFFSET_MACRO_COL = 44
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Type declarations
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- @class SourceFile
|
||||
--- @field path string -- absolute path to the source file
|
||||
--- @field text string -- the full source text
|
||||
--- @field dir string -- the directory containing the source
|
||||
--- @field basename string -- filename without extension
|
||||
--- @field scan table -- pre-scanned SourceScan payload (from duffle.scan_source)
|
||||
|
||||
--- @class PassCtx
|
||||
--- @field shared table -- cross-pass shared state
|
||||
--- @field shared.corpus table -- canonical corpus projection
|
||||
--- @field shared.word_counts table
|
||||
--- @field out_root string -- output root (e.g. "build/gen")
|
||||
|
||||
--- @class PassResult
|
||||
--- @field outputs table[] -- {kind=, path=} entries describing emit files
|
||||
--- @field errors table[] -- {line=, msg=} entries; build-stops
|
||||
--- @field warnings table[] -- {line=, msg=} entries; build-succeeds
|
||||
|
||||
--- @class BranchOffset
|
||||
--- @field tag string -- the marker tag (e.g. "F" in `atom_offset(F, T)`)
|
||||
--- @field target string -- the target label name (e.g. "T" in `atom_offset(F, T)`)
|
||||
--- @field branch_word integer -- branch word position within the atom body
|
||||
--- @field offset integer -- computed `target_word - branch_word - 1`
|
||||
|
||||
--- @class AtomData
|
||||
--- @field name string -- atom name
|
||||
--- @field total_words integer -- total word count of the atom body
|
||||
--- @field offsets BranchOffset[] -- per-branch offset list
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Canonical marker projection
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- MARKER_PROJECTORS is the marker-kind data table.
|
||||
-- The emission-model pass already records marker word positions;
|
||||
-- this pass only projects those records into the label/branch lookup shape needed by offset computation.
|
||||
local MARKER_PROJECTORS = {
|
||||
label = function(state, marker)
|
||||
state.labels[marker.name] = marker.word_index
|
||||
end,
|
||||
offset = function(state, marker)
|
||||
state.branches[#state.branches + 1] = {
|
||||
tag = marker.name,
|
||||
target = marker.target,
|
||||
branch_word = marker.word_index,
|
||||
}
|
||||
end,
|
||||
}
|
||||
|
||||
--- Project canonical marker records into the two lookup tables used by the offset renderer.
|
||||
--- No source text, body text, or body token is inspected.
|
||||
--- @param markers table[] -- atom.paths.markers
|
||||
--- @return table<string, integer>, table[]
|
||||
local function project_markers(markers)
|
||||
local state = { labels = {}, branches = {} }
|
||||
for _, marker in ipairs(markers or {}) do
|
||||
local project = MARKER_PROJECTORS[marker.kind]
|
||||
if project then project(state, marker) end
|
||||
end
|
||||
return state.labels, state.branches
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Offset computation + header generation
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Compute branch offsets as `target_word - branch_word - 1` (the standard MIPS branch-immediate encoding).
|
||||
--- @param labels table<string, integer>
|
||||
--- @param branches table[]
|
||||
--- @return BranchOffset[]
|
||||
local function compute_offsets(labels, branches)
|
||||
local results = {}
|
||||
for _, br in ipairs(branches) do
|
||||
local target = labels[br.target]
|
||||
if not target then
|
||||
error("Branch target '" .. br.target .. "' has no atom_label (at word " .. br.branch_word .. ")")
|
||||
end
|
||||
results[#results + 1] = {
|
||||
target = br.target,
|
||||
tag = br.tag,
|
||||
branch_word = br.branch_word,
|
||||
offset = target - br.branch_word - 1,
|
||||
}
|
||||
end
|
||||
return results
|
||||
end
|
||||
|
||||
--- Right-pad `s` with spaces to width `w`. If `s` is already `w` or wider, no padding is added.
|
||||
--- @param s string
|
||||
--- @param w integer
|
||||
--- @return string
|
||||
local function pad_right(s, w)
|
||||
return s .. string.rep(" ", math.max(0, w - #s))
|
||||
end
|
||||
|
||||
--- (internal) Build a constant-table entry `{macro_name, enum_name, value}` from a BranchOffset.
|
||||
--- @param bo BranchOffset
|
||||
--- @return table
|
||||
local function make_offset_const(bo)
|
||||
return {
|
||||
macro_name = OFFSET_MACRO_PREFIX .. bo.tag .. "_" .. bo.target,
|
||||
enum_name = OFFSET_ENUM_PREFIX .. bo.tag .. "_" .. bo.target,
|
||||
value = bo.offset,
|
||||
}
|
||||
end
|
||||
|
||||
--- (internal) Emit one atom's offset constants + enum into the lines buffer.
|
||||
--- @param add fun(s: string)
|
||||
--- @param atom AtomData
|
||||
local function emit_atom_offsets(add, atom)
|
||||
if #atom.offsets == 0 then return end
|
||||
add("// --- atom: " .. atom.name .. " (" .. atom.total_words .. " words) ---")
|
||||
add("")
|
||||
local consts = {}
|
||||
for _, r in ipairs(atom.offsets) do
|
||||
consts[#consts + 1] = make_offset_const(r)
|
||||
end
|
||||
for _, c in ipairs(consts) do
|
||||
add("#define " .. pad_right(c.macro_name, OFFSET_MACRO_COL) .. " " .. c.value)
|
||||
end
|
||||
add("")
|
||||
add("enum {")
|
||||
for _, c in ipairs(consts) do
|
||||
add(" " .. c.enum_name .. " = " .. c.macro_name .. ",")
|
||||
end
|
||||
add("};")
|
||||
add("")
|
||||
end
|
||||
|
||||
--- Generate the per-source .offsets.h header.
|
||||
--- @param source_path string
|
||||
--- @param atoms_data AtomData[]
|
||||
--- @return string
|
||||
local function generate_header(source_path, atoms_data)
|
||||
local basename = duffle.basename_no_ext(source_path)
|
||||
|
||||
local lines = {}
|
||||
local function add(s) lines[#lines + 1] = s end
|
||||
|
||||
add("// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT")
|
||||
add("// Source: " .. source_path)
|
||||
add("#pragma once")
|
||||
add("")
|
||||
add("#pragma region " .. basename)
|
||||
add("")
|
||||
add("")
|
||||
for _, atom in ipairs(atoms_data) do
|
||||
emit_atom_offsets(add, atom)
|
||||
end
|
||||
add("#pragma endregion " .. basename)
|
||||
add("")
|
||||
return table.concat(lines, "\n") .. "\n"
|
||||
end
|
||||
|
||||
local M = {}
|
||||
|
||||
--- (internal) Process one source: render offsets from canonical atom paths.
|
||||
--- Returns the offsets_h path if a header was written, or nil.
|
||||
--- @param ctx PassCtx
|
||||
--- @param src SourceFile
|
||||
--- @return string|nil -- the offsets_h path
|
||||
local function process_source(ctx, src)
|
||||
local atoms_data = {}
|
||||
local scan = src.scan or {}
|
||||
|
||||
local function append_atom(atom)
|
||||
local paths = atom and atom.paths
|
||||
if not paths then return end
|
||||
local labels, branches = project_markers(paths.markers)
|
||||
atoms_data[#atoms_data + 1] = {
|
||||
name = atom.raw_name or atom.name,
|
||||
total_words = #(paths.word_events or {}),
|
||||
offsets = compute_offsets(labels, branches),
|
||||
}
|
||||
end
|
||||
|
||||
for _, atom in ipairs(scan.atoms or {}) do append_atom(atom) end
|
||||
for _, atom in ipairs(scan.raw_atoms or {}) do append_atom(atom) end
|
||||
if #atoms_data == 0 then return nil end
|
||||
|
||||
local out_path = src.dir .. "/gen/" .. duffle.basename_no_ext(src.dir) .. ".offsets.h"
|
||||
duffle.ensure_dir(duffle.dirname(out_path))
|
||||
duffle.write_file(out_path, generate_header(src.path:gsub("/", "\\"), atoms_data))
|
||||
return out_path
|
||||
end
|
||||
|
||||
--- Run the offsets pass.
|
||||
--- For each canonical source, emits a per-module `<dir_basename>.offsets.h`
|
||||
--- containing constants for every marker recorded in atom.paths.
|
||||
--- @param ctx PassCtx
|
||||
--- @return PassResult
|
||||
function M.run(ctx)
|
||||
local outputs = {}
|
||||
local errors = {}
|
||||
local warnings = {}
|
||||
|
||||
local corpus = ctx.shared and ctx.shared.corpus
|
||||
if type(corpus) ~= "table" or type(corpus.source_order) ~= "table" then
|
||||
error("offsets.run requires ctx.shared.corpus.source_order (canonical corpus).", 0)
|
||||
end
|
||||
|
||||
for _, src in ipairs(corpus.source_order) do
|
||||
local out_path = process_source(ctx, src)
|
||||
if out_path then
|
||||
outputs[#outputs + 1] = { offsets_h = out_path }
|
||||
end
|
||||
end
|
||||
|
||||
return { outputs = outputs, errors = errors, warnings = warnings }
|
||||
end
|
||||
|
||||
return M
|
||||
@@ -0,0 +1,475 @@
|
||||
--- passes/report.lua — Per-MODULE annotation report renderer +
|
||||
--- project-wide summary writer.
|
||||
---
|
||||
--- Two output files per build:
|
||||
--- - `build/gen/<dir_basename>.annotations.txt` — one per source-directory containing atoms; aggregates across all sources in the directory.
|
||||
--- - `build/gen/annotation_validation.txt` — the project summary.
|
||||
---
|
||||
--- The annotation pass emits `errors.h` files per module and the canonical `corpus.sources_by_dir` projection groups sources by directory.
|
||||
--- This pass iterates the canonical dir projection directly and re-validates each source via `annotation.validate()` to get the detailed per-source results.
|
||||
---
|
||||
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
|
||||
--- Lua 5.3 compatible.
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Module-scope requires + package.path setup
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Resolve `arg[0]` to an absolute-ish script directory so that `require("duffle")` resolves against `scripts/` regardless of CWD.
|
||||
-- Bootstrap: see `ps1_meta.lua` for the rationale.
|
||||
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
||||
-- Uses `debug.getinfo` to find this file's own directory, so it works
|
||||
-- both standalone and when require'd from the orchestrator.
|
||||
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
|
||||
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
|
||||
-- Load the annotation pass so we can re-validate each source against the canonical corpus projection.
|
||||
-- The annotation pass exposes `M.validate`, which returns the per-source AnnotationResult (atoms / annots / macros / binds / errors / warnings)
|
||||
-- that the report pass renders into the per-module `<dir_basename>.annotations.txt` output.
|
||||
local annotation = dofile(_bootstrap_dir .. "annotation.lua")
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Constants
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Section separators used in the rendered text reports.
|
||||
-- The thin rules are hand-tuned to align with the per-section content width; do not change without also checking the section renderers below.
|
||||
local RULE_THICK = "========================================================"
|
||||
local SECTION_HEADER_ATOMS = "── Atoms ────────────────────────────────────────────────"
|
||||
local SECTION_HEADER_ANNOTS = "── Annotations ──────────────────────────────────────────"
|
||||
local SECTION_HEADER_BINDS = "── Binds_* structs ──────────────────────────────────────"
|
||||
local SECTION_HEADER_MACROS = "── Macro word-count declarations ─────────────────────────"
|
||||
local SECTION_HEADER_ERRORS = "── Errors ──────────────────────────────────────────────"
|
||||
local SECTION_HEADER_WARNINGS = "── Warnings ────────────────────────────────────────────"
|
||||
|
||||
-- Lua pattern that captures the basename (last path segment) of a
|
||||
-- forward- or back-slash separated path.
|
||||
local BASENAME_PATTERN = "([^/\\]+)$"
|
||||
|
||||
-- Debug flag name — set to truthy in `_G` to enable verbose logging.
|
||||
local DEBUG_FLAG = "_DEBUG_REPORT"
|
||||
|
||||
-- Pass identifier for log messages.
|
||||
local PASS_NAME = "report"
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Type declarations
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- @class SourceFile
|
||||
--- @field path string -- absolute path to the source file
|
||||
--- @field text string -- the full source text
|
||||
--- @field dir string -- the directory containing the source
|
||||
--- @field basename string -- filename without extension
|
||||
|
||||
--- @class PassCtx
|
||||
--- @field sources SourceFile[] -- all source files in the build
|
||||
--- @field metadata_path string -- path to word_count.metadata.h
|
||||
--- @field shared table -- cross-pass shared state
|
||||
--- @field out_root string -- output root (e.g. "build/gen")
|
||||
--- @field project_root string -- project root (e.g. "code/")
|
||||
--- @field upstream table<string, table> -- per-pass upstream outputs
|
||||
--- @field flags table -- CLI flags + per-pass stash
|
||||
--- @field verbose boolean -- if true, log diagnostic info
|
||||
|
||||
--- @class PassResult
|
||||
--- @field outputs table[] -- {kind=, path=} entries describing emit files
|
||||
--- @field errors table[] -- {line=, msg=} entries; build-stops
|
||||
--- @field warnings table[] -- {line=, msg=} entries; build-succeeds
|
||||
|
||||
-- Shapes produced by `passes/annotation.lua`'s `M.validate()`.
|
||||
|
||||
--- @class AtomEntry
|
||||
--- @field name string -- atom name (e.g. "cube_g4_face")
|
||||
--- @field line integer -- source line of the atom declaration
|
||||
|
||||
--- @class AnnotEntry
|
||||
--- @field line integer -- source line
|
||||
--- @field macro string -- the macro name (e.g. "atom_reads")
|
||||
--- @field name string -- the atom name (if a `name(...)` was given)
|
||||
--- @field kind string -- "atom_info" | "atom_bind" | ...
|
||||
--- @field binds string|nil -- Binds_X name if any
|
||||
--- @field reads string[] -- R_* names (read targets)
|
||||
--- @field writes string[] -- R_* names (write targets)
|
||||
--- @field error string|nil -- error message if annotation was malformed
|
||||
|
||||
--- @class BindsField
|
||||
--- @field name string -- field name
|
||||
--- @field offset integer -- byte offset within the Binds_X struct
|
||||
|
||||
--- @class BindsStruct
|
||||
--- @field name string -- struct name (e.g. "Binds_Floor")
|
||||
--- @field line integer -- source line of the typedef
|
||||
--- @field bytes integer -- total byte size
|
||||
--- @field fields BindsField[] -- the field list
|
||||
|
||||
--- @class MacroEntry
|
||||
--- @field name string -- macro name (e.g. "WORD_COUNT(my_macro, 4)")
|
||||
--- @field line integer -- source line
|
||||
--- @field words integer -- declared word count
|
||||
|
||||
--- @class Finding
|
||||
--- @field line integer -- source line
|
||||
--- @field msg string -- finding message
|
||||
|
||||
--- @class AnnotationResult
|
||||
--- @field source string -- set by this pass; original source path
|
||||
--- @field atoms AtomEntry[] -- atom declarations in this source
|
||||
--- @field annots AnnotEntry[] -- annotation entries
|
||||
--- @field macros MacroEntry[] -- macro word-count declarations
|
||||
--- @field binds BindsStruct[] -- Binds_* struct declarations
|
||||
--- @field errors Finding[] -- errors from validation
|
||||
--- @field warnings Finding[] -- warnings from validation
|
||||
--- @field info table -- info summary (not rendered here)
|
||||
|
||||
--- @class ModuleEntry
|
||||
--- @field dir string -- absolute directory path
|
||||
--- @field dir_basename string -- basename (e.g. "duffle", "gte_hello")
|
||||
--- @field atoms_count integer -- pre-counted atoms for filtering
|
||||
|
||||
--- @class ModuleReport
|
||||
--- @field dir string -- module directory
|
||||
--- @field sources SourceFile[] -- sources in this module
|
||||
--- @field results AnnotationResult[] -- per-source validate() results
|
||||
|
||||
--- @class ProjectReport
|
||||
--- @field results AnnotationResult[] -- all per-source results
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Per-MODULE annotation report (aggregated across all sources in a dir)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Extract the basename (last path segment) of a forward- or back-slash separated path. Returns the input unchanged if no separator is found.
|
||||
--- @param path string
|
||||
--- @return string
|
||||
local function source_basename(path)
|
||||
return path:match(BASENAME_PATTERN) or path
|
||||
end
|
||||
|
||||
--- (internal) Format a single annotation entry as one rendered line.
|
||||
--- @param a AnnotEntry
|
||||
--- @param src_name string
|
||||
--- @return string
|
||||
local function format_annot_line(a, src_name)
|
||||
if a.error then
|
||||
return string.format(" ✗ line %d %s [ERROR: %s] [%s]", a.line, a.macro or "?", a.error, src_name)
|
||||
end
|
||||
local line = string.format(" ● line %d %s [%s]", a.line, a.name, src_name)
|
||||
if a.binds then line = line .. " binds=" .. a.binds end
|
||||
if #a.reads > 0 then line = line .. " reads={" .. table.concat(a.reads, ",") .. "}" end
|
||||
if #a.writes > 0 then line = line .. " writes={" .. table.concat(a.writes, ",") .. "}" end
|
||||
return line
|
||||
end
|
||||
|
||||
--- (internal) Tally totals across all results in a module.
|
||||
--- @param results AnnotationResult[]
|
||||
--- @return integer, integer, integer, integer, integer, integer
|
||||
local function tally_module_totals(results)
|
||||
local total_atoms, total_annots, total_binds, total_macros = 0, 0, 0, 0
|
||||
local total_errors, total_warnings = 0, 0
|
||||
for _, r in ipairs(results) do
|
||||
total_atoms = total_atoms + #r.atoms
|
||||
total_annots = total_annots + #r.annots
|
||||
total_binds = total_binds + #r.binds
|
||||
total_macros = total_macros + #r.macros
|
||||
total_errors = total_errors + #r.errors
|
||||
total_warnings = total_warnings + #r.warnings
|
||||
end
|
||||
return total_atoms, total_annots, total_binds, total_macros, total_errors, total_warnings
|
||||
end
|
||||
|
||||
-- (internal) Section renderer: per-source atom declarations.
|
||||
local function render_module_atoms_section(add, results)
|
||||
add(SECTION_HEADER_ATOMS)
|
||||
for _, r in ipairs(results) do
|
||||
local src_name = source_basename(r.source)
|
||||
for _, a in ipairs(r.atoms) do
|
||||
add(string.format(" MipsAtom_(%s) line %d [%s]", a.name, a.line, src_name))
|
||||
end
|
||||
end
|
||||
add("")
|
||||
end
|
||||
|
||||
-- (internal) Section renderer: per-source annotation entries.
|
||||
local function render_module_annots_section(add, results)
|
||||
add(SECTION_HEADER_ANNOTS)
|
||||
for _, r in ipairs(results) do
|
||||
local src_name = source_basename(r.source)
|
||||
for _, a in ipairs(r.annots) do
|
||||
add(format_annot_line(a, src_name))
|
||||
end
|
||||
end
|
||||
add("")
|
||||
end
|
||||
|
||||
-- (internal) Section renderer: per-source Binds_* struct declarations.
|
||||
local function render_module_binds_section(add, results)
|
||||
add(SECTION_HEADER_BINDS)
|
||||
for _, r in ipairs(results) do
|
||||
local src_name = source_basename(r.source)
|
||||
for _, b in ipairs(r.binds) do
|
||||
add(string.format(" %s line %d %d bytes [%s]", b.name, b.line, b.bytes, src_name))
|
||||
for _, f in ipairs(b.fields) do
|
||||
add(string.format(" +%2d: %s", f.offset, f.name))
|
||||
end
|
||||
end
|
||||
end
|
||||
add("")
|
||||
end
|
||||
|
||||
-- (internal) Section renderer: per-source macro word-count declarations.
|
||||
local function render_module_macros_section(add, results)
|
||||
add(SECTION_HEADER_MACROS)
|
||||
for _, r in ipairs(results) do
|
||||
local src_name = source_basename(r.source)
|
||||
for _, m in ipairs(r.macros) do
|
||||
add(string.format(" %s line %d words=%d [%s]", m.name, m.line, m.words, src_name))
|
||||
end
|
||||
end
|
||||
add("")
|
||||
end
|
||||
|
||||
-- (internal) Section renderer: per-source errors (one-line + "(none)" if empty).
|
||||
local function render_module_errors_section(add, results, total_errors)
|
||||
add(SECTION_HEADER_ERRORS)
|
||||
if total_errors == 0 then
|
||||
add(" (none)")
|
||||
else
|
||||
for _, r in ipairs(results) do
|
||||
local src_name = source_basename(r.source)
|
||||
for _, e in ipairs(r.errors) do
|
||||
add(string.format(" ✗ line %d %s [%s]", e.line, e.msg, src_name))
|
||||
end
|
||||
end
|
||||
end
|
||||
add("")
|
||||
end
|
||||
|
||||
-- (internal) Section renderer: per-source warnings (one-line + "(none)" if empty).
|
||||
local function render_module_warnings_section(add, results, total_warnings)
|
||||
add(SECTION_HEADER_WARNINGS)
|
||||
if total_warnings == 0 then
|
||||
add(" (none)")
|
||||
else
|
||||
for _, r in ipairs(results) do
|
||||
local src_name = source_basename(r.source)
|
||||
for _, w in ipairs(r.warnings) do
|
||||
add(string.format(" ⚠ line %d %s [%s]", w.line, w.msg, src_name))
|
||||
end
|
||||
end
|
||||
end
|
||||
add("")
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- SECTION_RENDERERS — data-driven section dispatch (the plex pattern)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
--
|
||||
-- Each entry maps a section to its (header, render_fn). The render_fn signature:
|
||||
-- render_fn(add, results, totals)
|
||||
-- add -- the `add(line)` closure from the surrounding report renderer
|
||||
-- results -- AnnotationResult[] (per-source results)
|
||||
-- totals -- {atoms, annots, binds, macros, errors, warnings} counts
|
||||
--
|
||||
-- Sections that need to render "(none)" vs iterate use totals.errors / totals.warnings;
|
||||
-- other sections ignore the totals arg.
|
||||
-- Adding a new section = 1 row here + 1 render_<thing>_section function.
|
||||
local SECTION_RENDERERS = {
|
||||
{ header = SECTION_HEADER_ATOMS, render = render_module_atoms_section },
|
||||
{ header = SECTION_HEADER_ANNOTS, render = render_module_annots_section },
|
||||
{ header = SECTION_HEADER_BINDS, render = render_module_binds_section },
|
||||
{ header = SECTION_HEADER_MACROS, render = render_module_macros_section },
|
||||
{ header = SECTION_HEADER_ERRORS, render = function(add, results, totals) return render_module_errors_section(add, results, totals.errors) end },
|
||||
{ header = SECTION_HEADER_WARNINGS, render = function(add, results, totals) return render_module_warnings_section(add, results, totals.warnings) end },
|
||||
}
|
||||
|
||||
--- Render the per-MODULE annotation report (one `<dir_basename>.annotations.txt`).
|
||||
--- @param dir string -- module directory path
|
||||
--- @param sources SourceFile[] -- sources in this module
|
||||
--- @param results AnnotationResult[] -- per-source validate() results
|
||||
--- @return string -- the rendered report text
|
||||
local function render_module_report(dir, sources, results)
|
||||
local lines = {}
|
||||
local function add(s) lines[#lines + 1] = s end
|
||||
|
||||
add(RULE_THICK)
|
||||
add("ANNOTATION PASS — module " .. source_basename(dir))
|
||||
add(RULE_THICK)
|
||||
add(string.format("Sources: %d", #sources))
|
||||
for _, s in ipairs(sources) do add(" " .. s.path) end
|
||||
add("")
|
||||
|
||||
local total_atoms, total_annots, total_binds, total_macros, total_errors, total_warnings = tally_module_totals(results)
|
||||
add(string.format("Atoms: %d Annotations: %d Binds structs: %d Macro decls: %d",
|
||||
total_atoms, total_annots, total_binds, total_macros))
|
||||
add("")
|
||||
|
||||
-- Bundle the totals so the section renderers don't need separate parameter lists.
|
||||
-- Errors/warnings sections need their total count to decide "(none)" vs iterate.
|
||||
-- Sections without totals (atoms/annots/binds/macros) ignore this arg.
|
||||
local totals = {
|
||||
atoms = total_atoms, annots = total_annots, binds = total_binds,
|
||||
macros = total_macros, errors = total_errors, warnings = total_warnings,
|
||||
}
|
||||
|
||||
-- THE per-section dispatch. ONE loop over SECTION_RENDERERS.
|
||||
-- Each renderer writes its header + content via the `add` closure (pre-bound above).
|
||||
-- Adding a new section = 1 row here + 1 render_<thing>_section function.
|
||||
for _, section in ipairs(SECTION_RENDERERS) do
|
||||
section.render(add, results, totals)
|
||||
end
|
||||
|
||||
return table.concat(lines, "\n") .. "\n"
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Per-project summary
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Render the per-project summary (`build/gen/annotation_validation.txt`).
|
||||
--- Aggregates totals across all sources; lists per-source error counts if any source has errors.
|
||||
--- @param all_results AnnotationResult[]
|
||||
--- @return string
|
||||
local function render_project_report(all_results)
|
||||
local lines = {}
|
||||
local function add(s) lines[#lines + 1] = s end
|
||||
|
||||
local total_atoms, total_annots, total_macros, total_binds = 0, 0, 0, 0
|
||||
local total_errors, total_warnings = 0, 0
|
||||
for _, r in ipairs(all_results) do
|
||||
total_atoms = total_atoms + #r.atoms
|
||||
total_annots = total_annots + #r.annots
|
||||
total_macros = total_macros + #r.macros
|
||||
total_binds = total_binds + #r.binds
|
||||
total_errors = total_errors + #r.errors
|
||||
total_warnings = total_warnings + #r.warnings
|
||||
end
|
||||
|
||||
add(RULE_THICK)
|
||||
add("ANNOTATION VALIDATION — project summary")
|
||||
add(RULE_THICK)
|
||||
add("")
|
||||
add(string.format("Atoms: %d", total_atoms))
|
||||
add(string.format("Annotations: %d", total_annots))
|
||||
add(string.format("Macros: %d", total_macros))
|
||||
add(string.format("Binds: %d", total_binds))
|
||||
add("")
|
||||
add(string.format("Errors: %d", total_errors))
|
||||
add(string.format("Warnings: %d", total_warnings))
|
||||
add("")
|
||||
|
||||
if total_errors > 0 then
|
||||
add("Per-source error counts:")
|
||||
for _, r in ipairs(all_results) do
|
||||
if #r.errors > 0 then
|
||||
local src_name = source_basename(r.source)
|
||||
add(string.format(" %s : %d error(s)", src_name, #r.errors))
|
||||
end
|
||||
end
|
||||
add("")
|
||||
end
|
||||
|
||||
return table.concat(lines, "\n") .. "\n"
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Orchestration helpers
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- (internal) Re-validate every source in a directory against the canonical corpus projection.
|
||||
--- Calls `annotation.validate()` per source to produce the per-source AnnotationResult (atoms / annots / macros / binds / errors / warnings)
|
||||
--- that the report renderer consumes. Eeach report pass run is reproducible from the corpus.
|
||||
--- Returns the list of module results + the flat list of all results (for the project-wide summary).
|
||||
--- @param ctx PassCtx
|
||||
--- @param dir_sources SourceFile[]
|
||||
--- @return AnnotationResult[], AnnotationResult[]
|
||||
local function lookup_module_results(ctx, dir_sources)
|
||||
local module_results = {}
|
||||
local all_results = {}
|
||||
for _, src in ipairs(dir_sources) do
|
||||
if src.scan then
|
||||
local result = annotation.validate(ctx, src, nil)
|
||||
result.source = src.path -- tag for downstream rendering
|
||||
module_results[#module_results + 1] = result
|
||||
all_results[#all_results + 1] = result
|
||||
end
|
||||
end
|
||||
return module_results, all_results
|
||||
end
|
||||
|
||||
--- (internal) Does this module's results contain anything worth emitting?
|
||||
--- @param module_results AnnotationResult[]
|
||||
--- @return boolean
|
||||
local function module_has_content(module_results)
|
||||
for _, r in ipairs(module_results) do
|
||||
if #r.atoms > 0 or #r.annots > 0 or #r.binds > 0
|
||||
or #r.macros > 0 or #r.errors > 0 or #r.warnings > 0 then
|
||||
return true
|
||||
end
|
||||
end
|
||||
return false
|
||||
end
|
||||
|
||||
--- (internal) Log a debug message if `_G[DEBUG_FLAG]` is truthy.
|
||||
--- @param fmt string
|
||||
local function debug_log(fmt, ...)
|
||||
if _G[DEBUG_FLAG] then
|
||||
io.stderr:write(string.format("[%s] " .. fmt, PASS_NAME, ...))
|
||||
end
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- M — module exports
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
local M = {}
|
||||
|
||||
--- Run the report pass.
|
||||
--- Renders one `<dir_basename>.annotations.txt` per source-directory that has content, plus the project-wide `annotation_validation.txt` summary.
|
||||
--- @param ctx PassCtx
|
||||
--- @return PassResult
|
||||
function M.run(ctx)
|
||||
local outputs = {}
|
||||
local errors = {}
|
||||
local warnings = {}
|
||||
|
||||
-- Module grouping comes from `corpus.sources_by_dir` (the canonical projection).
|
||||
-- Iterate it directly; no private cache, no per-pass stash.
|
||||
local corpus = ctx.shared and ctx.shared.corpus
|
||||
local by_dir = (corpus and corpus.sources_by_dir) or {}
|
||||
|
||||
duffle.ensure_dir(ctx.out_root)
|
||||
|
||||
local all_results_for_summary = {}
|
||||
for dir, dir_sources in pairs(by_dir) do
|
||||
local dir_basename = dir:match("([^/\\]+)$") or dir
|
||||
debug_log("dir=%s basename=%s sources=%d\n", dir, dir_basename, #dir_sources)
|
||||
|
||||
if #dir_sources > 0 then
|
||||
local module_results, all_results = lookup_module_results(ctx, dir_sources)
|
||||
for _, r in ipairs(all_results) do
|
||||
all_results_for_summary[#all_results_for_summary + 1] = r
|
||||
end
|
||||
|
||||
if module_has_content(module_results) then
|
||||
local out_path = ctx.out_root .. "/" .. dir_basename .. ".annotations.txt"
|
||||
duffle.write_file(out_path, render_module_report(dir, dir_sources, module_results))
|
||||
outputs[#outputs + 1] = { annotations_txt = out_path }
|
||||
else
|
||||
debug_log(" -> no content; skipping\n")
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
if #all_results_for_summary > 0 then
|
||||
local summary_path = ctx.out_root .. "/annotation_validation.txt"
|
||||
duffle.write_file(summary_path, render_project_report(all_results_for_summary))
|
||||
outputs[#outputs + 1] = { summary_txt = summary_path }
|
||||
end
|
||||
|
||||
return { outputs = outputs, errors = errors, warnings = warnings }
|
||||
end
|
||||
|
||||
return M
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,128 @@
|
||||
--- word_count_eval.lua — Word-counting logic for the tape-atom metaprogram pipeline.
|
||||
---
|
||||
--- Two responsibilities:
|
||||
--- 1. **Public utility** `M.count_token_words(token, wc)`: Used by `passes/offsets.lua`, `passes/annotation.lua`, and other passes.
|
||||
--- 2. **Pass entry** `M.run(ctx)`: Loads the authored `word_count.metadata.h` into `ctx.shared.corpus.word_counts` for downstream passes.
|
||||
--- The generated `.macs.h` files are OUTPUT artifacts and are NOT inputs to this pass;
|
||||
--- Current component counts are owned by `passes/components.lua` (which populates `corpus.word_counts` and `corpus.component_body_index`
|
||||
--- AFTER computing each current count from the just-built body + `corpus.word_counts`).
|
||||
---
|
||||
--- **Canonical contract**:
|
||||
--- * `ctx.shared.corpus.word_counts` is the count table.
|
||||
--- * `corpus.word_counts` is the sole count table. Consumers read `corpus.word_counts` directly.
|
||||
--- * `ctx.shared.components` and `ctx.shared.component_body_index` are NOT created by this pass (projections only).
|
||||
--- * No `.macs.h` recursive discovery (no `scan_dir`, no scan cache, no `_invalidate_scan_cache`).
|
||||
---
|
||||
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
|
||||
--- Lua 5.3 compatible.
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Module-scope requires + package.path setup
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
||||
-- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator.
|
||||
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Type declarations
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- @class WordCounts
|
||||
--- @field [string] integer -- macro name -> word count
|
||||
|
||||
--- @class SourceFile
|
||||
--- @field path string -- absolute path to the source file
|
||||
--- @field text string -- the full source text
|
||||
--- @field dir string -- the directory containing the source
|
||||
--- @field basename string -- filename without extension
|
||||
|
||||
--- @class PassCtx
|
||||
--- @field sources SourceFile[] -- all source files in the build
|
||||
--- @field metadata_path string -- path to word_count.metadata.h
|
||||
--- @field shared table -- cross-pass shared state
|
||||
--- @field shared.corpus table -- canonical corpus (required)
|
||||
--- @field shared.corpus.word_counts WordCounts -- canonical count table (populated by this pass)
|
||||
--- @field out_root string -- output root (e.g. "build/gen")
|
||||
--- @field project_root string -- project root (e.g. "code/")
|
||||
--- @field upstream table<string, table> -- per-pass upstream outputs
|
||||
--- @field flags table -- CLI flags
|
||||
--- @field verbose boolean -- if true, log diagnostic info
|
||||
|
||||
--- @class PassResult
|
||||
--- @field outputs table[] -- {kind=, path=} entries describing emit files
|
||||
--- @field errors table[] -- {line=, msg=} entries; build-stops
|
||||
--- @field warnings table[] -- {line=, msg=} entries; build-succeeds
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Module exports
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
local M = {}
|
||||
|
||||
-- ┌────────────────────────────────────────────────────────────────────┐
|
||||
-- │ Shared utility: count_token_words │
|
||||
-- └────────────────────────────────────────────────────────────────────┘
|
||||
|
||||
--- Count words emitted by a single comma-separated token inside an atom body.
|
||||
--- For most tokens (regular MIPS instructions) this returns 1.
|
||||
--- For `mac_X(...)` calls, this returns the resolved word count from `wc` (recursively if needed). For `nop2` etc., returns wc[name].
|
||||
--- For unknown macros, returns 1 and (optionally) warns.
|
||||
--- @param token string -- a single token from split_top_level_commas
|
||||
--- @param wc WordCounts -- the shared word-count table
|
||||
--- @return integer
|
||||
function M.count_token_words(token, wc)
|
||||
local s = duffle.trim(token)
|
||||
if s == "" then return 0 end
|
||||
local name, after = duffle.read_ident(s, 1)
|
||||
if not name then return 1 end
|
||||
if wc[name] then return wc[name] end
|
||||
local paren_pos = duffle.skip_ws_and_cmt(s, after)
|
||||
if s:sub(paren_pos, paren_pos) == "(" then
|
||||
io.stderr:write(" warning: unknown macro '" .. name .. "', assuming 1 word\n")
|
||||
end
|
||||
return 1
|
||||
end
|
||||
|
||||
-- ┌────────────────────────────────────────────────────────────────────┐
|
||||
-- │ Pass entry: M.run(ctx) — "word-counts" pass │
|
||||
-- └────────────────────────────────────────────────────────────────────┘
|
||||
|
||||
--- Load the authored `word_count.metadata.h` into `ctx.shared.corpus.word_counts`.
|
||||
--- Generated `.macs.h` files are OUTPUT artifacts and are NOT scanned as inputs.
|
||||
--- Current component counts are computed and inserted by `passes/components.lua`
|
||||
--- after the components pass iterates `corpus.source_order` and writes each source's `<dir_basename>.macs.h` file.
|
||||
---
|
||||
--- Contract:
|
||||
--- * `ctx.shared.corpus` MUST exist (canonical corpus ownership).
|
||||
--- * `ctx.metadata_path` MUST be a readable file path to the authored `word_count.metadata.h`.
|
||||
--- * The pass assigns exactly one table to `corpus.word_counts`.
|
||||
--- Consumers read the corpus-owned table directly.
|
||||
--- Consumers must read `corpus.word_counts` directly.
|
||||
--- @param ctx PassCtx
|
||||
--- @return PassResult
|
||||
function M.run(ctx)
|
||||
-- 1. Canonical-corpus ownership gate.
|
||||
local corpus = ctx.shared and ctx.shared.corpus
|
||||
if type(corpus) ~= "table" then
|
||||
error("word_count_eval.run requires ctx.shared.corpus (canonical corpus). The fixture must install the corpus before running this pass.", 0)
|
||||
end
|
||||
|
||||
-- 2. metadata_path gate.
|
||||
if type(ctx.metadata_path) ~= "string" or ctx.metadata_path == "" then
|
||||
error("word_count_eval.run requires ctx.metadata_path (path to the authored word_count.metadata.h).", 0)
|
||||
end
|
||||
|
||||
-- 3. Load authored metadata. Generated .macs.h files are NOT scanned
|
||||
-- (the pass computes their counts from the just-built bodies after disk emission; see passes/components.lua).
|
||||
local wc = duffle.load_word_counts(ctx.metadata_path)
|
||||
|
||||
-- 4. Assign the count table. ONE assignment, no copy. The assignment creates no secondary alias.
|
||||
corpus.word_counts = wc
|
||||
|
||||
return { outputs = {}, errors = {}, warnings = {} }
|
||||
end
|
||||
|
||||
return M
|
||||
Binary file not shown.
@@ -0,0 +1,51 @@
|
||||
-- autoexec.lua - pcsx_debug_helper plugin entry point.
|
||||
-- Packaged in scripts/pcsx_debug_helper.zip. Loaded by pcsx-redux via the -archive CLI flag (see scripts/launch_pcsx_debug.ps1).
|
||||
--
|
||||
-- Registers two web handlers for external CLI tools:
|
||||
-- /api/v1/lua/gte - full GTE state (32 data + 32 control regs + PC)
|
||||
-- /api/v1/lua/gp - GP state summary (screenshot endpoint + VRAM endpoint refs)
|
||||
--
|
||||
-- The GTE handler reads COP2 regs via PCSX.getRegisters().CP2D/CP2C.
|
||||
-- The pcsx-redux gdb stub doesn't expose COP2, so this is the only way for external tools to see GTE state.
|
||||
--
|
||||
-- The GP handler is a thin pointer:
|
||||
-- pcsx-redux's Lua API exposes only PCSX.GPU.takeScreenShot() (no GPUSTAT, no GP0/GP1 command log, no display state). For richer GP state, the existing web endpoints are the practical path:
|
||||
-- /api/v1/state/still - PNG screenshot
|
||||
-- /api/v1/gpu/vram/raw - VRAM raw bytes (1MB)
|
||||
--
|
||||
-- Companion: scripts/gdb/gdb_tape_atoms.gdb (covers GPRs + atom-aware stepping).
|
||||
|
||||
local function register_handlers()
|
||||
if not PCSX.WebServer then PCSX.WebServer = {} end
|
||||
if not PCSX.WebServer.Handlers then PCSX.WebServer.Handlers = {} end
|
||||
|
||||
-- ── GTE state ──
|
||||
PCSX.WebServer.Handlers.gte = function(req)
|
||||
local r = PCSX.getRegisters()
|
||||
local out = { "pc=0x" .. string.format("%x", r.pc) }
|
||||
for i = 0, 31 do
|
||||
out[#out + 1] = string.format("D[%d]=0x%08x C[%d]=0x%08x",
|
||||
i, r.CP2D.r[i], i, r.CP2C.r[i])
|
||||
end
|
||||
return table.concat(out, "\n")
|
||||
end
|
||||
|
||||
-- ── GP state (pointer to existing endpoints) ──
|
||||
-- pcsx-redux's Lua GPU API exposes only takeScreenShot(); no GPUSTAT / GP0 / GP1 command log / display state.
|
||||
-- We point to the existing web endpoints that DO expose those (when the emulator is actually rendering. Paused-at-BP frames won't have a fresh frame).
|
||||
PCSX.WebServer.Handlers.gp = function(req)
|
||||
local out = {
|
||||
"gpu_screenshot_png=http://localhost:8080/api/v1/state/still",
|
||||
"vram_raw=http://localhost:8080/api/v1/gpu/vram/raw (1MB VRAM)",
|
||||
"gpustat=NOT_AVAILABLE_VIA_LUA",
|
||||
"gp_command_log=NOT_AVAILABLE_VIA_LUA (use pcsx-redux Debug > GPU Logger)",
|
||||
"hint_run_emulator_unpaused_for_screenshot",
|
||||
}
|
||||
return table.concat(out, "\n")
|
||||
end
|
||||
end
|
||||
|
||||
local ok, err = pcall(register_handlers)
|
||||
if ok then print("[pcsx_debug_helper] handlers registered: gte, gp")
|
||||
else print("[pcsx_debug_helper] registration failed: " .. tostring(err))
|
||||
end
|
||||
@@ -0,0 +1,771 @@
|
||||
--- ps1_meta.lua — Orchestrator entry point for the tape-atom metaprogram.
|
||||
---
|
||||
--- Dispatches to pass modules under `scripts/passes/`, resolving dependencies topologically (Kahn's algorithm + cycle detection).
|
||||
---
|
||||
--- **Architecture**:
|
||||
--- - **PASSES table** — declarative dep graph (data, not code).
|
||||
--- - **FLAG_HANDLERS table** — maps CLI flags to handlers.
|
||||
--- - **parse_args** → **build_ctx** (resolves unity/direct includes or exact sources; no semantic scanning) → **topo_sort** → **dispatch_passes**.
|
||||
--- - The first pass in the dep graph is `scan-source` (see `passes/scan_source.lua`).
|
||||
--- It calls `duffle.scan_source` once per source to produce the fat `SourceScan` payload, which is attached to each `src.scan`.
|
||||
--- Every other pass that reads source structure depends on `scan-source` and consumes `src.scan` as a read-only.
|
||||
---
|
||||
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
|
||||
--- Lua 5.3 compatible.
|
||||
---
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Module-scope requires + package.path setup
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Bootstrap: load `duffle_paths.lua` via this script's own path.
|
||||
-- Use `arg[0]` when this file is the entry script (`arg[0]` ends in "ps1_meta.lua");
|
||||
-- fall back to `debug.getinfo(1, "S").source` when this file is being dofile()'d or require()'d (in which case `arg[0]` is the *caller's* path, not ours).
|
||||
-- That single statement: (a) sets `package.path` + `package.cpath`, (b) at the bottom returns `require("duffle")`.
|
||||
-- So the dofile's return value is the duffle module.
|
||||
local _is_entry_script = arg and arg[0] and arg[0]:match("ps1_meta%.lua$") ~= nil
|
||||
local _bootstrap_src
|
||||
if _is_entry_script then
|
||||
_bootstrap_src = arg[0]
|
||||
else
|
||||
-- debug.getinfo(1, "S").source returns "@<path>" for the current chunk;
|
||||
-- strip the leading "@" so the directory match works in both cases.
|
||||
_bootstrap_src = debug.getinfo(1, "S").source:sub(2)
|
||||
end
|
||||
local duffle = dofile((_bootstrap_src:match("(.*[/\\])") or "./") .. "duffle_paths.lua")
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Constants
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Exit codes (per the --help text and the post-build summary convention).
|
||||
local EXIT_OK = 0
|
||||
local EXIT_VALIDATION_ERRORS = 1
|
||||
local EXIT_INTERNAL_ERROR = 2
|
||||
|
||||
-- Default --out-root value if not provided.
|
||||
local DEFAULT_OUT_ROOT = "build/gen"
|
||||
|
||||
-- Sentinel for "all passes" in `PASS_FLAG_TO_NAME`. Distinguishes `--all` from the per-pass flags (which map to individual pass names).
|
||||
local ALL_PASSES_SENTINEL = "__all__"
|
||||
|
||||
-- Sentinel key for the pass-flag dispatcher in `FLAG_HANDLERS`.
|
||||
-- The actual pass names are looked up via `PASS_FLAG_TO_NAME`, not direct dispatch, so this key never matches a real flag.
|
||||
local PASS_FLAG_DISPATCH_KEY = "__pass__"
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Type declarations
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- @class PassDescriptor
|
||||
--- @field module string -- module name passed to require()
|
||||
--- @field kind string -- "shared" | "header-output" | "validation" | "diagnostic" | "report"
|
||||
--- -- Report severity is independent from process exit policy (see PASS_KIND_STOP_ON_ERROR).
|
||||
--- @field deps string[] -- names of upstream passes
|
||||
--- @field groups string[]? -- OPTIONAL build-phase groups this pass is a root of
|
||||
--- -- (e.g. { "pre-link" }, { "post-link" }); absent ⇒ dependency-only
|
||||
|
||||
--- @class SourceFile
|
||||
--- @field path string -- absolute path to the source file
|
||||
--- @field text string -- the full source text
|
||||
--- @field dir string -- the directory containing the source
|
||||
--- @field basename string -- filename without extension
|
||||
|
||||
--- @class PassCtx
|
||||
--- @field metadata_path string -- path to word_count.metadata.h
|
||||
--- @field shared table -- cross-pass shared state
|
||||
--- @field shared.corpus table -- canonical authored-source/project projection
|
||||
--- @field out_root string -- output root (e.g. "build/gen")
|
||||
--- @field project_root string -- PS1 repository root
|
||||
--- @field flags table -- CLI flags + per-pass stash
|
||||
--- @field verbose boolean -- if true, log diagnostic info
|
||||
|
||||
--- @class Finding
|
||||
--- @field line integer -- source line (or 0 for pass-level)
|
||||
--- @field msg string -- finding message
|
||||
|
||||
--- @class PassResult
|
||||
--- @field outputs PassOutputEntry[] -- emitted file paths
|
||||
--- @field errors Finding[] -- build-stops (per-pass kind policy)
|
||||
--- @field warnings Finding[] -- informational
|
||||
|
||||
--- @class ParsedArgs
|
||||
--- @field requested_set string[] -- pass names to run (explicit --all expanded)
|
||||
--- @field sources string[] -- exact --source values, retained in CLI order
|
||||
--- @field unity_root string|nil -- --unity-root value; mutually exclusive with sources
|
||||
--- @field metadata string -- --metadata value
|
||||
--- @field out_root string -- --out-root value (default "build/gen")
|
||||
--- @field project_root string -- PS1 repository root (derived from metadata by default)
|
||||
--- @field verbose boolean -- if true, log diagnostic info
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- PASSES Table
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Build-phase groups: Each PASSES row may declare membership in one or more named groups via `groups = { ... }`.
|
||||
-- The CLI flags --pre-link and --post-link request the *roots* of their group; topo_sort then closes transitive dependencies from those roots,
|
||||
-- and dispatch_passes runs every pass in the resulting closure without phase-filtering.
|
||||
--
|
||||
-- A row without a `groups` entry is dependency-only: it runs only when a transitive dep requests it,
|
||||
-- but it remains directly requestable through its explicit CLI flag (e.g. --atoms-source-map, --scan-source).
|
||||
|
||||
local PASSES = {
|
||||
["scan-source"] = {
|
||||
module = "passes.scan_source",
|
||||
kind = "shared", deps = {},
|
||||
},
|
||||
["word-counts"] = {
|
||||
module = "passes.word_count_eval",
|
||||
kind = "shared", deps = {},
|
||||
},
|
||||
components = {
|
||||
module = "passes.components",
|
||||
kind = "header-output",
|
||||
deps = {"scan-source", "word-counts"},
|
||||
},
|
||||
["emission-model"] = {
|
||||
module = "passes.emission_model",
|
||||
kind = "validation",
|
||||
deps = {"components"},
|
||||
},
|
||||
annotation = {
|
||||
module = "passes.annotation",
|
||||
kind = "validation",
|
||||
deps = {"scan-source", "word-counts"},
|
||||
},
|
||||
offsets = {
|
||||
module = "passes.offsets",
|
||||
kind = "header-output",
|
||||
deps = {"scan-source", "word-counts", "components", "emission-model"},
|
||||
groups = { "pre-link" },
|
||||
},
|
||||
["static-analysis"] = {
|
||||
module = "passes.static_analysis",
|
||||
-- "diagnostic" — every `error`/`warning` finding is written to the report file;
|
||||
-- the orchestrator does NOT exit non-zero on these findings (see PASS_KIND_STOP_ON_ERROR).
|
||||
-- Report severity is independent from process exit policy.
|
||||
kind = "diagnostic",
|
||||
deps = {"scan-source", "word-counts", "components", "emission-model"},
|
||||
},
|
||||
["atoms-source-map"] = {
|
||||
module = "passes.atoms_source_map",
|
||||
kind = "header-output",
|
||||
deps = {"word-counts", "components", "emission-model"},
|
||||
},
|
||||
["dwarf-injection"] = {
|
||||
module = "passes.dwarf_injection",
|
||||
kind = "shared",
|
||||
deps = {"scan-source", "atoms-source-map"},
|
||||
groups = { "post-link" },
|
||||
},
|
||||
report = {
|
||||
module = "passes.report",
|
||||
kind = "report",
|
||||
deps = {"annotation", "static-analysis"},
|
||||
groups = { "pre-link" },
|
||||
},
|
||||
}
|
||||
|
||||
-- ────────────────────────────────────────────────────────────────────────────
|
||||
-- Phase-root selection: derive the sorted set of roots belonging to a named build-phase group, then append them to `args.requested_set`.
|
||||
-- topo_sort closes the transitive deps from there; dispatch_passes runs every resolved pass without phase-filtering.
|
||||
-- ────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
--- @param group_name string -- the build-phase group ("pre-link" | "post-link")
|
||||
--- @return string[] -- sorted root pass names belonging to that group
|
||||
local function roots_for_group(group_name)
|
||||
local names = {}
|
||||
for name, pass in pairs(PASSES) do
|
||||
if pass.groups then
|
||||
for _, g in ipairs(pass.groups) do
|
||||
if g == group_name then
|
||||
names[#names + 1] = name
|
||||
break
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
table.sort(names)
|
||||
return names
|
||||
end
|
||||
|
||||
--- Append every root belonging to `group_name` to `args.requested_set`.
|
||||
--- Errors loudly if no PASSES row declares the group, so a typo'd or future-removed group name
|
||||
--- cannot silently fall through to pre-link (or any other default) and dispatch nothing.
|
||||
--- @param args ParsedArgs
|
||||
--- @param group_name string
|
||||
local function request_roots_for_group(args, group_name)
|
||||
local roots = roots_for_group(group_name)
|
||||
if #roots == 0 then
|
||||
error(string.format("ps1_meta: build-phase group %q has zero roots in PASSES; check PASSES rows for a `groups = { %q }` field"
|
||||
, group_name, group_name))
|
||||
end
|
||||
for _, name in ipairs(roots) do
|
||||
args.requested_set[#args.requested_set + 1] = name
|
||||
end
|
||||
end
|
||||
|
||||
-- Pass-kind taxonomy: Which kinds stop the build on errors?
|
||||
--
|
||||
-- Report severity is independent from process exit policy. A "diagnostic" pass still writes every `error`/`warning` finding into its report file,
|
||||
-- but `report_validation_errors` returns early for non-stopping kinds, so nothing is printed to stderr and the orchestrator does not exit non-zero.
|
||||
-- Adding a new pass kind requires listing it here explicitly; an unknown kind must not silently fall back to "true".
|
||||
local PASS_KIND_STOP_ON_ERROR = {
|
||||
["shared"] = false,
|
||||
["header-output"] = true,
|
||||
["validation"] = true,
|
||||
["diagnostic"] = false,
|
||||
["report"] = false,
|
||||
}
|
||||
|
||||
-- Closed set of CLI flags -> pass names.
|
||||
-- Per-pass flags (e.g. --word-counts) live here; phase flags (--pre-link, --post-link, --all)
|
||||
-- live in FLAG_HANDLERS because they own side effects or invoke group-derivation logic.
|
||||
-- dwarf-injection is *also* a per-pass opt-in flag, but its selection + opt-in state are both owned by the explicit FLAG_HANDLERS entry below
|
||||
-- (it sets args.flags.dwarf_injection and appends "dwarf-injection" to requested_set), so it is intentionally absent from this table.
|
||||
local PASS_FLAG_TO_NAME = {
|
||||
["--word-counts"] = "word-counts",
|
||||
["--components"] = "components",
|
||||
["--validate"] = "annotation",
|
||||
["--offsets"] = "offsets",
|
||||
["--static-analysis"] = "static-analysis",
|
||||
["--atoms-source-map"] = "atoms-source-map",
|
||||
["--report"] = "report",
|
||||
["--scan-source"] = "scan-source",
|
||||
["--all"] = ALL_PASSES_SENTINEL,
|
||||
}
|
||||
|
||||
--- Append every pass name to args.requested_set.
|
||||
--- Names are derived from PASSES (no parallel name list); used by --all and by any caller that wants the full closure.
|
||||
--- @param args ParsedArgs
|
||||
local function request_all_passes(args)
|
||||
local names = {}
|
||||
for name in pairs(PASSES) do names[#names + 1] = name end
|
||||
table.sort(names)
|
||||
for _, n in ipairs(names) do
|
||||
args.requested_set[#args.requested_set + 1] = n
|
||||
end
|
||||
end
|
||||
|
||||
-- Per-flag handlers. Each handler takes (args, argv, arg_idx) and returns the new arg_idx (so multi-arg flags like --source FILE advance it).
|
||||
-- Returning nil + os.exit() handles termination flags (--help).
|
||||
|
||||
local FLAG_HANDLERS = {}
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- CLI parsing
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Print the CLI usage to stdout and exit 0.
|
||||
local function print_help()
|
||||
io.write([[
|
||||
ps1_meta.lua - Tape-atom metaprogram orchestrator
|
||||
|
||||
USAGE:
|
||||
ps1_meta.lua [PASS_FLAGS] [COMMON_FLAGS]
|
||||
|
||||
PASS_FLAGS:
|
||||
Pick a phase or one-or-more individual passes:
|
||||
--pre-link [phase; default] Run the pre-link group + transitive deps.
|
||||
The root set is data-driven from each PASSES row's groups` field; no parallel name list is maintained.
|
||||
--post-link [phase] Run the post-link group + transitive deps.
|
||||
Requires --elf. Sets --gdb-runtime and --dwarf-injection opt-in flags as well.
|
||||
--all Select every row of the PASSES table. Pass-local opt-in guards remain active, so --dwarf-injection still requires
|
||||
--elf and --gdb-runtime still requires a runtime emission.
|
||||
Or pick any subset:
|
||||
--scan-source Scan sources into the fat SourceScan payload
|
||||
--word-counts Load metadata.h + scan for existing .macs.h
|
||||
--components Generate <module>/gen/<basename>.macs.h
|
||||
--validate Run atom annotation DSL validation
|
||||
--offsets Generate <module>/gen/<basename>.offsets.h
|
||||
--atoms-source-map Generate <basename>.atoms.sourcemap.txt per source
|
||||
--dwarf-injection [opt-in] Select the post-link dwarf-injection pass + set the opt-in flag. Requires --elf.
|
||||
--static-analysis Static analysis: GTE pipeline-fill, mac_yield, ABI handoff, cycle budget
|
||||
--report Render per-project summary
|
||||
|
||||
COMMON_FLAGS:
|
||||
--unity-root FILE Unity source root: load root + direct quoted authored includes only. Mutually exclusive with --source.
|
||||
--source FILE Exact source file to process (repeatable, never expands includes). Mutually exclusive with --unity-root.
|
||||
--metadata PATH Path to metadata.h (required)
|
||||
--out-root DIR Output root for reports (default: build/gen)
|
||||
--project-root DIR PS1 repository root (default: derived from <repo>/code/duffle/word_count.metadata.h)
|
||||
--gdb-runtime Also emit <out_root>/gdb_tape_atoms_runtime.gdb (post-link, requires --elf)
|
||||
--elf PATH Path to linked .elf (for --gdb-runtime / --dwarf-injection)
|
||||
--verbose Print per-pass debug output
|
||||
--help Show this help and exit
|
||||
|
||||
EXIT CODES:
|
||||
0 All requested passes succeeded
|
||||
1 Validation errors found
|
||||
2 Metaprogram internal error
|
||||
|
||||
EXAMPLES:
|
||||
ps1_meta.lua --pre-link --metadata code/duffle/word_count.metadata.h --unity-root code/gte_hello/hello_gte.c
|
||||
ps1_meta.lua --post-link --metadata code/duffle/word_count.metadata.h --unity-root code/gte_hello/hello_gte.c --elf build/hello_gte.elf
|
||||
ps1_meta.lua --all --metadata metadata.h --source code/foo.c --source code/bar.c
|
||||
]])
|
||||
end
|
||||
|
||||
local FLAG_VALUE_NAMES = {
|
||||
["--source"] = "FILE",
|
||||
["--unity-root"] = "FILE",
|
||||
["--metadata"] = "PATH",
|
||||
["--out-root"] = "DIR",
|
||||
["--project-root"] = "DIR",
|
||||
["--elf"] = "PATH",
|
||||
}
|
||||
|
||||
local function require_flag_value(argv, arg_idx, flag)
|
||||
local value = argv[arg_idx + 1]
|
||||
local next_known = type(value) == "string"
|
||||
and (FLAG_HANDLERS[value] ~= nil or PASS_FLAG_TO_NAME[value] ~= nil)
|
||||
if value == nil or next_known then
|
||||
io.stderr:write("ps1_meta: " .. flag .. " requires "
|
||||
.. FLAG_VALUE_NAMES[flag] .. "\n")
|
||||
os.exit(EXIT_INTERNAL_ERROR)
|
||||
end
|
||||
return value, arg_idx + 1
|
||||
end
|
||||
|
||||
-- Per-flag handlers. Each takes (args, argv, arg_idx) and returns the new arg_idx (so multi-arg flags like --source FILE advance it).
|
||||
-- Termination flags like --help call os.exit() instead.
|
||||
-- Populated AFTER print_help so the --help handler can reference it as an upvalue (Lua resolves locals at closure-call time,
|
||||
-- but if the closure is defined before the local, it falls back to _G).
|
||||
FLAG_HANDLERS["--help"] = function(args)
|
||||
print_help()
|
||||
os.exit(0)
|
||||
end
|
||||
|
||||
FLAG_HANDLERS["--verbose"] = function(args) args.verbose = true end
|
||||
FLAG_HANDLERS["--source"] = function(args, argv, arg_idx)
|
||||
local value, value_idx = require_flag_value(argv, arg_idx, "--source")
|
||||
args.sources[#args.sources + 1] = value
|
||||
return value_idx
|
||||
end
|
||||
FLAG_HANDLERS["--unity-root"] = function(args, argv, arg_idx)
|
||||
local value, value_idx = require_flag_value(argv, arg_idx, "--unity-root")
|
||||
args.unity_root = value
|
||||
return value_idx
|
||||
end
|
||||
FLAG_HANDLERS["--metadata"] = function(args, argv, arg_idx)
|
||||
local value, value_idx = require_flag_value(argv, arg_idx, "--metadata")
|
||||
args.metadata = value
|
||||
return value_idx
|
||||
end
|
||||
FLAG_HANDLERS["--out-root"] = function(args, argv, arg_idx)
|
||||
local value, value_idx = require_flag_value(argv, arg_idx, "--out-root")
|
||||
args.out_root = value
|
||||
return value_idx
|
||||
end
|
||||
FLAG_HANDLERS["--project-root"] = function(args, argv, arg_idx)
|
||||
local value, value_idx = require_flag_value(argv, arg_idx, "--project-root")
|
||||
args.project_root = value
|
||||
return value_idx
|
||||
end
|
||||
|
||||
-- Per-pass stash flags. Read by `passes/atoms_source_map.lua` to opt into the post-link gdb-runtime emission.
|
||||
-- Same shape as the existing per-flag handlers. mutates `args.flags` (which propagates into `ctx.flags`).
|
||||
FLAG_HANDLERS["--gdb-runtime"] = function(args)
|
||||
args.flags = args.flags or {}
|
||||
args.flags.gdb_runtime = true
|
||||
end
|
||||
FLAG_HANDLERS["--elf"] = function(args, argv, arg_idx)
|
||||
local value, value_idx = require_flag_value(argv, arg_idx, "--elf")
|
||||
args.flags = args.flags or {}
|
||||
args.flags.elf_path = value
|
||||
return value_idx
|
||||
end
|
||||
-- Enable DWARF injection (default OFF). Opts in to the post-link pass and sets the flag in one shot.
|
||||
-- The explicit handler below owns both selection and opt-in state, so --dwarf-injection is intentionally absent from PASS_FLAG_TO_NAME.
|
||||
FLAG_HANDLERS["--dwarf-injection"] = function(args)
|
||||
args.flags = args.flags or {}
|
||||
args.flags.dwarf_injection = true
|
||||
args.requested_set[#args.requested_set + 1] = "dwarf-injection"
|
||||
end
|
||||
-- Build-phase flags: --pre-link and --post-link request the roots of their declared groups (see roots_for_group).
|
||||
-- topo_sort closes transitive deps from those roots; dispatch_passes runs every pass in the resolved closure without phase-filtering.
|
||||
FLAG_HANDLERS["--pre-link"] = function(args)
|
||||
request_roots_for_group(args, "pre-link")
|
||||
end
|
||||
-- Batch post-link phase: gdb-runtime + dwarf-injection in one luajit cold start.
|
||||
-- Sets the same opt-in flags as --gdb-runtime + --dwarf-injection and selects the post-link build-phase group.
|
||||
-- elf is required; parse_args enforces it after all flags are parsed.
|
||||
FLAG_HANDLERS["--post-link"] = function(args)
|
||||
args.flags = args.flags or {}
|
||||
args.flags.gdb_runtime = true
|
||||
args.flags.dwarf_injection = true
|
||||
request_roots_for_group(args, "post-link")
|
||||
end
|
||||
|
||||
-- `--dwarf-injection` also emits atom-local debug data.
|
||||
|
||||
-- Pass-flag handler. Reads the closed-set table, expands --all, appends to requested_set.
|
||||
FLAG_HANDLERS[PASS_FLAG_DISPATCH_KEY] = function(args, a)
|
||||
local name = PASS_FLAG_TO_NAME[a]
|
||||
if name == ALL_PASSES_SENTINEL then
|
||||
request_all_passes(args)
|
||||
return
|
||||
end
|
||||
args.requested_set[#args.requested_set + 1] = name
|
||||
end
|
||||
|
||||
--- Parse argv into a structured table. Validates against a closed enum.
|
||||
--- @param argv string[]
|
||||
--- @return ParsedArgs
|
||||
local function parse_args(argv)
|
||||
local args = {
|
||||
requested_set = {},
|
||||
sources = {},
|
||||
unity_root = nil,
|
||||
metadata = nil,
|
||||
out_root = DEFAULT_OUT_ROOT,
|
||||
project_root = nil,
|
||||
verbose = false,
|
||||
}
|
||||
|
||||
local pos = 1
|
||||
while pos <= #argv do
|
||||
local a = argv[pos]
|
||||
local handler = FLAG_HANDLERS[a]
|
||||
if handler then
|
||||
pos = handler(args, argv, pos) or pos
|
||||
elseif PASS_FLAG_TO_NAME[a] then
|
||||
FLAG_HANDLERS[PASS_FLAG_DISPATCH_KEY](args, a)
|
||||
else
|
||||
io.stderr:write("ps1_meta: unknown flag '" .. a .. "'\n")
|
||||
io.stderr:write("Run with --help for usage.\n")
|
||||
os.exit(EXIT_INTERNAL_ERROR)
|
||||
end
|
||||
pos = pos + 1
|
||||
end
|
||||
|
||||
-- Default: --pre-link if no explicit pass flags were given.
|
||||
-- The first invocation of a build is always pre-link, so this avoids silently also invoking post-link work in builds without an ELF artifact.
|
||||
if #args.requested_set == 0 then request_roots_for_group(args, "pre-link") end
|
||||
|
||||
if not args.metadata then
|
||||
io.stderr:write("ps1_meta: --metadata PATH is required\n")
|
||||
os.exit(EXIT_INTERNAL_ERROR)
|
||||
end
|
||||
|
||||
-- `<repo>/code/duffle/word_count.metadata.h` is the canonical metadata location.
|
||||
-- `project_root` names `<repo>`; the resolver derives `<project_root>/code` separately.
|
||||
if not args.project_root then
|
||||
local metadata_dir = duffle.dirname(duffle.normalize_path(args.metadata))
|
||||
local code_root = duffle.dirname(metadata_dir)
|
||||
args.project_root = duffle.dirname(code_root)
|
||||
else
|
||||
args.project_root = duffle.normalize_path(args.project_root)
|
||||
end
|
||||
|
||||
local has_unity = type(args.unity_root) == "string" and args.unity_root ~= ""
|
||||
if has_unity and #args.sources > 0 then
|
||||
io.stderr:write("ps1_meta: --unity-root FILE and --source FILE are mutually exclusive\n")
|
||||
os.exit(EXIT_INTERNAL_ERROR)
|
||||
end
|
||||
if not has_unity and #args.sources == 0 then
|
||||
io.stderr:write("ps1_meta: either --unity-root FILE or at least one --source FILE is required\n")
|
||||
os.exit(EXIT_INTERNAL_ERROR)
|
||||
end
|
||||
|
||||
-- Post-link opt-ins (--gdb-runtime, --dwarf-injection) write output that depends on the linked ELF.
|
||||
-- Without --elf the metaprogram can't satisfy those requests, so refuse loud and early.
|
||||
-- This covers the explicit --post-link batch, --dwarf-injection by itself, and --gdb-runtime by itself.
|
||||
local flags = args.flags or {}
|
||||
local elf_path = flags.elf_path
|
||||
local has_elf = type(elf_path) == "string" and #elf_path > 0
|
||||
local post_links = flags.gdb_runtime or flags.dwarf_injection
|
||||
if post_links and not has_elf then
|
||||
io.stderr:write("ps1_meta: --elf PATH is required for post-link output\n")
|
||||
os.exit(EXIT_INTERNAL_ERROR)
|
||||
end
|
||||
|
||||
return args
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Build ctx from parsed args
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Build the PassCtx from parsed args. Exact mode opens only the repeated `--source` inputs;
|
||||
--- unity mode delegates direct-include resolution to duffle.resolve_source_corpus`.
|
||||
--- Scanning remains pass-owned (`src.scan`).
|
||||
--- @param args ParsedArgs
|
||||
--- @return PassCtx
|
||||
local function build_ctx(args)
|
||||
local normalized_project_root = duffle.normalize_path(args.project_root)
|
||||
local project_root = normalized_project_root
|
||||
local project_root_is_absolute = normalized_project_root:match("^%a:/")
|
||||
or normalized_project_root:sub(1, 2) == "//"
|
||||
or normalized_project_root:sub(1, 1) == "/"
|
||||
if not project_root_is_absolute then
|
||||
-- canonical_path_key validates ordinary relative paths and rejects
|
||||
-- drive-relative paths before the absolute-path rewrite is performed.
|
||||
duffle.canonical_path_key(normalized_project_root)
|
||||
project_root = duffle.normalize_path(duffle.to_absolute_path(normalized_project_root))
|
||||
else
|
||||
-- Do not route POSIX/UNC/drive-absolute paths through to_absolute_path.
|
||||
duffle.canonical_path_key(project_root)
|
||||
end
|
||||
local resolution
|
||||
if args.unity_root then
|
||||
local ok_resolve, resolved = pcall(duffle.resolve_source_corpus, {
|
||||
unity_root = args.unity_root,
|
||||
project_root = project_root,
|
||||
})
|
||||
if not ok_resolve then
|
||||
io.stderr:write("ps1_meta: cannot resolve --unity-root "
|
||||
.. tostring(args.unity_root) .. ": " .. tostring(resolved) .. "\n")
|
||||
os.exit(EXIT_INTERNAL_ERROR)
|
||||
end
|
||||
resolution = resolved
|
||||
else
|
||||
local source_order = {}
|
||||
local sources_by_path = {}
|
||||
local resolver = {
|
||||
resolved = {},
|
||||
skipped = {},
|
||||
shadowed = {},
|
||||
}
|
||||
for _, input_path in ipairs(args.sources) do
|
||||
local path = duffle.normalize_path(input_path)
|
||||
local key_ok, key_or_error = pcall(duffle.canonical_path_key, path)
|
||||
if not key_ok then
|
||||
error("ps1_meta: invalid --source " .. input_path .. ": "
|
||||
.. tostring(key_or_error), 0)
|
||||
end
|
||||
local file = io.open(path, "r")
|
||||
if not file then
|
||||
io.stderr:write("ps1_meta: cannot open --source " .. input_path .. "\n")
|
||||
os.exit(EXIT_INTERNAL_ERROR)
|
||||
end
|
||||
local text = file:read("*a")
|
||||
file:close()
|
||||
|
||||
local source = {
|
||||
path = path,
|
||||
text = text,
|
||||
dir = duffle.dirname(path),
|
||||
basename = duffle.basename_no_ext(path),
|
||||
}
|
||||
source_order[#source_order + 1] = source
|
||||
local key = key_or_error
|
||||
if not sources_by_path[key] then sources_by_path[key] = source end
|
||||
resolver.resolved[#resolver.resolved + 1] = {
|
||||
include_path = path,
|
||||
include_text = nil,
|
||||
root_source = nil,
|
||||
root_line = nil,
|
||||
candidate_a = path,
|
||||
candidate_b = nil,
|
||||
selected_path = path,
|
||||
disposition = "exact",
|
||||
}
|
||||
end
|
||||
resolution = {
|
||||
unity_root = nil,
|
||||
project_root = project_root,
|
||||
code_root = duffle.normalize_path(project_root .. "/code"),
|
||||
source_order = source_order,
|
||||
sources_by_path = sources_by_path,
|
||||
sources_by_dir = duffle.group_sources_by_dir(source_order),
|
||||
resolver = resolver,
|
||||
}
|
||||
end
|
||||
|
||||
local corpus = {
|
||||
unity_root = resolution.unity_root,
|
||||
project_root = resolution.project_root,
|
||||
code_root = resolution.code_root,
|
||||
source_order = resolution.source_order,
|
||||
sources_by_path = resolution.sources_by_path,
|
||||
sources_by_dir = resolution.sources_by_dir,
|
||||
atoms_by_name = {},
|
||||
binds_by_name = {},
|
||||
atom_infos = {},
|
||||
register_alias_registry = {},
|
||||
type_name_registry = {},
|
||||
atom_views = {},
|
||||
atom_ctxs = {},
|
||||
atom_phases = {},
|
||||
word_counts = {},
|
||||
components = {},
|
||||
component_body_index = {},
|
||||
collisions = {},
|
||||
resolver = resolution.resolver,
|
||||
}
|
||||
local ctx = {
|
||||
metadata_path = args.metadata,
|
||||
shared = { corpus = corpus },
|
||||
out_root = args.out_root,
|
||||
project_root = corpus.project_root,
|
||||
flags = args.flags or {},
|
||||
verbose = args.verbose,
|
||||
}
|
||||
|
||||
-- Source records and directory buckets are owned by the corpus.
|
||||
-- Consumers read `corpus.source_order` and `corpus.sources_by_dir` directly.
|
||||
-- The corpus is the sole source of truth for source records and module grouping; `ctx` only holds per-pass execution state.
|
||||
return ctx
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Topological sort (Kahn's algorithm + cycle detection)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Topologically sort the requested pass set, augmented with all transitive deps.
|
||||
--- Detects cycles and errors out with details.
|
||||
--- @param passes table<string, PassDescriptor>
|
||||
--- @param requested_set string[]
|
||||
--- @return string[] -- execution order
|
||||
---
|
||||
--- Dependency closure, in-degree calculation, queue seeding, and sorting are local blocks.
|
||||
--- Keeping these blocks local makes the topological sort self-contained.
|
||||
local function topo_sort(passes, requested_set)
|
||||
-- Dependency closure: include every pass transitively required by `requested_set`.
|
||||
local needed = {}
|
||||
for _, name in ipairs(requested_set) do needed[name] = true end
|
||||
local changed = true
|
||||
while changed do
|
||||
changed = false
|
||||
for name, _ in pairs(needed) do
|
||||
local pass = passes[name]
|
||||
if not pass then
|
||||
error("unknown pass '" .. name .. "' requested")
|
||||
end
|
||||
for _, dep in ipairs(pass.deps) do
|
||||
if not needed[dep] then
|
||||
needed[dep] = true
|
||||
changed = true
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
-- In-degree calculation: count each needed pass's needed dependencies.
|
||||
local in_degree = {}
|
||||
for name, _ in pairs(needed) do in_degree[name] = 0 end
|
||||
for name, _ in pairs(needed) do
|
||||
for _, dep in ipairs(passes[name].deps) do
|
||||
if needed[dep] then
|
||||
in_degree[name] = in_degree[name] + 1
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
-- Ready-queue seeding: add zero-in-degree passes in deterministic order.
|
||||
local ready = {}
|
||||
for name, deg in pairs(in_degree) do
|
||||
if deg == 0 then ready[#ready + 1] = name end
|
||||
end
|
||||
table.sort(ready)
|
||||
|
||||
-- Ready-queue drain: decrement dependents when each pass is emitted.
|
||||
-- Newly-zero-degree passes are inserted back into the ready queue (kept sorted).
|
||||
local order = {}
|
||||
while #ready > 0 do
|
||||
local just_finished = table.remove(ready, 1)
|
||||
order[#order + 1] = just_finished
|
||||
for name, _ in pairs(needed) do
|
||||
if name ~= just_finished then
|
||||
for _, dep in ipairs(passes[name].deps) do
|
||||
if dep == just_finished then
|
||||
in_degree[name] = in_degree[name] - 1
|
||||
if in_degree[name] == 0 then
|
||||
ready[#ready + 1] = name
|
||||
table.sort(ready)
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
-- Cycle detection: if `order` doesn't include all needed passes, some are stuck with in_degree > 0
|
||||
-- (the cycle closed on itself before Kahn could process them).
|
||||
-- Without this check, a fully-closed cycle (e.g. A -> B -> A) would silently return an empty order list, leaving the orchestrator to dispatch nothing.
|
||||
local needed_count = 0
|
||||
for _ in pairs(needed) do needed_count = needed_count + 1 end -- count hash entries; Lua's #t doesn't work
|
||||
if #order ~= needed_count then
|
||||
for name, deg in pairs(in_degree) do
|
||||
if deg > 0 then
|
||||
error("dependency cycle detected involving pass '" .. name .. "'")
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
return order
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Main Orchestrator
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- (internal) If the pass's kind is in PASS_KIND_STOP_ON_ERROR and it reported errors, write each error to stderr.
|
||||
--- Returns true if any validation errors were reported.
|
||||
--- @param pass_name string
|
||||
--- @param pass PassDescriptor
|
||||
--- @param result PassResult
|
||||
--- @return boolean
|
||||
local function report_validation_errors(pass_name, pass, result)
|
||||
local has_errors = result.errors and #result.errors > 0
|
||||
if not (has_errors and PASS_KIND_STOP_ON_ERROR[pass.kind]) then return false end
|
||||
for _, e in ipairs(result.errors) do
|
||||
io.stderr:write(string.format("[%s] line %d: %s\n", pass_name, e.line or 0, e.msg or ""))
|
||||
end
|
||||
return true
|
||||
end
|
||||
|
||||
--- (internal) Run each pass in `order` in topological sequence.
|
||||
--- @param ctx PassCtx
|
||||
--- @param order string[]
|
||||
--- @return boolean -- true if any validation errors were reported
|
||||
local function dispatch_passes(ctx, order)
|
||||
local had_errors = false
|
||||
for _, pass_name in ipairs(order) do
|
||||
local pass = PASSES[pass_name]
|
||||
local mod = require(pass.module)
|
||||
local result = mod.run(ctx)
|
||||
if report_validation_errors(pass_name, pass, result) then
|
||||
had_errors = true
|
||||
end
|
||||
end
|
||||
return had_errors
|
||||
end
|
||||
|
||||
--- Main entry point. Runs the requested passes in dep-topological order.
|
||||
--- @param argv string[]
|
||||
local function main(argv)
|
||||
local ok, err = pcall(function()
|
||||
local args = parse_args(argv)
|
||||
local ctx = build_ctx(args)
|
||||
|
||||
local requested = args.requested_set
|
||||
local closed = topo_sort(PASSES, requested)
|
||||
|
||||
local had_errors = dispatch_passes(ctx, closed)
|
||||
if had_errors then os.exit(EXIT_VALIDATION_ERRORS) end
|
||||
end)
|
||||
|
||||
if not ok then
|
||||
io.stderr:write("[ps1_meta] internal error: " .. tostring(err) .. "\n")
|
||||
os.exit(EXIT_INTERNAL_ERROR)
|
||||
end
|
||||
|
||||
os.exit(EXIT_OK)
|
||||
end
|
||||
|
||||
-- Module export for in-process consumers (tests that dofile this script).
|
||||
-- The conditional `main(...)` call below only fires when this file is invoked as the entry script (arg[0] ends in "ps1_meta.lua");
|
||||
-- in dofile() mode (test's arg[0] does not match), main() is skipped and the chunk returns `_M` to the caller.
|
||||
local _M = {
|
||||
PASSES = PASSES,
|
||||
PASS_KIND_STOP_ON_ERROR = PASS_KIND_STOP_ON_ERROR,
|
||||
parse_args = parse_args,
|
||||
build_ctx = build_ctx,
|
||||
}
|
||||
|
||||
if arg and arg[0] and arg[0]:match("ps1_meta%.lua$") then
|
||||
main({...})
|
||||
end
|
||||
|
||||
return _M
|
||||
+86
-7
@@ -4,25 +4,24 @@ $path_code = join-path $path_root 'code'
|
||||
$path_scripts = join-path $path_root 'scripts'
|
||||
$path_toolchain = join-path $path_root 'toolchain'
|
||||
|
||||
# Halt on any error (instead of PowerShell's default `Continue`).
|
||||
$ErrorActionPreference = 'Stop'
|
||||
|
||||
$misc = join-path $PSScriptRoot 'helpers/misc.ps1'
|
||||
. $misc
|
||||
|
||||
# TODO(Ed): Review usage of these deps
|
||||
# I orgiinally cloned them when starting to get to the C runtime usage of the course
|
||||
# However, based on the heavy reliance of the PSX.Dev extension I might fallback; also
|
||||
# The gdb server doesn't need the full repo and were only using the src/mips
|
||||
# which has a standalone repo (nuggets)
|
||||
# armips may not be used at all but I'm not sure...
|
||||
|
||||
$url_armips = 'https://github.com/Kingcom/armips.git'
|
||||
$url_pcsx_redux = 'https://github.com/grumpycoders/pcsx-redux.git'
|
||||
$url_psyq_iwyu = 'https://github.com/johnbaumann/psyq_include_what_you_use.git'
|
||||
$url_lpeg = 'https://github.com/roberto-ieru/LPeg.git'
|
||||
|
||||
$path_armips = join-path $path_toolchain 'armips'
|
||||
$path_pcsx_redux = join-path $path_toolchain 'pcsx-redux'
|
||||
$path_psyq_iwyu = join-path $path_toolchain 'psyq_iwyu'
|
||||
$path_lpeg = join-path $path_toolchain 'lpeg'
|
||||
|
||||
clone-gitrepo $path_armips $url_armips
|
||||
clone-gitrepo $path_lpeg $url_lpeg
|
||||
clone-gitrepo $path_pcsx_redux $url_pcsx_redux
|
||||
clone-gitrepo $path_psyq_iwyu $url_psyq_iwyu
|
||||
|
||||
@@ -37,3 +36,83 @@ pop-location
|
||||
# $path_pcsx_redux_binaries = join-path $path_pcsx_redux_vsprojects 'x64/Release'
|
||||
|
||||
# $psyq_obj_parser = join-path $path_pcsx_redux_binaries 'psyq-obj-parser.exe'
|
||||
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
# PCSX-Redux — built via MSBuild (VS2022)
|
||||
# Requires: Visual Studio 2022 with the C++ desktop workload.
|
||||
# Output: toolchain\pcsx-redux\vsprojects\x64\Debug\pcsx-redux.exe
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
# Locate MSBuild from the VS2022 install (no hardcoded path — uses vswhere).
|
||||
$vswhere = "${env:ProgramFiles(x86)}\Microsoft Visual Studio\Installer\vswhere.exe"
|
||||
if (-not (Test-Path $vswhere)) {
|
||||
write-error "vswhere not found at '$vswhere'. Install Visual Studio 2022 with the C++ desktop workload."
|
||||
exit 1
|
||||
}
|
||||
$msbuild_exe = & $vswhere -latest -products * -requires Microsoft.Component.MSBuild -find "MSBuild\**\Bin\MSBuild.exe" 2>$null | Select-Object -First 1
|
||||
if (-not $msbuild_exe) {
|
||||
write-error "MSBuild not found via vswhere. Install Visual Studio 2022 with the C++ desktop workload."
|
||||
exit 1
|
||||
}
|
||||
|
||||
$path_pcsx_sln = join-path $path_pcsx_redux 'vsprojects\pcsx-redux.sln'
|
||||
& $msbuild_exe $path_pcsx_sln /p:Configuration=Release /p:Platform=x64 /p:PlatformToolset=v143 /m /v:minimal
|
||||
|
||||
# Locate luajit via scoop. `luajit.exe` is on PATH via scoop's shim;
|
||||
# we use `scoop prefix` to find the install root for the include dir (needed to compile lpeg against luajit's headers).
|
||||
# If scoop or luajit is missing, fail fast with an actionable message.
|
||||
$luajit_prefix = & scoop prefix luajit 2>$null
|
||||
if (-not $luajit_prefix -or -not (Test-Path (Join-Path $luajit_prefix 'bin/luajit.exe'))) {
|
||||
write-error "luajit not found via 'scoop prefix luajit'. Install via: scoop install luajit"
|
||||
exit 1
|
||||
}
|
||||
|
||||
# Discover the luajit include dir by globbing `include/luajit-*`.
|
||||
# This avoids hardcoding a specific version (e.g. `luajit-2.1`).
|
||||
$luajit_include_root = Join-Path $luajit_prefix 'include'
|
||||
$lua_inc_dir = Get-ChildItem -Path $luajit_include_root -Directory -Filter 'luajit-*' -ErrorAction SilentlyContinue |
|
||||
Select-Object -First 1 -ExpandProperty FullName
|
||||
if (-not $lua_inc_dir) {
|
||||
write-error "No 'luajit-*' include dir found under '$luajit_include_root'. The scoop luajit install may be broken."
|
||||
exit 1
|
||||
}
|
||||
|
||||
# Generate lpeg.dll by compiling the 6 source files directly.
|
||||
# `gcc` is on PATH (scoop's shim puts it there).
|
||||
# The source files: lpcap.c lpcode.c lpcset.c lpprint.c lptree.c lpvm.c
|
||||
# Link against luajit's import library (`libluajit-5.1.a`) for the Lua C API symbols (lua_*, luaL_*).
|
||||
$luajit_lib_dir = Join-Path $luajit_prefix 'lib'
|
||||
$lpeg_sources = @('lpcap.c', 'lpcode.c', 'lpcset.c', 'lpprint.c', 'lptree.c', 'lpvm.c')
|
||||
$lpeg_compile_args = @(
|
||||
'-O2', '-shared',
|
||||
"-I$lua_inc_dir",
|
||||
"-L$luajit_lib_dir",
|
||||
'-o', 'lpeg.dll'
|
||||
) + $lpeg_sources + @('-lluajit-5.1')
|
||||
push-location $path_lpeg
|
||||
& gcc @lpeg_compile_args
|
||||
pop-location
|
||||
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
# lfs (LuaFileSystem) — compiled from pcsx-redux's vendored luafilesystem source.
|
||||
# Source: toolchain/pcsx-redux/third_party/luafilesystem/src/lfs.c
|
||||
# Output: toolchain/lfs/lfs.dll
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
$path_lfs = join-path $path_toolchain 'lfs'
|
||||
verify-path $path_lfs
|
||||
$lfs_src = join-path $path_pcsx_redux 'third_party\luafilesystem\src\lfs.c'
|
||||
$lfs_dll = join-path $path_lfs 'lfs.dll'
|
||||
$lfs_dll_import = join-path $luajit_lib_dir 'libluajit-5.1.dll.a'
|
||||
& gcc -O2 -shared "-I$lua_inc_dir" -o $lfs_dll $lfs_src $lfs_dll_import
|
||||
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
# OpenBIOS — built from the PCSX-Redux source tree via make + mipsel-none-elf
|
||||
# Output: toolchain\pcsx-redux\src\mips\openbios\openbios.bin
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
$path_openbios = join-path $path_pcsx_redux 'src\mips\openbios'
|
||||
push-location $path_openbios
|
||||
& make clean
|
||||
& make
|
||||
pop-location
|
||||
|
||||
Submodule
+1
Submodule toolchain/psyq_iwyu added at 5cbf9f68d1
Reference in New Issue
Block a user