12 Commits
Author SHA1 Message Date
ed 338f1fe46e Better reports from dsl metaprogram 2026-07-27 10:06:23 -04:00
ed 27a9038e0d req c11, 2026-07-26 17:36:46 -04:00
ed 8c8d2e54aa remove cruft 2026-07-26 14:40:57 -04:00
ed 80a35aa23a WIP: Better step debug on atom components, better db_skip annotation, lots of curation passes on lua.
Still don't have this thing in its final state for  the curse but its close.
2026-07-26 13:55:47 -04:00
ed f247d56c32 Debug vis ergonomics 2026-07-25 13:19:35 -04:00
ed 590ff1e2ec Curation pass: reduce nested conditional branching in some defnitions. 2026-07-25 13:00:36 -04:00
ed 653e18ee28 remove code related to dry run and dep graph rendering (ps1 meta) 2026-07-25 11:59:41 -04:00
ed ebb876fe89 report.lua: Remove redudnant section formatting/header 2026-07-25 11:25:12 -04:00
ed 1b40b16c0e Review pass. 2026-07-25 11:20:53 -04:00
ed 9ffd6592bc Better static analysis for C0 <-> C2 data race hazards. 2026-07-25 04:09:48 -04:00
ed d56adab38f branch delay slot better support.
Still reviewing. Need to see if gte is handled properly.
2026-07-23 18:35:02 -04:00
ed 08af73d0d2 Lua Metaprogram: Improvements to static analysis + others. 2026-07-23 10:18:30 -04:00
32 changed files with 6094 additions and 3722 deletions
+2
View File
@@ -17,3 +17,5 @@ toolchain/PSn00bSDK
.vscode/settings.json .vscode/settings.json
toolchain/lfs toolchain/lfs
toolchain/lpeg toolchain/lpeg
scratch
+10 -5
View File
@@ -11,7 +11,7 @@
* Pure macro anntation. * Pure macro anntation.
* --------------- * ---------------
* Don't want to constraint the macro usage to some attribute placment constraint, etc, don't want ot dela with the compiler. * Don't want to constraint the macro usage to some attribute placment constraint, etc, don't want ot dela with the compiler.
* atom_info, atom_bind, atom_reads, atom_writes, atom_label, atom_dbg_skip_over each expand to a C comment or to nothing * atom_info, atom_bind, atom_reads, atom_writes, atom_label, atom_dbg_skip each expand to a C comment or to nothing
* (C preprocessor strips them to whitespace). * (C preprocessor strips them to whitespace).
* *
* ============================================================================ * ============================================================================
@@ -90,13 +90,18 @@
#define atom_info(...) /* atom_info(__VA_ARGS__) */ #define atom_info(...) /* atom_info(__VA_ARGS__) */
/* ---------------------------------------------------------------------------- /* ----------------------------------------------------------------------------
* DEBUG SOURCE-STEP MARKERS * DEBUG SOURCE-STEP MARKER
* *
* Place atom_dbg_skip_over() before a MipsAtom_, MipsAtomComp_, or MipsAtomComp_Proc_. * Place `atom_dbg_skip` (BARE) before a MipsAtom_, MipsAtomComp_, or MipsAtomComp_Proc_.
* The following declaration kind determines whether the marker selects a whole atom or a component inline view. * The following declaration kind determines whether the marker selects a whole atom or a component inline view.
* The source scanner associates the marker with that declaration; placement diagnostics are handled by the annotation pass. * The source scanner associates the marker with that declaration; placement diagnostics are handled by the annotation pass.
*
* Example:
* atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(rret_addr), nop };
* atom_dbg_skip MipsAtomComp_(ac_yield) { ... };
* atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, { ... });
* ----------------------------------------------------------------------------*/ * ----------------------------------------------------------------------------*/
#define atom_dbg_skip_over() /* atom_dbg_skip_over: skip the following atom or component source view */ #define atom_dbg_skip /* atom_dbg_skip: skip the following atom or component source view */
/* ---------------------------------------------------------------------------- /* ----------------------------------------------------------------------------
* Typed-view annotations (Registry for DWARF RR_<R_X> chain resolution) * Typed-view annotations (Registry for DWARF RR_<R_X> chain resolution)
@@ -117,7 +122,7 @@
* The preferred correlation mechanism; atom_ctx is the escape hatch for non-natural cases. * The preferred correlation mechanism; atom_ctx is the escape hatch for non-natural cases.
* *
* All three expand to C comments * All three expand to C comments
* (the bare-token convention matching `atom_reg` and `atom_dbg_skip_over`). * (the bare-token convention matching `atom_reg` and `atom_dbg_skip`).
* The Lua scanner reads the bare tokens in source-as-written; the C preprocessor strips them. * The Lua scanner reads the bare tokens in source-as-written; the C preprocessor strips them.
* ----------------------------------------------------------------------------*/ * ----------------------------------------------------------------------------*/
#define atom_type(T) /* atom_type: associate <T> with the preceding enum entry (enum site) or this register (atom-info site) */ #define atom_type(T) /* atom_type: associate <T> with the preceding enum entry (enum site) or this register (atom-info site) */
+3
View File
@@ -218,3 +218,6 @@ IA_ void assert(U8 cond) { if(cond){return;} else{debug_trap(); ms_exit_process(
#endif #endif
#pragma endregion Debug #pragma endregion Debug
#endif #endif
#define GCC_OPTIMIZATION_DISABLE _Pragma("GCC push_options") _Pragma("GCC optimize(\"O0\")")
#define GCC_OPTIMIZATION_ENABLE _Pragma("GCC pop_options")
+14
View File
@@ -9,6 +9,12 @@
#define WORD_COUNT(name, count) enum { words_##name = (count) }; #define WORD_COUNT(name, count) enum { words_##name = (count) };
#endif #endif
/* atom_dbg_skip */
/* ---------------------------------------------------------------------------
* MACRO ATOM Components (Reusable Assembly Components)
* These do NOT yield. They are expanded inline inside Tape Atoms.
* ---------------------------------------------------------------------------*/
// The 'Yield' sequence for Tape Atoms (mac_yield).
#define mac_yield(...) \ #define mac_yield(...) \
load_word(R_AtomJmp, R_TapePtr, 0) \ load_word(R_AtomJmp, R_TapePtr, 0) \
, add_ui_self( R_TapePtr, S_(MipsCode)) \ , add_ui_self( R_TapePtr, S_(MipsCode)) \
@@ -16,6 +22,7 @@
, nop , nop
WORD_COUNT(mac_yield, 4) WORD_COUNT(mac_yield, 4)
/* atom_dbg_skip */
/* Words: 3; Loads 3 S2 indices from the face array */ /* Words: 3; Loads 3 S2 indices from the face array */
#define mac_load_tri_indices(...) \ #define mac_load_tri_indices(...) \
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)) \ load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)) \
@@ -23,6 +30,8 @@ WORD_COUNT(mac_yield, 4)
, load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)) , load_half_u(R_T2, R_FaceCursor, 2 * S_(S2))
WORD_COUNT(mac_load_tri_indices, 3) WORD_COUNT(mac_load_tri_indices, 3)
/* atom_dbg_skip */
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
#define mac_gte_load_tri_verts(...) \ #define mac_gte_load_tri_verts(...) \
shift_lleft(R_AT, R_T0, v3s2_byteoff) \ shift_lleft(R_AT, R_T0, v3s2_byteoff) \
, add_u_self(R_AT, R_VertBase) \ , add_u_self(R_AT, R_VertBase) \
@@ -74,16 +83,19 @@ WORD_COUNT(mac_insert_ot_tag_f3, 11)
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */ , store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
WORD_COUNT(mac_insert_ot_tag_g4, 11) WORD_COUNT(mac_insert_ot_tag_g4, 11)
/* atom_dbg_skip */
#define mac_pack_color_word(off, cmd, r, g, b) \ #define mac_pack_color_word(off, cmd, r, g, b) \
load_upper_i(R_AT, (cmd) << 8 | (b)) \ load_upper_i(R_AT, (cmd) << 8 | (b)) \
, or_i_self( R_AT, ((g) << 8) | (r)) \ , or_i_self( R_AT, ((g) << 8) | (r)) \
, store_word( R_AT, R_PrimCursor, (off)) , store_word( R_AT, R_PrimCursor, (off))
WORD_COUNT(mac_pack_color_word, 3) WORD_COUNT(mac_pack_color_word, 3)
/* atom_dbg_skip */
#define mac_format_f3_color(r, g, b) \ #define mac_format_f3_color(r, g, b) \
mac_pack_color_word(O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) mac_pack_color_word(O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b)
WORD_COUNT(mac_format_f3_color, 3) WORD_COUNT(mac_format_f3_color, 3)
/* atom_dbg_skip */
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3. /* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */ * PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
#define mac_gte_store_f3_post_rtpt(...) \ #define mac_gte_store_f3_post_rtpt(...) \
@@ -99,6 +111,7 @@ WORD_COUNT(mac_gte_store_f3_post_rtpt, 3)
, mac_pack_color_word(O_(Poly_G4,c3), 0, r3,g3,b3) , mac_pack_color_word(O_(Poly_G4,c3), 0, r3,g3,b3)
WORD_COUNT(mac_format_g4_color, 12) WORD_COUNT(mac_format_g4_color, 12)
/* atom_dbg_skip */
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the /* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
* G4 triangle portion to p0/p1/p2. * G4 triangle portion to p0/p1/p2.
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). * PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
@@ -113,6 +126,7 @@ WORD_COUNT(mac_format_g4_color, 12)
, gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2)) , gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2))
WORD_COUNT(mac_gte_store_g4_p012_post_rtpt_pre_rtps, 3) WORD_COUNT(mac_gte_store_g4_p012_post_rtpt_pre_rtps, 3)
/* atom_dbg_skip */
/* Words: 1; Stores the V3 screen coord to the G4's p3 slot. /* Words: 1; Stores the V3 screen coord to the G4's p3 slot.
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its * PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its
* single-vertex result to SXY2; SXY0 still holds v0.screen from the * single-vertex result to SXY2; SXY0 still holds v0.screen from the
+2 -2
View File
@@ -2,7 +2,7 @@
* duffle DSL — GPU Vendor Mnemonics (opt-in) * duffle DSL — GPU Vendor Mnemonics (opt-in)
* ============================================================================ * ============================================================================
* *
* Provides the PSYQ-style CamelCase aliases for the canonical duffle GPU primitive setters and OT operations. * Provides the PSYQ-style CamelCase aliases for the duffle GPU primitive setters and OT operations.
* The duffle snake_case names are primary; this header is for users who prefer the PSYQ SDK function names from the legacy C API. * The duffle snake_case names are primary; this header is for users who prefer the PSYQ SDK function names from the legacy C API.
* *
* USAGE: #include "duffle/gp_vendor_sym.h" // after gp.h * USAGE: #include "duffle/gp_vendor_sym.h" // after gp.h
@@ -24,7 +24,7 @@
* The gp0_cmd_* / gp1_cmd_* byte constants are already short and descriptive; no vendor alias is provided for them. * The gp0_cmd_* / gp1_cmd_* byte constants are already short and descriptive; no vendor alias is provided for them.
* *
* The vendor mnemonics are NOT registered with the duffle word-count metadata (word_counts.metadata.h). * The vendor mnemonics are NOT registered with the duffle word-count metadata (word_counts.metadata.h).
* They expand to the duffle canonical macros which DO have word-count entries * They expand to the duffle macros which DO have word-count entries
* (the ones emitted by mac_format_f3_color / mac_gte_store_f3 / etc.). Verification: V13 (objdump byte-identical) holds. * (the ones emitted by mac_format_f3_color / mac_gte_store_f3 / etc.). Verification: V13 (objdump byte-identical) holds.
* ============================================================================ */ * ============================================================================ */
+30 -43
View File
@@ -353,41 +353,34 @@ enum { _C2_TX_SUBS_ = 0
/* GTE command words for the common cases. /* GTE command words for the common cases.
* *
* These are pure compile-time integer constants — the C compiler * These are pure compile-time integer constants — the C compiler constant-folds them into `.word` directives in .rodata.
* constant-folds them into `.word` directives in .rodata. Use them * Use them inside `asm_inline(...)` blocks (see `gte_rtpt` below for the idiom).
* inside `asm_inline(...)` blocks (see `gte_rtpt` below for the
* canonical idiom).
* *
* Decomposition (per the `enc_gte_<field>` definitions above): * Decomposition (per the `enc_gte_<field>` definitions above):
* gte_cmdw_<name> = gte_cmd_base | enc_gte_cmd(<cmd>) * gte_cmdw_<name> = gte_cmd_base | enc_gte_cmd(<cmd>)
* The SF/MX/V/CV/LM fields are all zero in the common cases (standard * The SF/MX/V/CV/LM fields are all zero in the common cases
* rotation-matrix, no scaling factor, V0 vector, translation vector, * (standard rotation-matrix, no scaling factor, V0 vector, translation vector, no clamp),
* no clamp), so the only varying bits are the `cmd` field. * so the only varying bits are the `cmd` field.
* *
* Naming follows the file's convention: `gte_cmd_*` is the raw * Naming follows the file's convention: `gte_cmd_*` is the raw 6-bit `cmd` field id, `gte_cmdw_*`
* 6-bit `cmd` field id, `gte_cmdw_*` is the fully-encoded 32-bit * is the fully-encoded 32-bit instruction word ready to drop into a `.word` directive.
* instruction word ready to drop into a `.word` directive.
* *
* -------------------------------------------------------------------------- * --------------------------------------------------------------------------
* PsyQ-compatibility note (RTPS/RTPT): * PsyQ-compatibility note (RTPS/RTPT):
* The original Sony PsyQ `inline_n.h` ships RTPT as `cop2 0x0280030` and * The original Sony PsyQ `inline_n.h` ships RTPT as `cop2 0x0280030` and RTPS as `cop2 0x0180001`.
* RTPS as `cop2 0x0180001`. Both have `0x20` set in the upper-reserved * Both have `0x20` set in the upper-reserved region (bit 21) AND `sf=1` (bit 19) — i.e. the "no division" flag.
* region (bit 21) AND `sf=1` (bit 19) — i.e. the "no division" flag. * Per psx-spec these bits are reserved/must-be-zero,
* Per psx-spec these bits are reserved/must-be-zero, but the real GTE * but the real GTE hardware and PCSX-Redux's GTE model both IGNORE them on these two commands
* hardware and PCSX-Redux's GTE model both IGNORE them on these two * (the perspective divide happens regardless of `sf`).
* commands (the perspective divide happens regardless of `sf`).
* *
* If we emit a strictly-spec-compliant word (`sf=0`, reserved bits * If we emit a strictly-spec-compliant word (`sf=0`, reserved bits clear),
* clear), PCSX-Redux's GTE checks those bits more strictly than the * PCSX-Redux's GTE checks those bits more strictly than the silicon does and RTPT silently no-ops —
* silicon does and RTPT silently no-ops — the floor's screen * the floor's screen coordinates come out as raw projection-of-rotation (Z never divided),
* coordinates come out as raw projection-of-rotation (Z never * `nclip` ends up wrong, and the triangle is culled.
* divided), `nclip` ends up wrong, and the triangle is culled.
* *
* So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to * So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern everyone has shipped for 25 years.
* match the working bit pattern everyone has shipped for 25 years. * NCLIP/OP/MVMVA stay spec-clean — their reserved bits really are zero in the original PsyQ source.
* NCLIP/OP/MVMVA stay spec-clean — their reserved bits really are
* zero in the original PsyQ source.
* -------------------------------------------------------------------------- * --------------------------------------------------------------------------
*/ */
#define gte_cmdw_psyq_compat (1u << 21 | enc_gte_sf(gte_sf_integer)) #define gte_cmdw_psyq_compat (1u << 21 | enc_gte_sf(gte_sf_integer))
@@ -425,8 +418,8 @@ enum { _C2_TX_SUBS_ = 0
* (XY at offset 0) and C2_VZ0 (Z at offset 4) using `lwc2`. * (XY at offset 0) and C2_VZ0 (Z at offset 4) using `lwc2`.
* *
* Uses string-style GCC inline asm with `%0` substitution because the * Uses string-style GCC inline asm with `%0` substitution because the
* base register `r0` is a runtime GPR chosen by the compiler — it cannot * base register `r0` is a runtime GPR chosen by the compiler.
* be encoded into a static `.word` constant. * It cannot be encoded into a static `.word` constant.
* *
* Usage: * Usage:
* asm_gte_load_v0(svector_ptr); * asm_gte_load_v0(svector_ptr);
@@ -458,26 +451,21 @@ enum {
/* gte_load_vN(r_ptr, base) — placeholder-punned lwc2 loaders /* gte_load_vN(r_ptr, base) — placeholder-punned lwc2 loaders
* *
* Emits `.word` constants encoding `lwc2 $N, off(<base>)` for the chosen * Emits `.word` constants encoding `lwc2 $N, off(<base>)` for the chosen GTE vector register, where `<base>` is the GPR number you pass in
* GTE vector register, where `<base>` is the GPR number you pass in
* (typically one of R_T4..R_T9 for the standard "3-pointer" pattern). * (typically one of R_T4..R_T9 for the standard "3-pointer" pattern).
* *
* The caller MUST bind `r_ptr` to that same GPR via a register variable: * The caller MUST bind `r_ptr` to that same GPR via a register variable:
* register V3_S2* p_in_12 __asm__("$12") = my_ptr; * register V3_S2* p_in_12 __asm__("$12") = my_ptr;
* gte_load_v0(p_in_12, R_T4); // R_T4 = 12, base is $12 * gte_load_v0(p_in_12, R_T4); // R_T4 = 12, base is $12
* *
* Then `"r"(r_ptr)` inside the asm binds to $12 (the only register * Then `"r"(r_ptr)` inside the asm binds to $12 (the only register `p_in_12` can live in),
* `p_in_12` can live in), which is exactly the register the .word * which is exactly the register the .word constants expect. A `"$12"` clobber would conflict with the register-variable binding
* constants expect. A `"$12"` clobber would conflict with the * ("asm specifier for variable conflicts with asm clobber list"), so we omit it.
* register-variable binding ("asm specifier for variable conflicts * The other ABI-clobbers ($2/$8/$9/$31) stay because the GTE instructions don't touch caller-saved GPRs but the kernel does treat them as volatile.
* with asm clobber list"), so we omit it. The other ABI-clobbers
* ($2/$8/$9/$31) stay because the GTE instructions don't touch
* caller-saved GPRs but the kernel does treat them as volatile.
* *
* WHICH REGISTER TO PICK * WHICH REGISTER TO PICK
* ---------------------- * ----------------------
* Any caller-saved GPR is safe. Recommended default for an RTPT-style * Any caller-saved GPR is safe. Recommended default for an RTPT-style 3-pointer pipeline:
* 3-pointer pipeline:
* gte_load_v0(p0, R_T4); // $12 * gte_load_v0(p0, R_T4); // $12
* gte_load_v1(p1, R_T5); // $13 * gte_load_v1(p1, R_T5); // $13
* gte_load_v2(p2, R_T6); // $14 * gte_load_v2(p2, R_T6); // $14
@@ -490,8 +478,7 @@ enum {
* clobbers section : "$2", "$8", ..., "memory" (from asm_clobber) * clobbers section : "$2", "$8", ..., "memory" (from asm_clobber)
* 3 colons total, GCC-legal. No string-syntax mnemonics in the .word body. * 3 colons total, GCC-legal. No string-syntax mnemonics in the .word body.
* *
* The `asm_clobber(...)` helper from gcc_asm.h prepends the colon that * The `asm_clobber(...)` helper from gcc_asm.h prepends the colon that starts the clobbers section. */
* starts the clobbers section. */
#define gte_load_v0(r_ptr, base) asm volatile( \ #define gte_load_v0(r_ptr, base) asm volatile( \
asm_words( gte_lw_v0_xy(base), gte_lw_v0_z(base) ) \ asm_words( gte_lw_v0_xy(base), gte_lw_v0_z(base) ) \
asm_rpins, r_use(r_ptr) \ asm_rpins, r_use(r_ptr) \
@@ -510,11 +497,11 @@ enum {
asm_clobber: rlit(R_V0), rlit(R_T0), rlit(R_T1), rlit(R_RA), clb_mem_drain \ asm_clobber: rlit(R_V0), rlit(R_T0), rlit(R_T1), rlit(R_RA), clb_mem_drain \
) )
/* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — the canonical prelude to gte_cmd_rtpt. /* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — prelude to gte_cmd_rtpt.
* *
* Loads all three GTE input vectors (6 words) from three separate pointers, * Loads all three GTE input vectors (6 words) from three separate pointers,
* one per GTE vector register, each loaded from its own base GPR. Caller * one per GTE vector register, each loaded from its own base GPR.
* must bind each `pN` to `bN` via a register variable. * Caller must bind each `pN` to `bN` via a register variable.
* *
* register V3_S2* p0 rgcc(R_T4) = verts[0].ptr; // → __asm__("$12") * register V3_S2* p0 rgcc(R_T4) = verts[0].ptr; // → __asm__("$12")
* register V3_S2* p1 rgcc(R_T5) = verts[1].ptr; // → __asm__("$13") * register V3_S2* p1 rgcc(R_T5) = verts[1].ptr; // → __asm__("$13")
+1 -1
View File
@@ -2,7 +2,7 @@
* duffle DSL — GTE Vendor Mnemonics (opt-in) * duffle DSL — GTE Vendor Mnemonics (opt-in)
* ============================================================================ * ============================================================================
* *
* Provides the textbook MIPS assembly mnemonics for the GTE/COP2 instructions as thin aliases to the canonical duffle macros in gte.h. * Provides the textbook MIPS assembly mnemonics for the GTE/COP2 instructions as thin aliases to the duffle macros in gte.h.
* The duffle names are primary; this header is for users who prefer the textbook mnemonics. * The duffle names are primary; this header is for users who prefer the textbook mnemonics.
* *
* USAGE: #include "duffle/gte_vendor_sym.h" // after gte.h * USAGE: #include "duffle/gte_vendor_sym.h" // after gte.h
+12 -15
View File
@@ -57,10 +57,10 @@ enum {
* ---------------------------------------------------------------------------*/ * ---------------------------------------------------------------------------*/
/* The 'Exit' Atom */ /* The 'Exit' Atom */
MipsAtom_(tape_exit) { jump_reg(rret_addr), nop }; atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(rret_addr), nop };
/* Generalized Tape Engine Runner */ /* Generalized Tape Engine Runner */
FI_ void tape_run(Slice_U4 tape) { register U4* tp rgcc(R_TapePtr) = tape.ptr; asm volatile( NI_ void tape_run(Slice_MipsCode tape) { register U4* tp rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile(
asm_words( asm_words(
add_ui( R_SP, R_SP, -MipsStackAlignment) /* Allocate stack space */ add_ui( R_SP, R_SP, -MipsStackAlignment) /* Allocate stack space */
, store_word( R_RA, R_SP, 0) /* Safely backup $ra to the stack */ , store_word( R_RA, R_SP, 0) /* Safely backup $ra to the stack */
@@ -91,8 +91,8 @@ FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ mem.ptr, mem.len, 0 }
FI_ void tb_emit(TapeBuilder* tb, MipsCode* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; } FI_ void tb_emit(TapeBuilder* tb, MipsCode* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; } FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; }
FI_ Slice_U4 tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Slice_U4){ C_(U4*,tb->ptr), tb->used }; } FI_ Slice_MipsCode tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Slice_MipsCode){ C_(U4*,tb->ptr), tb->used }; }
FI_ Slice_U4 tb_slice(TapeBuilder tb) { return (Slice_U4){ C_(U4*,tb.ptr), tb.used }; } FI_ Slice_MipsCode tb_slice(TapeBuilder tb) { return (Slice_MipsCode){ C_(U4*,tb.ptr), tb.used }; }
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit)) #define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit))
#pragma endregion Tape Drive #pragma endregion Tape Drive
@@ -104,22 +104,21 @@ FI_ Slice_U4 tb_slice(TapeBuilder tb) { return (Sli
* ---------------------------------------------------------------------------*/ * ---------------------------------------------------------------------------*/
// The 'Yield' sequence for Tape Atoms (mac_yield). // The 'Yield' sequence for Tape Atoms (mac_yield).
MipsAtomComp_(ac_yield) { atom_dbg_skip MipsAtomComp_(ac_yield) {
load_word(R_AtomJmp, R_TapePtr, 0), load_word(R_AtomJmp, R_TapePtr, 0),
add_ui_self( R_TapePtr, S_(MipsCode)), add_ui_self( R_TapePtr, S_(MipsCode)),
jump_reg( R_AtomJmp), nop, jump_reg( R_AtomJmp), nop,
}; };
/* Words: 3; Loads 3 S2 indices from the face array */ /* Words: 3; Loads 3 S2 indices from the face array */
MipsAtomComp_(ac_load_tri_indices) { atom_dbg_skip MipsAtomComp_(ac_load_tri_indices) {
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)), load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)), load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)), load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
}; };
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */ /* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
atom_dbg_skip_over() atom_dbg_skip MipsAtomComp_(ac_gte_load_tri_verts) {
MipsAtomComp_(ac_gte_load_tri_verts) {
shift_lleft(R_AT, R_T0, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0), shift_lleft(R_AT, R_T0, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
shift_lleft(R_AT, R_T1, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1), shift_lleft(R_AT, R_T1, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
shift_lleft(R_AT, R_T2, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2), shift_lleft(R_AT, R_T2, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
@@ -158,7 +157,7 @@ MipsAtomComp_(ac_insert_ot_tag_g4) {
/* Words: 3; Emits one (cmd|color) word to R_PrimCursor at the given /* Words: 3; Emits one (cmd|color) word to R_PrimCursor at the given
* byte offset. Internal helper used by the *_format_*_color macros. */ * byte offset. Internal helper used by the *_format_*_color macros. */
FI_ MipsAtom ac_pack_color_word(U4 off, U4 cmd, U1 r, U1 g, U1 b) FI_ MipsAtom ac_pack_color_word(U4 off, U4 cmd, U1 r, U1 g, U1 b)
MipsAtomComp_Proc_(ac_pack_color_word, { atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, {
load_upper_i(R_AT, (cmd) << 8 | (b)), load_upper_i(R_AT, (cmd) << 8 | (b)),
or_i_self( R_AT, ((g) << 8) | (r)), or_i_self( R_AT, ((g) << 8) | (r)),
store_word( R_AT, R_PrimCursor, (off)), store_word( R_AT, R_PrimCursor, (off)),
@@ -167,11 +166,11 @@ MipsAtomComp_Proc_(ac_pack_color_word, {
/* Words: 3; Emits the F3 command+color word (cmd byte | BLUE | GREEN | RED) /* Words: 3; Emits the F3 command+color word (cmd byte | BLUE | GREEN | RED)
* Args: _r, _g, _b are 8-bit RGB byte values (not raw 16-bit fields). */ * Args: _r, _g, _b are 8-bit RGB byte values (not raw 16-bit fields). */
FI_ MipsAtom ac_format_f3_color(U1 r, U1 g, U1 b) FI_ MipsAtom ac_format_f3_color(U1 r, U1 g, U1 b)
MipsAtomComp_Proc_(ac_format_f3_color, { mac_pack_color_word(O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) }) atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, { mac_pack_color_word(O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3. /* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */ * PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
MipsAtomComp_(ac_gte_store_f3_post_rtpt) { atom_dbg_skip MipsAtomComp_(ac_gte_store_f3_post_rtpt) {
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_F3,p0)), gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_F3,p0)),
gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_F3,p1)), gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_F3,p1)),
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_F3,p2)), gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_F3,p2)),
@@ -199,7 +198,7 @@ MipsAtomComp_Proc_(ac_format_g4_color, {
* three registers aligned with v0/v1/v2 you must store before RTPS). * three registers aligned with v0/v1/v2 you must store before RTPS).
* The macro name declares the pipeline position; check #6 (GTE state- * The macro name declares the pipeline position; check #6 (GTE state-
* machine validation) verifies the call site matches the declaration. */ * machine validation) verifies the call site matches the declaration. */
MipsAtomComp_(ac_gte_store_g4_p012_post_rtpt_pre_rtps) { atom_dbg_skip MipsAtomComp_(ac_gte_store_g4_p012_post_rtpt_pre_rtps) {
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_G4,p0)), gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_G4,p0)),
gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_G4,p1)), gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_G4,p1)),
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2)), gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2)),
@@ -211,7 +210,7 @@ MipsAtomComp_(ac_gte_store_g4_p012_post_rtpt_pre_rtps) {
* earlier RTPT — DO NOT read SXY0 here, that's the bug this name * earlier RTPT — DO NOT read SXY0 here, that's the bug this name
* prevents). * prevents).
*/ */
MipsAtomComp_(ac_gte_store_g4_p3_post_rtps) { gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p3)) }; atom_dbg_skip MipsAtomComp_(ac_gte_store_g4_p3_post_rtps) { gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p3)) };
#pragma endregion Macro Atom Components #pragma endregion Macro Atom Components
@@ -295,7 +294,6 @@ internal MipsAtom_(set_gte_world) atom_info(
/* DIAGNOSTIC 1: Pure tape loop test */ /* DIAGNOSTIC 1: Pure tape loop test */
internal MipsAtom_(diag_yield) { mac_yield() }; internal MipsAtom_(diag_yield) { mac_yield() };
// TODO(Ed): Reduce magic numbers/offsets
/* DIAGNOSTIC 2: Pure memory test (No GTE). Draws a fixed cyan triangle. */ /* DIAGNOSTIC 2: Pure memory test (No GTE). Draws a fixed cyan triangle. */
internal MipsAtom_(diag_color) { internal MipsAtom_(diag_color) {
store_word( R_0, R_T7, 0), store_word( R_0, R_T7, 0),
@@ -324,7 +322,6 @@ internal MipsAtom_(diag_color) {
mac_yield() mac_yield()
}; };
// TODO(Ed): Reduce magic numbers/offsets
/* DIAGNOSTIC 3: Pure GTE test (No Memory Writes) */ /* DIAGNOSTIC 3: Pure GTE test (No Memory Writes) */
internal MipsAtom_(diag_gte) { internal MipsAtom_(diag_gte) {
/* Load 3 indices */ /* Load 3 indices */
+1 -1
View File
@@ -103,7 +103,7 @@ FI_ void farena_init(FArena_R arena, Slice mem) { assert(arena != nullptr);
arena->used = 0; arena->used = 0;
} }
FI_ FArena farena_make(Slice mem) { FArena a; farena_init(& a, mem); return a; } FI_ FArena farena_make(Slice mem) { FArena a; farena_init(& a, mem); return a; }
I_ Slice farena_push(FArena_R arena, U4 amount, Opt_farena o) { I_ Slice farena_push(FArena_R arena, U4 amount, Opt_farena o) {
if (amount == 0) { return (Slice){}; } if (amount == 0) { return (Slice){}; }
U4 desired = amount * (o.type_width == 0 ? 1 : o.type_width); U4 desired = amount * (o.type_width == 0 ? 1 : o.type_width);
U4 to_commit = align_pow2(desired, o.alignment ? o.alignment : MEM_ALIGNMENT_DEFAULT); U4 to_commit = align_pow2(desired, o.alignment ? o.alignment : MEM_ALIGNMENT_DEFAULT);
+1 -1
View File
@@ -436,7 +436,7 @@ enum { _BitOffsets = 0
/* --- Shift-amount alias (matches the gas convention `\p3 = shamt`) --- */ /* --- Shift-amount alias (matches the gas convention `\p3 = shamt`) --- */
#define shift_amount(rd, rt, n) shift_lleft(rd, rt, n) #define shift_amount(rd, rt, n) shift_lleft(rd, rt, n)
/* nop — canonical sll $0, $0, 0 */ /* nop — sll $0, $0, 0 */
#define nop shift_lleft(rdiscard, rdiscard, 0) #define nop shift_lleft(rdiscard, rdiscard, 0)
#define nop2 nop, nop #define nop2 nop, nop
+1 -1
View File
@@ -2,7 +2,7 @@
* duffle DSL — MIPS Vendor Mnemonics (opt-in) * duffle DSL — MIPS Vendor Mnemonics (opt-in)
* ============================================================================ * ============================================================================
* *
* Provides the textbook MIPS assembly mnemonics as thin aliases to the canonical duffle macros in mips.h. * Provides the textbook MIPS assembly mnemonics as thin aliases to the duffle macros in mips.h.
* The duffle names are primary; this header is for users who prefer the textbook mnemonics. * The duffle names are primary; this header is for users who prefer the textbook mnemonics.
* *
* USAGE: #include "duffle/mips_vendor_sym.h" // after mips.h * USAGE: #include "duffle/mips_vendor_sym.h" // after mips.h
+2 -2
View File
@@ -15,9 +15,9 @@ enum {
atom_offset_bounds_chk_cube_g4_face_exit = _atom_offset_bounds_chk_cube_g4_face_exit, atom_offset_bounds_chk_cube_g4_face_exit = _atom_offset_bounds_chk_cube_g4_face_exit,
}; };
// --- atom: floor_f3_face (66 words) --- // --- atom: floor_f3_face (58 words) ---
#define _atom_offset_culling_floor_f3_face_exit 29 #define _atom_offset_culling_floor_f3_face_exit 25
#define _atom_offset_bounds_chk_floor_f3_face_exit 13 #define _atom_offset_bounds_chk_floor_f3_face_exit 13
enum { enum {
+16 -17
View File
@@ -1,6 +1,6 @@
#include "stdio.h" #include <stdio.h>
#include <stdlib.h> #include <stdlib.h>
#include "assert.h" #include <assert.h>
// #include "libgpu.h" // #include "libgpu.h"
// #include "libetc.h" // #include "libetc.h"
// #include "libgte.h" // #include "libgte.h"
@@ -99,8 +99,13 @@ typedef Struct_(Ent_Floor) {
A2_V3_S2 faces; A2_V3_S2 faces;
}; };
enum { scratchpad_size = 1024, }; enum {
Scratchpad_Len = 1024,
MemTape_Len = 512,
};
typedef Struct_(SMemory) { typedef Struct_(SMemory) {
U4 MemTape[MemTape_Len];
DoubleBuffer screen_buf; DoubleBuffer screen_buf;
A2_OrderingTable_Buffer ordering_tbl; A2_OrderingTable_Buffer ordering_tbl;
PrimitiveArena primitives; PrimitiveArena primitives;
@@ -182,6 +187,7 @@ void gp_display_frame(DoubleBuffer* screen_buf, S4* active_buf_id, U4* ordering_
void render(void) { void render(void) {
} }
GCC_OPTIMIZATION_DISABLE
void update(PrimitiveArena* pa, U4* ordering_buf) void update(PrimitiveArena* pa, U4* ordering_buf)
{ {
orderingtbl_clear_reverse(ordering_buf, OrderingTbl_Len); orderingtbl_clear_reverse(ordering_buf, OrderingTbl_Len);
@@ -207,6 +213,8 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
A2_S2 p; //??? A2_S2 p; //???
S4 flag; //???? S4 flag; //????
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
// Draw Cube // Draw Cube
if (0) if (0)
{ {
@@ -259,8 +267,7 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
U4 prim_base = u4_(pa->buf[smem.active_buf_id]); U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
U4 prim_cursor = prim_base + pa->used; U4 prim_cursor = prim_base + pa->used;
LP_ U4 mem_temp_tape[512]; tb.used = 0; tb_scope(& tb) {
TapeBuilder tb = tb_make(slice_ut_arr(mem_temp_tape)); tb_scope(& tb) {
tb_emit(& tb, rbind_cube_g4_face); tb_emit(& tb, rbind_cube_g4_face);
tb_data(& tb, prim_cursor); tb_data(& tb, prim_cursor);
tb_data(& tb, u4_(smem.cube.faces)); tb_data(& tb, u4_(smem.cube.faces));
@@ -344,12 +351,11 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
U4 prim_base = u4_(pa->buf[smem.active_buf_id]); U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
U4 prim_cursor = prim_base + pa->used; U4 prim_cursor = prim_base + pa->used;
// TODO(Ed): We should do a bounds check beforehand to confirm pa can hold all tris. // TODO(Ed): We should do a bounds check beforehand to confirm pa can hold all tris?
// The tape atoms in-flight should not need to care. // The tape atoms in-flight should not need to care.
// Prepare the tape. (Push protocol to tape) // Prepare the tape. (Push protocol to tape)
LP_ U4 mem_temp_tape[512]; tb.used = 0; tb_scope(& tb) {
TapeBuilder tb = tb_make(slice_ut_arr(mem_temp_tape)); tb_scope(& tb) {
tb_emit(& tb, set_gte_world); tb_emit(& tb, set_gte_world);
tb_data(& tb, u4_(& smem.tform_world)); tb_data(& tb, u4_(& smem.tform_world));
@@ -367,25 +373,18 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
tb_data(& tb, u4_(& pa->used)); tb_data(& tb, u4_(& pa->used));
tb_data(& tb, prim_base); tb_data(& tb, prim_base);
} }
tape_run(tb_slice(tb));// Fire off the tape. tape_run(tb_slice(tb));// Fire off the tape.
// C-side state (pa->used) has already been updated by the tape! // C-side state (pa->used) has already been updated by the tape!
smem.floor.rot.y += 5; smem.floor.rot.y += 5;
} }
// --- TAPE DIAGNOSTICS --- // --- TAPE DIAGNOSTICS ---
if (1) if (0)
{ {
LP_ U4 mem_temp_tape[512]; FArena tape_arena; farena_init(& tape_arena, slice_ut_arr(mem_temp_tape)); LP_ U4 mem_temp_tape[512]; FArena tape_arena; farena_init(& tape_arena, slice_ut_arr(mem_temp_tape));
TapeBuilder tb = tb_make_old(& tape_arena); tb_scope(& tb) { TapeBuilder tb = tb_make_old(& tape_arena); tb_scope(& tb) {
// Skip set_gte_world atom for diagnostics to isolate the triangle loop // Skip set_gte_world atom for diagnostics to isolate the triangle loop
for (U4 i = 0; i < Floor_num_faces; i++) { for (U4 i = 0; i < Floor_num_faces; i++) {
// =======================================================
// SWAP EMIT TO TEST DIFFERENT PARTS OF THE PIPELINE:
// =======================================================
// 1. code_diag_yield -> Tests Tape Engine jump logic
// 2. code_diag_color -> Tests OT and Prim Arena memory
// 3. code_diag_gte -> Tests Vertex arrays and GTE Math
// tb_emit(& tb, code_diag_yield); // tb_emit(& tb, code_diag_yield);
// tb_emit(& tb, code_diag_color); // tb_emit(& tb, code_diag_color);
// tb_emit(& tb, code_diag_gte); // tb_emit(& tb, code_diag_gte);
@@ -394,9 +393,9 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
B1* prim_cursor = (B1*)r_(pa->buf)[smem.active_buf_id] + pa->used; B1* prim_cursor = (B1*)r_(pa->buf)[smem.active_buf_id] + pa->used;
tape_run(tb_slice(tb)); tape_run(tb_slice(tb));
pa->used = (U4)prim_cursor - (U4)r_(pa->buf)[smem.active_buf_id]; pa->used = (U4)prim_cursor - (U4)r_(pa->buf)[smem.active_buf_id];
smem.floor.rot.y += 5;
} }
} }
GCC_OPTIMIZATION_ENABLE
int main(void) int main(void)
{ {
+8 -9
View File
@@ -103,28 +103,27 @@ MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(f
mac_yield() mac_yield()
}; };
atom_dbg_skip
internal internal
atom_dbg_skip_over()
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3) MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase) , atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
, atom_writes(R_PrimCursor, R_FaceCursor) , atom_writes(R_PrimCursor, R_FaceCursr)
) { ) {
mac_load_tri_indices( R_T0, R_T1, R_T2), mac_load_tri_indices( R_T0, R_T1, R_T2),
mac_gte_load_tri_verts(R_T0, R_T1, R_T2), mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
nop2, gte_cmdw_rotate_translate_perspective_triple, nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
nop2, gte_cmdw_nclip, gte_cmdw_nclip,
/* Culling (Branch forward if Backface) */ /* Culling (Branch forward if Backface) */
nop2, gte_mv_from_data_r(R_T0, C2_MAC0), gte_mv_from_data_r(R_T0, C2_MAC0),
nop, nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop,
/* Format Primitive */ /* Format Primitive */
mac_format_f3_color(0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white) mac_format_f3_color(0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
mac_gte_store_f3_post_rtpt(), mac_gte_store_f3_post_rtpt(),
/* Calculate Depth */ /* Calculate Depth */
nop2, gte_avg_sort_z3, gte_avg_sort_z3,
nop2, gte_mv_from_data_r(R_T1, C2_OTZ), gte_mv_from_data_r(R_T1, C2_OTZ),
/* Bounds Check OTZ < 2048 (Branch forward to skip insertion) */ /* Bounds Check OTZ < 2048 (Branch forward to skip insertion) */
add_ui( R_AT, R_0, OrderingTbl_Len), add_ui( R_AT, R_0, OrderingTbl_Len),
set_lt_u( R_AT, R_T1, R_AT), set_lt_u( R_AT, R_T1, R_AT),
+63 -59
View File
@@ -81,19 +81,6 @@ $path_psyq = join-path $path_toolchain 'psyq-4_7'
$path_psyq_iwyu = join-path $path_toolchain 'psyq_iwyu' $path_psyq_iwyu = join-path $path_toolchain 'psyq_iwyu'
$path_psyq_imyu_inc = join-path $path_psyq_iwyu 'include' $path_psyq_imyu_inc = join-path $path_psyq_iwyu 'include'
function Get-SourceFiles { param([Parameter(Mandatory=$true)] [string[]]$paths, [Parameter(Mandatory=$true)] [string[]]$extensions)
$files = @()
foreach ($p in $paths) {
if (-not (test-path $p)) { continue }
foreach ($ext in $extensions) {
Get-ChildItem -Path $p -File -Recurse -Filter "*$ext" -ErrorAction SilentlyContinue | ForEach-Object {
$files += $_.FullName
}
}
}
return ($files | Sort-Object -Unique)
}
function assemble-unit { param( function assemble-unit { param(
[string] $unit, [string] $unit,
[string] $link_module, [string] $link_module,
@@ -153,7 +140,7 @@ function compile-unit { param(
$f_arch_no_shared, $f_arch_no_shared,
$f_arch_no_stack_prot $f_arch_no_stack_prot
) )
# $compile_args += $f_std_c23 $compile_args += $f_std_c11
$compile_args += ($f_include + $path_psyq_imyu_inc) $compile_args += ($f_include + $path_psyq_imyu_inc)
$compile_args += ($f_include + $path_nugget) $compile_args += ($f_include + $path_nugget)
@@ -243,6 +230,52 @@ function make-binary { param([string]$elf, [string]$exe)
if ($LASTEXITCODE -ne 0) { Write-Error "Objcopy failed. Aborting."; exit 1 } if ($LASTEXITCODE -ne 0) { Write-Error "Objcopy failed. Aborting."; exit 1 }
} }
function ps1-meta { param(
[string]$unity_root,
[string[]]$sources,
[Parameter(Mandatory=$true)][string]$metadata,
[string]$out_root = (join-path $path_build 'gen'),
[string[]]$passes = @('--pre-link'),
[string[]]$extra_args = @()
)
# `--unity-root` and `--source` are
# mutually exclusive. Exactly one of `$unity_root` / `$sources` must
# be supplied; the other must be absent.
if ($null -ne $unity_root -and $unity_root -ne '')
{
if ($null -ne $sources -and $sources.Count -gt 0) {
write-error 'ps1-meta: -unity_root and -sources are mutually exclusive'
exit 2
}
}
elseif ($null -eq $sources -or $sources.Count -eq 0) {
write-error 'ps1-meta: either -unity_root <file> or -sources <file...> is required'
exit 2
}
$script = join-path $path_scripts 'ps1_meta.lua'
$input_summary = if ($null -ne $unity_root -and $unity_root -ne '') {
"unity=$unity_root"
}
else {
"$($sources.Count) source(s)"
}
write-host "ps1-meta $input_summary, passes=$($passes -join ',')" ` -ForegroundColor Magenta
$arg_list = @($passes) + @('--metadata', $metadata) + @('--out-root', $out_root) + @($extra_args)
if ($null -ne $unity_root -and $unity_root -ne '') {
$arg_list += @('--unity-root', $unity_root)
}
else {
foreach ($s in $sources) { $arg_list += @('--source', $s) }
}
& luajit $script @arg_list
if ($LASTEXITCODE -ne 0) {
write-error "ps1-meta failed (exit $LASTEXITCODE). Aborting."
exit $LASTEXITCODE
}
}
function build-hello_psyqo { function build-hello_psyqo {
$includes += @() $includes += @()
@@ -317,34 +350,16 @@ function build-graphis_hello {
} }
# build-graphis_hello # build-graphis_hello
function ps1-meta { param(
[Parameter(Mandatory=$true)][string[]]$sources,
[Parameter(Mandatory=$true)][string]$metadata,
[string]$out_root = (join-path $path_build 'gen'),
[string[]]$passes = @('--pre-link'),
[string[]]$extra_args = @()
)
$script = join-path $path_scripts 'ps1_meta.lua'
write-host "ps1-meta $($sources.Count) source(s), passes=$($passes -join ',')" ` -ForegroundColor Magenta
$arg_list = @($passes) + @('--metadata', $metadata) + @('--out-root', $out_root) + @($extra_args)
foreach ($s in $sources) { $arg_list += @('--source', $s) }
& luajit $script @arg_list
if ($LASTEXITCODE -ne 0) {
write-error "ps1-meta failed (exit $LASTEXITCODE). Aborting."
exit $LASTEXITCODE
}
}
function build-gte_hello { function build-gte_hello {
$includes += @() $includes += @()
$path_module = join-path $path_code 'gte_hello' $path_module = join-path $path_code 'gte_hello'
$path_duffle = join-path $path_code 'duffle' $path_duffle = join-path $path_code 'duffle'
$path_atom_metadata = join-path $path_duffle 'word_count.metadata.h' $path_atom_metadata = join-path $path_duffle 'word_count.metadata.h'
$path_build_gen = join-path $path_build 'gen'
$source_dirs = @($path_duffle, $path_module) $src_c = join-path $path_module 'hello_gte.c'
$atom_sources = Get-SourceFiles -paths $source_dirs -extensions @('.h', '.c') ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen
ps1-meta -sources $atom_sources -metadata $path_atom_metadata -out_root (join-path $path_build 'gen')
$assemble_args = @() $assemble_args = @()
$assemble_args += $f_debug $assemble_args += $f_debug
@@ -360,7 +375,6 @@ function build-gte_hello {
# assemble-unit $src_asm $module_asm $includes $assemble_args # assemble-unit $src_asm $module_asm $includes $assemble_args
$src_c = join-path $path_module 'hello_gte.c'
$module_c = join-path $path_build 'hello_gte_c.o' $module_c = join-path $path_build 'hello_gte_c.o'
$compile_args = @() $compile_args = @()
@@ -386,27 +400,17 @@ function build-gte_hello {
make-binary $elf $exe make-binary $elf $exe
# Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start). # Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start).
ps1-meta -sources $atom_sources -metadata $path_atom_metadata ` ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
-out_root (join-path $path_build 'gen') `
-passes @('--post-link') `
-extra_args @('--elf', $elf)
# F' + G' splice: collapse 9 objcopy subprocess invocations into 3. $dwarfLineBin = join-path $path_build_gen 'hello_gte.dwarf_line.bin'
# - 1 call: 3x --update-section for F' (line / aranges / rnglists) $dwarfArangesBin = join-path $path_build_gen 'hello_gte.dwarf_aranges.bin'
# - 1 call: 3x --update-section for G' (info / abbrev / str) $dwarfRnglistsBin = join-path $path_build_gen 'hello_gte.dwarf_rnglists.bin'
# - 1 call: 2x --add-section for G' (loc / loclists — these don't exist in the source ELF)
# - 1 call: 1x --set-section-flags (.rodata / .data enable code flag)
# = 4 objcopy calls (was 9; saved 5 spawns).
$dwarfLineBin = join-path (join-path $path_build 'gen') 'hello_gte.dwarf_line.bin'
$dwarfArangesBin = join-path (join-path $path_build 'gen') 'hello_gte.dwarf_aranges.bin'
$dwarfRnglistsBin = join-path (join-path $path_build 'gen') 'hello_gte.dwarf_rnglists.bin'
$injectElf = join-path $path_build 'hello_gte.dwarf-injected.elf' $injectElf = join-path $path_build 'hello_gte.dwarf-injected.elf'
if ((Test-Path $dwarfLineBin) -and (Test-Path $dwarfArangesBin) -and (Test-Path $dwarfRnglistsBin)) if ((Test-Path $dwarfLineBin) -and (Test-Path $dwarfArangesBin) -and (Test-Path $dwarfRnglistsBin))
{ {
Write-Host "[build] DWARF-injecting $elf -> $injectElf" Write-Host "[build] DWARF-injecting $elf -> $injectElf"
Copy-Item -LiteralPath $elf -Destination $injectElf -Force Copy-Item -LiteralPath $elf -Destination $injectElf -Force
# Objcopy call: 3x --update-section for (line, aranges, rnglists).
# Single objcopy call: 3x --update-section for F' (line, aranges, rnglists).
$f_args = @( $f_args = @(
"--update-section=.debug_line=$dwarfLineBin", "--update-section=.debug_line=$dwarfLineBin",
"--update-section=.debug_aranges=$dwarfArangesBin", "--update-section=.debug_aranges=$dwarfArangesBin",
@@ -419,12 +423,11 @@ function build-gte_hello {
return; return;
} }
# G' 5-section splice: 3 update-section (info / abbrev / str) + 2 add-section (loc / loclists). $dwarfInfoBin = join-path $path_build_gen 'hello_gte.dwarf_info.bin'
$dwarfInfoBin = join-path (join-path $path_build 'gen') 'hello_gte.dwarf_info.bin' $dwarfAbbrevBin = join-path $path_build_gen 'hello_gte.dwarf_abbrev.bin'
$dwarfAbbrevBin = join-path (join-path $path_build 'gen') 'hello_gte.dwarf_abbrev.bin' $dwarfStrBin = join-path $path_build_gen 'hello_gte.dwarf_str.bin'
$dwarfStrBin = join-path (join-path $path_build 'gen') 'hello_gte.dwarf_str.bin' $dwarfLocBin = join-path $path_build_gen 'hello_gte.dwarf_loc.bin'
$dwarfLocBin = join-path (join-path $path_build 'gen') 'hello_gte.dwarf_loc.bin' $dwarfLoclistsBin = join-path $path_build_gen 'hello_gte.dwarf_loclists.bin'
$dwarfLoclistsBin = join-path (join-path $path_build 'gen') 'hello_gte.dwarf_loclists.bin'
$g_args = @( $g_args = @(
"--update-section=.debug_info=$dwarfInfoBin", "--update-section=.debug_info=$dwarfInfoBin",
"--update-section=.debug_abbrev=$dwarfAbbrevBin", "--update-section=.debug_abbrev=$dwarfAbbrevBin",
@@ -441,7 +444,7 @@ function build-gte_hello {
# Baked atoms execute from RAM but are emitted as C data arrays, so their ELF sections lack SHF_EXECINSTR. # Baked atoms execute from RAM but are emitted as C data arrays, so their ELF sections lack SHF_EXECINSTR.
# GDB discards line rows for non-code sections. Mark only the debug-copy sections executable. # GDB discards line rows for non-code sections. Mark only the debug-copy sections executable.
# The shipping ELF and PS-EXE remain byte/flag unchanged. # The original ELF and PS-EXE remain byte/flag unchanged.
& $Objcopy ` & $Objcopy `
--set-section-flags ".rodata=alloc,load,readonly,code,contents" ` --set-section-flags ".rodata=alloc,load,readonly,code,contents" `
--set-section-flags ".data=alloc,load,data,code,contents" ` --set-section-flags ".data=alloc,load,data,code,contents" `
@@ -449,7 +452,8 @@ function build-gte_hello {
if ($LASTEXITCODE -ne 0) { if ($LASTEXITCODE -ne 0) {
Write-Warning "[build] atom-section flag update failed (exit $LASTEXITCODE); removing $injectElf" Write-Warning "[build] atom-section flag update failed (exit $LASTEXITCODE); removing $injectElf"
Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue
} else { }
else {
Write-Host "[build] DWARF-injected ELF: $injectElf" Write-Host "[build] DWARF-injected ELF: $injectElf"
} }
} }
+1848 -237
View File
File diff suppressed because it is too large Load Diff
+7 -6
View File
@@ -11,11 +11,10 @@
--- ``` --- ```
--- ---
--- That small bootstrap: (a) locates this helper via `arg[0]` / `debug.getinfo`, --- That small bootstrap: (a) locates this helper via `arg[0]` / `debug.getinfo`,
--- (b) loads it (which sets `package.path` + `package.cpath` via cached `git rev-parse`), --- (b) loads it (which sets `package.path` + `package.cpath`),
--- (c) at the bottom calls `require("duffle")` (now resolvable since `package.path` was just set) and returns the duffle M. --- (c) at the bottom calls `require("duffle")` (now resolvable since `package.path` was just set) and returns the duffle M.
--- Net effect: the caller gets the duffle module in one statement; no separate `dofile(...)` + `require("duffle")` dance. --- Net effect: the caller gets the duffle module in one statement; no separate `dofile(...)` + `require("duffle")` dance.
--- ---
--- Replaces the prior 2-line (entry) or 4-line (pass) pattern that had the call site do its own path resolution + duplicated setup.
local M = {} local M = {}
@@ -27,9 +26,6 @@ local CACHE_KEY = "__duffle_repo_root__"
--- parent of the directory containing this script. We derive it directly from `debug.getinfo(1, "S").source` --- parent of the directory containing this script. We derive it directly from `debug.getinfo(1, "S").source`
--- (returns `@<path>` for the currently-running chunk). --- (returns `@<path>` for the currently-running chunk).
--- ---
--- Replaces the prior `io.popen("git rev-parse --show-toplevel")` approach, which cost ~100-180ms per
--- LuaJIT process on Windows due to git's CLI startup. The path-derive approach costs <1ms.
---
--- If `debug.getinfo` can't parse this script's path (shouldn't happen — dofile always populates source), --- If `debug.getinfo` can't parse this script's path (shouldn't happen — dofile always populates source),
--- return nil and let `M.setup()` fail loud. --- return nil and let `M.setup()` fail loud.
--- @return string|nil --- @return string|nil
@@ -61,7 +57,12 @@ end
function M.setup() function M.setup()
local repo_root = find_repo_root() local repo_root = find_repo_root()
if not repo_root then if not repo_root then
io.stderr:write("[duffle_paths] git rev-parse failed -- not in a git repo?\n") -- Unreachable in practice: find_repo_root() derives the repo root from this script's
-- own source path via debug.getinfo(1, "S").source (no subprocess, no git CLI, <1ms).
-- A nil return means the source path did not match the expected
-- <repo>/scripts/duffle_paths.lua layout — a packaging bug, not a "missing git repo"
-- condition. os.exit(2) is retained so a real failure surfaces loud rather than
-- silently producing an unconfigured module table.
os.exit(2) os.exit(2)
end end
+6 -137
View File
@@ -240,7 +240,7 @@ end
-- Pure-Lua 5.3 LEB128 readers (no `bit` library). `2^shift` arithmetic matches the existing parser. -- Pure-Lua 5.3 LEB128 readers (no `bit` library). `2^shift` arithmetic matches the existing parser.
-- Offsets are 0-based; returns (value, next_pos). -- Offsets are 0-based; returns (value, next_pos).
-- Track A Task 10: promoted from `local function` to M.* exports so passes/dwarf_injection.lua -- Promoted from `local function` to M.* exports so passes/dwarf_injection.lua
-- can import them as file-scope locals per the 2nd-caller lift precedent -- can import them as file-scope locals per the 2nd-caller lift precedent
-- (the uleb128 + sleb128 encoders were promoted the same way). -- (the uleb128 + sleb128 encoders were promoted the same way).
function M.read_uleb128_at(buf, pos) function M.read_uleb128_at(buf, pos)
@@ -404,10 +404,8 @@ local function read_form_value(buf, str_buf, pos, form)
end end
--- Read a `DW_FORM_ref_sig8` value at 0-based offset `pos` from `buf`. --- Read a `DW_FORM_ref_sig8` value at 0-based offset `pos` from `buf`.
--- Returns the low 4 bytes (LE) as `low`, the high 4 bytes (LE) as `high`, and --- Returns the low 4 bytes (LE) as `low`, the high 4 bytes (LE) as `high`, and the cursor position after the 8-byte value as `next_pos`.
--- the cursor position after the 8-byte value as `next_pos`. --- Callers that need the full type-unit + type-offset pair (e.g. to resolve a type identifier embedded as a signature)
--- Callers that need the full type-unit + type-offset pair
--- (e.g. to resolve a type identifier embedded as a signature)
--- should use this directly rather than going through `read_form_value`, --- should use this directly rather than going through `read_form_value`,
--- which only exposes the low 4 bytes to preserve its existing (value, next_pos) return shape. --- which only exposes the low 4 bytes to preserve its existing (value, next_pos) return shape.
--- @param buf string --- @param buf string
@@ -427,8 +425,7 @@ end
-- --
-- Unit header layout (from pos 0): -- Unit header layout (from pos 0):
-- unit_length(4) + version(2) + unit_type(1) + address_size(1) + debug_abbrev_offset(4) -- unit_length(4) + version(2) + unit_type(1) + address_size(1) + debug_abbrev_offset(4)
-- -- followed by type_unit_specific fields: -- followed by type_unit_specific fields: type_signature(8) + type_offset(4)
-- type_signature(8) + type_offset(4)
-- The type_signature is at byte offset 8 of the body (right after debug_abbrev_offset). -- The type_signature is at byte offset 8 of the body (right after debug_abbrev_offset).
-- @param info string -- the .debug_info section bytes -- @param info string -- the .debug_info section bytes
-- @param target_sig_lo integer -- low 4 bytes (LE) of the desired signature -- @param target_sig_lo integer -- low 4 bytes (LE) of the desired signature
@@ -619,7 +616,7 @@ end
--- - ELF32 symtab entry = 16 bytes (`st_name:4 + st_value:4 + st_size:4 + st_info:1 + st_other:1 + st_shndx:2`); offsets within each entry are zero-based wire offsets. --- - ELF32 symtab entry = 16 bytes (`st_name:4 + st_value:4 + st_size:4 + st_info:1 + st_other:1 + st_shndx:2`); offsets within each entry are zero-based wire offsets.
--- - Direct Lua `string.byte`/`string.sub`/`string.find` boundaries receive `+ 1`. --- - Direct Lua `string.byte`/`string.sub`/`string.find` boundaries receive `+ 1`.
--- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded. --- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded.
--- - We strip the `code_` prefix to match the previous `read_nm` output. --- - The `code_` prefix is stripped (MipsAtom_ macros emit bare atom names, no `code_` prefix).
--- - `st_size > 0` filter excludes undefined/imported symbols. --- - `st_size > 0` filter excludes undefined/imported symbols.
--- @param elf_path Path --- @param elf_path Path
--- @return table<string, {integer, integer}> --- @return table<string, {integer, integer}>
@@ -657,7 +654,7 @@ function M.read_nm(elf_path)
-- Extract the name from .strtab (null-terminated C string). -- Extract the name from .strtab (null-terminated C string).
local name_end = strtab:find("\0", st_name_off + 1, true) or (st_name_off + 1) local name_end = strtab:find("\0", st_name_off + 1, true) or (st_name_off + 1)
local name = strtab:sub(st_name_off + 1, name_end - 1) local name = strtab:sub(st_name_off + 1, name_end - 1)
-- Filter: keep all symbol-table symbols (atoms emit their name as the bare `<name>` since the `code_` prefix was removed from the MipsAtom_ macro). -- Filter: keep all symbol-table symbols (atoms emit their name as the bare `<name>` — MipsAtom_ macros strip the `code_` prefix).
-- The atoms_source_map pass already filters out non-atom symbols via the source-map.txt cross-ref. -- The atoms_source_map pass already filters out non-atom symbols via the source-map.txt cross-ref.
if name and #name > 0 then if name and #name > 0 then
local st_value = M.read_u32_le(symtab, entry_off + SYM_ST_VALUE) local st_value = M.read_u32_le(symtab, entry_off + SYM_ST_VALUE)
@@ -794,132 +791,4 @@ end
-- I/O helpers: atoms source-map + native directory glob -- I/O helpers: atoms source-map + native directory glob
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- Parse a FORMAT_VERSION <expected_version> atoms-meta file (sourcemap or provenance).
--- Shared by M.parse_source_map_file + M.parse_provenance_file.
--- The two callers differ only in how they parse WORD lines; that's `extract_word(line)`.
--- Returns the standard `{name -> {total, words}}` shape.
--- Returns `{}` on format-version mismatch (and logs to stderr).
--- @param path string
--- @param expected_version integer
--- @param extract_word fun(line: string): table|nil -- caller-supplied per-line parser
--- @return table<string, table>
function M.parse_atom_records(path, expected_version, extract_word)
local out = {}
local cur_name, cur_words = nil, {}
for raw in io.lines(path) do
local line = raw
if line:match("^#") then
local ver = line:match("^# FORMAT_VERSION%s+(%d+)")
if ver and tonumber(ver) ~= expected_version then
io.stderr:write(string.format(
"[elf_dwarf.parse_atom_records] version mismatch (got %s, expected %d) in %s\n",
ver, expected_version, path))
return {}
end
-- skip other comments
elseif line:sub(1, 4) == "ATOM" then
-- ATOM <name> "<abs-source-path>" <total>
local _, _, name = line:find("ATOM%s+(%S+)%s+\"[^\"]*\"%s+(%d+)")
if name then
cur_name = name
cur_words = {}
out[name] = { total = 0, words = cur_words }
end
elseif line == "ENDATOM" then
-- Update the recorded total from the entries count
-- (matches the `lines[1] = lines[1]:gsub(" 0$", " " .. total)` patch in atoms_source_map.lua:170).
if cur_name and out[cur_name] then
out[cur_name].total = #cur_words
end
cur_name, cur_words = nil, {}
elseif line:sub(1, 4) == "WORD" and cur_name then
local field = extract_word(line)
if field then
cur_words[#cur_words + 1] = field
end
end
end
return out
end
--- Parse a FORMAT_VERSION <expected_version> `*.atoms.sourcemap.txt` file.
--- Returns `{name -> {total = N, words = {{pos, line}, ...}}}`.
--- Returns `{}` on format-version mismatch (and logs to stderr).
---
--- **Wire format** (emitted by `passes/atoms_source_map.lua`):
--- ```
--- # FORMAT_VERSION <n>
--- ATOM <name> "<abs-source-path>" <total>
--- WORD <n> LINE <line> TEXT <text...>
--- ...
--- ENDATOM
--- ```
---
--- **Conventions:** the in-memory shape uses `{pos, line, text}`
--- (`atoms_source_map.lua:142`); the `.txt` file uses `WORD <n>` so the parser maps `n` → `pos` field name.
--- @param sm_path Path
--- @param expected_version integer -- expected FORMAT_VERSION line
--- @return table<string, table>
function M.parse_source_map_file(sm_path, expected_version)
return M.parse_atom_records(sm_path, expected_version, function(line)
local _, n, _, src_line = line:find("WORD%s+(%d+)%s+LINE%s+(%d+)")
if n and src_line then
return { pos = tonumber(n), line = tonumber(src_line) }
end
end)
end
--- Parse a FORMAT_VERSION <expected_version> `*.atoms.provenance.txt` file.
--- Returns `{name -> {total = N, words = {{pos, call_file, call_line, comp_name, comp_file, comp_line}, ...}}}`.
--- Returns `{}` on format-version mismatch (and logs to stderr).
---
--- **Wire format** (emitted by `passes/atoms_source_map.lua`):
--- ```
--- # FORMAT_VERSION <n>
--- ATOM <name> "<abs-source-path>" <total>
--- WORD <n> CALL <src-file>:<src-line> RAW
--- WORD <n> CALL <src-file>:<src-line> MACRO <comp_name> "<comp-file>:<comp-line>"
--- ...
--- ENDATOM
--- ```
---
--- **Used by** `passes/dwarf_injection.lua` to:
--- - group consecutive MACRO rows into component invocations (one `DW_TAG_inlined_subroutine` each)
--- - emit abstract `DW_TAG_subprogram` per unique component name
--- - extend `.debug_line` so stepping into a `mac_X(...)` lands on the component's source line.
--- @param prov_path string -- path to *.atoms.provenance.txt
--- @param expected_version integer -- expected FORMAT_VERSION line
--- @return table<string, table>
function M.parse_provenance_file(prov_path, expected_version)
return M.parse_atom_records(prov_path, expected_version, function(line)
-- Two accepted shapes:
-- WORD <n> CALL <call-file>:<call-line> RAW
-- WORD <n> CALL <call-file>:<call-line> MACRO <comp_name> "<comp-file>:<comp-line>"
local pos, call_file, call_line, comp_name, comp_file, comp_line =
line:match('WORD%s+(%d+)%s+CALL%s+(.-):(%d+)%s+MACRO%s+(%S+)%s+"([^"]*):(%d+)"')
if pos then
return {
pos = tonumber(pos),
call_file = call_file,
call_line = tonumber(call_line),
comp_name = comp_name,
comp_file = comp_file,
comp_line = tonumber(comp_line),
}
end
-- RAW row.
local raw_pos, raw_file, raw_line = line:match('WORD%s+(%d+)%s+CALL%s+(.-):(%d+)%s+RAW')
if raw_pos then
return {
pos = tonumber(raw_pos),
call_file = raw_file,
call_line = tonumber(raw_line),
comp_name = nil,
comp_file = nil,
comp_line = nil,
}
end
end)
end
return M return M
+13 -13
View File
@@ -3,16 +3,16 @@
# Wrapper for the tape-atom step-debug helpers. # Wrapper for the tape-atom step-debug helpers.
# The 9 user commands are defined here as STUBS (degraded-state messages). # The 9 user commands are defined here as STUBS (degraded-state messages).
# The real implementations + the per-atom data tables are emitted by `passes/atoms_source_map.lua` # The real implementations + the per-atom data tables are emitted by `passes/atoms_source_map.lua`
# (post-link invocation: `ps1_meta.lua --atoms-source-map --gdb-runtime --elf <elf>`) into `build/gen/gdb_tape_atoms_runtime.gdb`. # (post-link invocation: `ps1_meta.lua --atoms-source-map --gdb-runtime --elf <elf>`) into `build/gdb_tape_atoms_runtime.gdb`.
# Sourcing that file RE-DEFINES the commands with real implementations. # Sourcing that file RE-DEFINES the commands with real implementations.
# #
# If `build/gen/gdb_tape_atoms_runtime.gdb` is missing or stale, the stubs remain (E1: no source map). # If `build/gdb_tape_atoms_runtime.gdb` is missing or stale, the stubs remain (E1: no source map).
# The user just needs to re-run `build_psyq.ps1` to regenerate. # The user just needs to re-run `build_psyq.ps1` to regenerate.
# ── Stub commands (defined here so they're always present, even if the runtime file is missing). The runtime file overrides these if sourced. ── # ?? Stub commands (defined here so they're always present, even if the runtime file is missing). The runtime file overrides these if sourced. ??
define tape_atoms define tape_atoms
echo "[gdb_tape_atoms] STUB: runtime file build/gen/gdb_tape_atoms_runtime.gdb not found." echo "[gdb_tape_atoms] STUB: runtime file build/gdb_tape_atoms_runtime.gdb not found."
echo "[gdb_tape_atoms] STUB: run .\\build_psyq.ps1 to regenerate, then re-source this file." echo "[gdb_tape_atoms] STUB: run .\\build_psyq.ps1 to regenerate, then re-source this file."
end end
document tape_atoms document tape_atoms
@@ -21,35 +21,35 @@ document tape_atoms
end end
define break_atom define break_atom
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1." echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
end end
document break_atom document break_atom
Set a breakpoint at the start of tape atom <name>. STUB state. Set a breakpoint at the start of tape atom <name>. STUB state.
end end
define step_atom define step_atom
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1." echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
end end
document step_atom document step_atom
Resume execution until the next atom boundary. STUB state. Resume execution until the next atom boundary. STUB state.
end end
define next_atom define next_atom
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1." echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
end end
document next_atom document next_atom
Alias for step_atom. STUB state. Alias for step_atom. STUB state.
end end
define where_in_atom define where_in_atom
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1." echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
end end
document where_in_atom document where_in_atom
Report current atom name, .rodata addr, word offset, and source line (if known). STUB state. Report current atom name, .rodata addr, word offset, and source line (if known). STUB state.
end end
define stepi_inside_atom define stepi_inside_atom
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1." echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
end end
document stepi_inside_atom document stepi_inside_atom
One MIPS-instruction step, then where_in_atom. STUB state. One MIPS-instruction step, then where_in_atom. STUB state.
@@ -89,17 +89,17 @@ document wave_ctx
end end
# ── Source the runtime file (re-defines commands with real impls + data). ── # ?? Source the runtime file (re-defines commands with real impls + data). ??
# Try to source from project-root-relative path first (the typical case). # Try to source from project-root-relative path first (the typical case).
# If the user is in a different CWD, the source will fail and stubs remain. # If the user is in a different CWD, the source will fail and stubs remain.
# The runtime file path is computed relative to the ELF's source map convention (build/gen/gdb_tape_atoms_runtime.gdb). # The runtime file path is computed relative to the ELF's source map convention (build/gdb_tape_atoms_runtime.gdb).
echo [gdb_tape_atoms] Wrapper loaded. Sourcing runtime file... echo [gdb_tape_atoms] Wrapper loaded. Sourcing runtime file...
# Suppress the "Redefine command" prompts that would otherwise appear when the runtime file overrides the 9 stub commands defined above. # Suppress the "Redefine command" prompts that would otherwise appear when the runtime file overrides the 9 stub commands defined above.
# The runtime's `define` blocks are intended to overwrite there's no ambiguity to confirm. # The runtime's `define` blocks are intended to overwrite ? there's no ambiguity to confirm.
set confirm off set confirm off
# Source the runtime file (re-defines commands with real impls + data). # Source the runtime file (re-defines commands with real impls + data).
source build/gen/gdb_tape_atoms_runtime.gdb source build/gdb_tape_atoms_runtime.gdb
set confirm on set confirm on
echo [gdb_tape_atoms] Runtime sourced successfully (9 commands now have real implementations). echo [gdb_tape_atoms] Runtime sourced successfully (9 commands now have real implementations).
-5
View File
@@ -28,11 +28,6 @@ param(
$ErrorActionPreference = 'Stop' $ErrorActionPreference = 'Stop'
$gdbInitPath = [System.IO.Path]::GetFullPath((Join-Path $PSScriptRoot '..\build\gen\hello_gte.gdbinit'))
if (-not (Test-Path -LiteralPath $gdbInitPath -PathType Leaf)) {
Write-Warning "Generated GDB skip sidecar missing (non-fatal): $gdbInitPath. Run the GTE build to regenerate it; debugger launch will continue without generated skip-over commands."
}
# ── Pre-checks ── # ── Pre-checks ──
foreach ($p in @($PcsxPath, $ExePath, $HelperZip)) { foreach ($p in @($PcsxPath, $ExePath, $HelperZip)) {
if (-not (Test-Path $p)) { if (-not (Test-Path $p)) {
+112 -154
View File
@@ -1,27 +1,19 @@
--- passes/annotation.lua — Atom-annotation DSL validator. --- passes/annotation.lua — Atom-annotation DSL validator.
--- ---
--- Validates `MipsAtom_(name) atom_info(atom_bind(Binds_X), atom_reads(...), atom_writes(...)) { ... }` declarations in source files. --- Validates `MipsAtom_(name) atom_info(atom_bind(Binds_X), atom_reads(...), atom_writes(...)) { ... }` declarations in source files.
--- Also reads: `Binds_*` struct declarations (`typedef Struct_(Binds_X) { ... };`) --- Also reads `Binds_*` struct declarations (`typedef Struct_(Binds_X) { ... };`).
--- ---
--- Source scanning: done ONCE upstream by `duffle.scan_source()` (ps1_meta.lua pre-scans each source and stashes the result in `src.scan`). --- `duffle.scan_source()` scans each source once upstream; `ps1_meta.lua` stores that result in `src.scan`.
--- ---
--- Writes: --- Ownership: the canonical `ctx.shared.corpus` supplies cross-source registries, while each `src.scan` supplies its source's declarations and bodies.
--- - `<ctx.out_root>/<dir_basename>.errors.h` — one per module, with `#error` directives on findings (the C compile will surface the error) --- A context without `ctx.shared.corpus` is rejected with an explicit canonical-corpus message.
--- - The annotations.txt report is rendered by `passes/report.lua` from the per-module results stashed in `ctx.flags._annot_results`
---
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex, Lua 5.3 compatible
-- Bootstrap: same as entry scripts. See `ps1_meta.lua` for the rationale. -- Bootstrap follows the entry scripts; `scripts/duffle_paths.lua` sets package.path and package.cpath. See `ps1_meta.lua` for the rationale.
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath). -- `debug.getinfo(1, "S").source` locates this file for standalone and orchestrated runs, then `duffle_paths.lua` returns the loaded `duffle` module.
-- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator.
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
local write_file = duffle.write_file
local ensure_dir = duffle.ensure_dir
-- The annotation pass now consults the source-derived registries built by scan_source: -- The annotation pass reads the source-derived registries from scan_source:
-- * pipe_ctx.register_alias_registry — for atom_dbg_reg_default(R_X, ...) and atom_reg_types(R_X, ...) member-identity checks -- * pipe_ctx.register_alias_registry — for atom_dbg_reg_default(R_X, ...) and atom_reg_types(R_X, ...) member-identity checks
-- * pipe_ctx.type_name_registry — for atom_dbg_reg_default(<T>, ...) and atom_reg_types(<T>, ...) type-identity checks -- * pipe_ctx.type_name_registry — for atom_dbg_reg_default(<T>, ...) and atom_reg_types(<T>, ...) type-identity checks
@@ -45,8 +37,6 @@ local ensure_dir = duffle.ensure_dir
--- @field project_root string --- @field project_root string
--- @field upstream table<string, table> --- @field upstream table<string, table>
--- @field flags table --- @field flags table
--- @field flags._annot_results table[] -- stashed by annotation pass; consumed by report.lua
--- @field dry_run boolean
--- @field verbose boolean --- @field verbose boolean
--- @class PassResult --- @class PassResult
@@ -64,15 +54,15 @@ local ensure_dir = duffle.ensure_dir
--- @field writes string[] -- R_* names (write targets) --- @field writes string[] -- R_* names (write targets)
--- @field errors string[]|nil -- parse-time errors from scan_source (atom_info body malformed) --- @field errors string[]|nil -- parse-time errors from scan_source (atom_info body malformed)
--- @class SkipOverMarker -- sub-shape of scan_source.lua's @class SkipOverMarker --- @class DebugSkipMarker -- sub-shape of scan_source.lua's @class DebugSkipMarker
--- @field marker_kind string -- exact marker ident (always "atom_dbg_skip_over") --- @field marker_kind string -- exact marker ident read from source. Only "atom_dbg_skip" (bare) is positive.
--- @field marker_line integer --- @field marker_line integer
--- @field args string|nil -- trimmed text inside the parens (nil when has_parens is false) --- @field args string|nil -- trimmed text inside the parens (nil when has_parens is false)
--- @field has_parens boolean --- @field has_parens boolean
--- @field is_bare boolean -- true iff marker_kind == "atom_dbg_skip" AND has_parens == false (the only positive form)
--- @field pending boolean -- true while awaiting the following declaration --- @field pending boolean -- true while awaiting the following declaration
--- @field superseded_by_marker_line integer|nil -- set on a marker that was bumped out of the pending slot --- @field superseded_by_marker_line integer|nil -- set on a marker that was bumped out of the pending slot
--- @field target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed --- @field target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed
--- @field declaration_line integer|nil
--- @class Finding --- @class Finding
--- @field line integer -- source line (or 0 for pass-level) --- @field line integer -- source line (or 0 for pass-level)
@@ -106,14 +96,11 @@ local ensure_dir = duffle.ensure_dir
-- Per-check functions (the CHECK_RULES table's payload) -- Per-check functions (the CHECK_RULES table's payload)
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- --
-- Each check has a uniform `append_to_findings` shape (errors[] / warnings[] / info[]). --- The dispatcher in `validate()` routes each result by convention: existence checks write errors[] and shape checks write warnings[].
-- The dispatcher in `validate()` decides which findings list each check writes to — by convention, --- `macro_word_drift` writes errors[] for missing or mismatched metadata and info[] for a match.
-- "existence" checks (declaration must exist, struct must exist) write errors[]; "shape" checks
-- (writes/reads must be wave-context) write warnings[].
-- The `macro_word_drift` check writes both errors[] (missing/mismatch) and info[] (match).
--- Check: every annotated atom must have a matching MipsAtom_(name) declaration. --- Check: every annotated atom must have a matching MipsAtom_(name) declaration.
--- @param a AtomAnnotation --- @param a AtomAnnotation
--- @param pipe_ctx PipeCtx --- @param pipe_ctx PipeCtx
--- @param findings Findings --- @param findings Findings
local function check_atom_decl_exists(a, pipe_ctx, findings) local function check_atom_decl_exists(a, pipe_ctx, findings)
@@ -141,10 +128,8 @@ local function check_unique_annotation(pipe_ctx, findings)
end end
--- Check: BIND atoms must reference a real Binds_* struct. --- Check: BIND atoms must reference a real Binds_* struct.
--- Emitting a warning here keeps the annotation pass from being stop-on-error for the common test-fixture case, --- I keep this as a warning so the annotation pass can report the common test-fixture case; `check_abi_handoff` in static analysis supplies the build-stopping error.
--- while still surfacing the issue in the report. --- @param a AtomAnnotation
--- The static-analysis report remains the source of truth for build-stopping errors.
--- @param a AtomAnnotation
--- @param pipe_ctx PipeCtx --- @param pipe_ctx PipeCtx
--- @param findings Findings --- @param findings Findings
local function check_binds_struct_exists(a, pipe_ctx, findings) local function check_binds_struct_exists(a, pipe_ctx, findings)
@@ -160,7 +145,7 @@ end
--- Check: TAPE_WORDS(mac_X, N) ↔ WORD_COUNT(mac_X, N) drift. --- Check: TAPE_WORDS(mac_X, N) ↔ WORD_COUNT(mac_X, N) drift.
--- Three outcomes: missing (error), mismatch (error), match (info). --- Three outcomes: missing (error), mismatch (error), match (info).
--- @param m MacroEntry --- @param m MacroEntry
--- @param wc table<string, integer> -- the shared word-count table (from ctx.shared.word_counts) --- @param wc table<string, integer> -- the shared word-count table (from ctx.shared.word_counts)
--- @param findings Findings --- @param findings Findings
local function check_macro_word_drift(m, wc, findings) local function check_macro_word_drift(m, wc, findings)
@@ -185,10 +170,9 @@ local function check_macro_word_drift(m, wc, findings)
} }
end end
--- Check: atom_dbg_reg_default(R_X, <type>) must target a register declared as a debug-visible alias in `pipe_ctx.register_alias_registry`, --- Check: atom_dbg_reg_default(R_X, <type>) targets an alias in `pipe_ctx.register_alias_registry` and a type in `pipe_ctx.type_name_registry`.
--- with a type name found in `pipe_ctx.type_name_registry`. --- Pointer depth remains bounded to 0 or 1, and duplicate defaults remain errors.
--- Pointer depth is still bounded to 0 or 1. Duplicate defaults are still detected. --- @param _src SourceFile -- unused (kept for the per_source shape)
--- @param _src SourceFile -- unused (kept for the per_source shape)
--- @param pipe_ctx PipeCtx --- @param pipe_ctx PipeCtx
--- @param findings Findings --- @param findings Findings
local function check_semantic_reg_defaults(_src, pipe_ctx, findings) local function check_semantic_reg_defaults(_src, pipe_ctx, findings)
@@ -236,11 +220,9 @@ local function check_semantic_reg_defaults(_src, pipe_ctx, findings)
end end
end end
--- Check: atom_reg_types(R_X, <type>) entries must point to a register declared in `pipe_ctx.register_alias_registry`, with a type name found in `pipe_ctx.type_name_registry`. --- Check: atom_reg_types(R_X, <type>) entries target an alias in `pipe_ctx.register_alias_registry` and a type in `pipe_ctx.type_name_registry`.
--- The alias ident `R_<n>` now encodes the GPR identity only for entries that are explicitly opted in via the bare `atom_reg` marker. --- A bare `atom_reg` marker opts the `R_<n>` alias into GPR identity; references to R_T0..R_T3 require the same explicit marker.
--- R_T0..R_T3 are intentionally NOT auto-included (per the prototype principle: no auto-include of wave-context; explicit opt-in only). --- @param _src SourceFile
--- The check fires for any R_T0..R_T3 reference that hasn't been opted in via `#define atom_reg`.
--- @param _src SourceFile
--- @param pipe_ctx PipeCtx --- @param pipe_ctx PipeCtx
--- @param findings Findings --- @param findings Findings
local function check_atom_reg_types(_src, pipe_ctx, findings) local function check_atom_reg_types(_src, pipe_ctx, findings)
@@ -270,8 +252,8 @@ local function check_atom_reg_types(_src, pipe_ctx, findings)
end end
end end
--- Check: atom_view(Binds_X) entries must reference a real Binds_* struct and that struct must declare at least one field. --- Check: atom_view(Binds_X) entries reference a Binds_* struct with at least one field.
--- @param _src SourceFile --- @param _src SourceFile
--- @param pipe_ctx PipeCtx --- @param pipe_ctx PipeCtx
--- @param findings Findings --- @param findings Findings
local function check_atom_view_layout(_src, pipe_ctx, findings) local function check_atom_view_layout(_src, pipe_ctx, findings)
@@ -299,8 +281,7 @@ local function check_atom_view_layout(_src, pipe_ctx, findings)
end end
end end
--- Check: Binds_* structs may not have duplicate field names --- Check: Binds_* structs require unique field names because atom_view uses those names for typed-field lookup in gdb.
--- (they would defeat the typed-field name lookup that atom_view exposes in gdb).
--- @param _src SourceFile --- @param _src SourceFile
--- @param pipe_ctx PipeCtx --- @param pipe_ctx PipeCtx
--- @param findings Findings --- @param findings Findings
@@ -323,26 +304,30 @@ local function check_binds_no_duplicate_fields(_src, pipe_ctx, findings)
end end
end end
-- Check: skip-over markers must satisfy shape + placement constraints. -- Check: debug-skip markers must satisfy shape + placement constraints.
--- Walks the priority list once; at most one error is appended per marker so that a single source-level defect does not cascade into multiple findings. --- Walks the priority list once; each marker produces at most one error, so one source defect yields one finding.
--- Priority order (first defect wins): --- Priority order (first defect wins):
--- 1. has_parens == false -> requires parentheses: marker() --- 1. marker_kind ~= "atom_dbg_skip" -> legacy/renamed spelling (use `atom_dbg_skip`)
--- 2. args ~= "" -> takes no arguments --- 2. marker_kind == "atom_dbg_skip" AND has_parens -> parenthesized form (the marker is bare-only)
--- 3. superseded_by_marker_line -> duplicate marker (cite superseding line) --- 3. args ~= "" -> takes no arguments
--- 4. pending + no target_kind -> dangling (no following declaration) --- 4. superseded_by_marker_line -> duplicate marker (cite superseding line)
--- 5. unsupported target_kind -> marker precedes an unrelated declaration --- 5. pending + no target_kind -> dangling (no following declaration)
--- Valid markers before whole-atom / bare-component / proc-component declarations emit no error and remain in src.scan.skip_over.atoms / .components. --- 6. unsupported target_kind -> marker precedes an unrelated declaration
--- @param marker SkipOverMarker --- Valid markers stamp `debug_skip` on whole-atom, bare-component, and proc-component declaration records in scan_source.lua.
--- @param marker DebugSkipMarker
--- @param _pipe_ctx PipeCtx -- unused today; kept for plex-shape consistency with per_annot --- @param _pipe_ctx PipeCtx -- unused today; kept for plex-shape consistency with per_annot
--- @param findings Findings --- @param findings Findings
local function check_skip_marker(marker, _pipe_ctx, findings) local function check_skip_marker(marker, _pipe_ctx, findings)
local kind = marker.marker_kind local kind = marker.marker_kind
local line = marker.marker_line local line = marker.marker_line
if not marker.has_parens then -- Left `scan.debug_skip_markers` with production records for `atom_dbg_skip` only; other identifiers take the walker's unrelated branch.
if marker.has_parens then
findings.errors[#findings.errors + 1] = { findings.errors[#findings.errors + 1] = {
line = line, line = line,
msg = string.format("%s marker at line %d requires parentheses: marker()", kind, line), msg = string.format("%s marker at line %d must be bare; the parenthesized form is no longer accepted (use `atom_dbg_skip MipsAtom_(name) { ... }`)",
kind, line),
} }
return return
end end
@@ -385,15 +370,11 @@ local function check_skip_marker(marker, _pipe_ctx, findings)
end end
end end
--- Migration warning emitted alongside the new registry-membership check. --- Warn when a source references an unregistered alias.
--- ---
--- R_TapePtr / R_AtomJmp / R_PrimCursor / R_FaceCursor / R_VertBase / R_OtBase --- R_TapePtr, R_AtomJmp, R_PrimCursor, R_FaceCursor, R_VertBase, and R_OtBase opt in through `#define atom_reg` in lottes_tape.h.
--- are the wave-context aliases opted in via `#define atom_reg` in lottes_tape.h (Task 21). --- When a source uses an unregistered R_X, this check emits one pass-level info entry for that source and directs C-ABI register names to explicit alias registration.
--- Any source referencing an R_X that's NOT in the registry will trip the new check; a single pass-level info entry --- @param _src SourceFile
--- (emitted only when at least one such rejection lands in this source) tells users where to look.
---
--- This check is a stop-gap until users migrate off raw C-ABI register names.
--- @param _src SourceFile
--- @param pipe_ctx PipeCtx --- @param pipe_ctx PipeCtx
--- @param findings Findings --- @param findings Findings
local function check_wave_context_migration(_src, pipe_ctx, findings) local function check_wave_context_migration(_src, pipe_ctx, findings)
@@ -425,7 +406,7 @@ end
-- per_annot(annot, pipe_ctx, findings) -- runs once per AtomAnnotation -- per_annot(annot, pipe_ctx, findings) -- runs once per AtomAnnotation
-- post(pipe_ctx, findings) -- runs once after all per_annot calls complete (full-corpus aggregation) -- post(pipe_ctx, findings) -- runs once after all per_annot calls complete (full-corpus aggregation)
-- per_macro(macro, wc, findings) -- runs once per TAPE_WORDS / _Pragma macro declaration -- per_macro(macro, wc, findings) -- runs once per TAPE_WORDS / _Pragma macro declaration
-- per_skip_marker(marker, pipe_ctx, findings) -- runs once per src.scan.skip_over.markers entry -- per_skip_marker(marker, pipe_ctx, findings) -- runs once per src.scan.debug_skip_markers entry
-- --
-- Adding a new check = 1 row here + 1 function above. The `validate()` dispatch loop never needs editing. -- Adding a new check = 1 row here + 1 function above. The `validate()` dispatch loop never needs editing.
@@ -446,14 +427,56 @@ local CHECK_RULES = {
-- Validation -- Validation
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- --
-- Pure check: read from src.scan, run validations, emit findings. -- Pure check: read from src.scan, run validations, emit findings. The scan was done once upstream.
-- No source walking; no parsing. The scan was done once upstream.
--- Validate one source against its pre-scanned SourceScan payload. --- Builds one pass-wide pipe_ctx from the merged `corpus.*` registries and source-ordered `corpus.atom_infos`; per-source declarations and bodies remain in `src.scan`.
--- The module ownership contract above requires callers to construct `ctx.shared.corpus` through `build_ctx`; the error message below enforces that gate.
--- @param ctx PassCtx --- @param ctx PassCtx
--- @param src SourceFile --- @return PipeCtx
local function build_corpus_pipe_ctx(ctx)
local corpus = ctx.shared and ctx.shared.corpus
if not corpus then
error("annotation requires ctx.shared.corpus "
.. "(the canonical corpus is the source of truth; "
.. "no per-source fallback is supported)", 0)
end
-- `corpus.atom_infos` preserves source order and duplicates; I precompute counts here for `check_unique_annotation` and the per-source checks.
local annot_counts = {}
for _, info in ipairs(corpus.atom_infos or {}) do
if info and info.atom_name then
annot_counts[info.atom_name] = (annot_counts[info.atom_name] or 0) + 1
end
end
-- Every consumer of these fields observes mutations via the canonical corpus without independently mutable registry construction.
return {
-- Cross-source lookup tables from corpus.
register_alias_registry = corpus.register_alias_registry or {},
type_name_registry = corpus.type_name_registry or {},
atom_views = corpus.atom_views or {},
atom_ctxs = corpus.atom_ctxs or {},
atom_phases = corpus.atom_phases or {},
binds_by_name = corpus.binds_by_name or {},
atoms_by_name = corpus.atoms_by_name or {},
-- Corpus-wide ordered list of atom_info records (source-order + duplicates).
atom_infos_list = corpus.atom_infos or {},
-- Corpus-wide annotation count aggregation (post-rule consumes this).
annot_counts = annot_counts,
-- Corpus-wide collisions (recorded by scan_source.merge_corpus_registries).
collisions = corpus.collisions or {},
-- `check_macro_word_drift` reads `corpus.word_counts`, populated by word_count_eval.run.
word_counts = corpus.word_counts or {},
}
end
--- Validate one source against its pre-scanned SourceScan payload + the corpus-wide pipe_ctx.
--- @param ctx PassCtx
--- @param src SourceFile
--- @param corpus_pipe_ctx PipeCtx|nil -- built once per pass from corpus registries; nil builds the same projection here.
--- @return AnnotatedResult --- @return AnnotatedResult
local function validate(ctx, src) local function validate(ctx, src, corpus_pipe_ctx)
corpus_pipe_ctx = corpus_pipe_ctx or build_corpus_pipe_ctx(ctx)
local scan = src.scan local scan = src.scan
-- Project the pre-scanned atoms to the AtomEntry shape this pass needs. -- Project the pre-scanned atoms to the AtomEntry shape this pass needs.
@@ -479,9 +502,7 @@ local function validate(ctx, src)
} }
end end
-- Build pipe_ctx (Fleury: expose structure). Pre-compute everything the per-check functions need. -- Build a per-source pipe_ctx: shared lookups come from `corpus_pipe_ctx`, while declarations, bodies, types, views, defaults, and occurrences come from `src.scan`.
-- Single source of truth for atom / binds / annotation-count lookups.
-- pipe_ctx.types / pipe_ctx.atom_views / pipe_ctx.seen_defaults are projected from the scan payload so per_source check rules can iterate.
local seen_defaults = {} local seen_defaults = {}
for reg, _ in pairs(scan.types or {}) do for reg, _ in pairs(scan.types or {}) do
seen_defaults[reg] = (seen_defaults[reg] or 0) + 1 seen_defaults[reg] = (seen_defaults[reg] or 0) + 1
@@ -494,33 +515,25 @@ local function validate(ctx, src)
local pipe_ctx = { local pipe_ctx = {
atom_index = {}, atom_index = {},
binds_index = {}, binds_index = {},
annot_counts = {}, annot_counts = corpus_pipe_ctx.annot_counts,
types = scan.types or {}, types = scan.types or {},
type_occurrences = scan.type_occurrences or {}, type_occurrences = scan.type_occurrences or {},
atom_views = scan.atom_views or {}, atom_views = scan.atom_views or {},
seen_defaults = seen_defaults, seen_defaults = seen_defaults,
atom_infos_list = atom_infos_list, atom_infos_list = atom_infos_list,
binds_list = scan.binds or {}, binds_list = scan.binds or {},
-- Project the source-derived registries from the scan payload so per_source checks consult them instead of the deleted -- See the module ownership contract; these shared lookup tables come from corpus_pipe_ctx.
-- SEMANTIC_DEFAULT_REGS / KNOWN_REG_DEFAULT_TYPES / etc. register_alias_registry = corpus_pipe_ctx.register_alias_registry,
register_alias_registry = scan.register_alias_registry or {}, type_name_registry = corpus_pipe_ctx.type_name_registry,
type_name_registry = scan.type_name_registry or {},
} }
for _, a in ipairs(atoms) do pipe_ctx.atom_index [a.name] = a end for _, a in ipairs(atoms) do pipe_ctx.atom_index [a.name] = a end
for _, b in ipairs(scan.binds) do pipe_ctx.binds_index[b.name] = b end for _, b in ipairs(scan.binds) do pipe_ctx.binds_index[b.name] = b end
for _, a in ipairs(annots) do
if a.name then
pipe_ctx.annot_counts[a.name] = (pipe_ctx.annot_counts[a.name] or 0) + 1
end
end
-- Findings live in a single struct with three lists (errors / warnings / info). -- Findings live in a single struct with three lists (errors / warnings / info).
-- Each check writes to the list appropriate for its severity. -- Each check writes to the list appropriate for its severity.
local findings = { errors = {}, warnings = {}, info = {} } local findings = { errors = {}, warnings = {}, info = {} }
-- Propagate parse-time errors from scan_source's atom_info parsing. -- Lift parse-time errors already recorded in scan_source's atom_info payload into this pass's findings list.
-- These are errors found in the atom_info(...) body itself (e.g., malformed args).
-- They are pre-existing in the scan payload — we just lift them into our findings list.
for _, a in ipairs(annots) do for _, a in ipairs(annots) do
if a.errors then if a.errors then
for _, msg in ipairs(a.errors) do for _, msg in ipairs(a.errors) do
@@ -544,11 +557,9 @@ local function validate(ctx, src)
if rule.post then rule.post(pipe_ctx, findings) end if rule.post then rule.post(pipe_ctx, findings) end
end end
-- Per-skip-marker rules. -- scan_source records each marker in scan.debug_skip_markers; this loop validates each record independently and emits at most one error per marker.
-- Each raw marker recorded by scan_source (in scan.skip_over.markers) is validated independently; -- Valid markers stamp `debug_skip = true` on the following atom or component declaration, which downstream consumers read directly.
-- the check emits at most one error per marker. local skip_markers = scan.debug_skip_markers or {}
-- Valid markers stay attached to scan.skip_over.atoms /.components for dwarf_injection.lua consumer.
local skip_markers = scan.skip_over and scan.skip_over.markers or {}
for _, marker in ipairs(skip_markers) do for _, marker in ipairs(skip_markers) do
for _, rule in ipairs(CHECK_RULES) do for _, rule in ipairs(CHECK_RULES) do
if rule.per_skip_marker then rule.per_skip_marker(marker, pipe_ctx, findings) end if rule.per_skip_marker then rule.per_skip_marker(marker, pipe_ctx, findings) end
@@ -556,7 +567,7 @@ local function validate(ctx, src)
end end
-- Per-macro rules (TAPE_WORDS vs WORD_COUNT drift). -- Per-macro rules (TAPE_WORDS vs WORD_COUNT drift).
local wc = ctx.shared.word_counts local wc = corpus_pipe_ctx.word_counts
for _, m in ipairs(scan.macros) do for _, m in ipairs(scan.macros) do
for _, rule in ipairs(CHECK_RULES) do for _, rule in ipairs(CHECK_RULES) do
if rule.per_macro then rule.per_macro(m, wc, findings) end if rule.per_macro then rule.per_macro(m, wc, findings) end
@@ -572,8 +583,8 @@ local function validate(ctx, src)
-- Information summary (always emitted). -- Information summary (always emitted).
findings.info[#findings.info + 1] = { findings.info[#findings.info + 1] = {
line = 0, line = 0,
msg = string.format("scanned: %d atom(s), %d annotation(s), %d macro-word-decl(s), %d binds struct(s)", msg = string.format("scanned: %d atom(s), %d annotation(s), %d macro-word-decl(s), %d binds struct(s)"
#atoms, #annots, #scan.macros, #scan.binds), , #atoms, #annots, #scan.macros, #scan.binds),
} }
return { return {
@@ -587,52 +598,6 @@ local function validate(ctx, src)
} }
end end
-- ════════════════════════════════════════════════════════════════════════════
-- Per-DIRECTORY (per-module) output: errors.h + annotations.txt
-- ════════════════════════════════════════════════════════════════════════════
--- Render `<dir_basename>.errors.h` with `#error` directives for every error found across all sources in the directory.
--- Empty directories (no errors, no atoms) produce no file.
local function emit_module_errors_h(ctx, dir_basename, atoms_count, errors, sources)
if ctx.dry_run then return nil end
if atoms_count == 0 and #errors == 0 then
return nil
end
local out_path = ctx.out_root .. "/" .. dir_basename .. ".errors.h"
local lines = {
"// Auto-generated by ps1_meta.lua (passes/annotation.lua) — DO NOT EDIT",
string.format("// Module: %s Sources: %d", dir_basename, #sources),
"#pragma once",
"",
}
if #errors == 0 then
lines[#lines + 1] = "// annotation pass OK"
else
for _, e in ipairs(errors) do
local src_tag = ""
if e.source then
local src_name = e.source:match("([^/\\]+)$") or e.source
src_tag = src_name .. ": "
end
lines[#lines + 1] = string.format('#error "%s%s (line %d)"', src_tag, e.msg, e.line)
end
end
ensure_dir(ctx.out_root)
write_file(out_path, table.concat(lines, "\n") .. "\n")
return out_path
end
--- Stash aggregated per-module results for the report pass to consume.
local function emit_module_annotations_stub(ctx, dir, dir_basename, atoms_count)
ctx.flags = ctx.flags or {}
ctx.flags._annot_results = ctx.flags._annot_results or {}
ctx.flags._annot_results[#ctx.flags._annot_results + 1] = {
dir = dir,
dir_basename = dir_basename,
atoms_count = atoms_count,
}
end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- M.run — orchestrator entry -- M.run — orchestrator entry
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -651,22 +616,22 @@ function M.run(ctx)
local errors = {} local errors = {}
local warnings = {} local warnings = {}
-- Per-DIRECTORY (per-module) aggregation. Group sources by `src.dir`, validate every source in the dir, then emit ONE errors.h per dir. -- Build the shared pipe_ctx once for this run; every validate() call sees the same cross-source registries.
-- `ctx.by_dir` is pre-computed in build_ctx (shared across all passes). -- The corpus owns the canonical cross-source registries; per-source scans retain body / declaration ownership.
local by_dir = ctx.by_dir or duffle.group_sources_by_dir(ctx.sources) local corpus_pipe_ctx = build_corpus_pipe_ctx(ctx)
local corpus = ctx.shared.corpus
-- Group `corpus.sources_by_dir` by module, validate every source in each bucket, and emit one errors.h per directory.
local by_dir = (corpus and corpus.sources_by_dir) or {}
for dir, dir_sources in pairs(by_dir) do for dir, dir_sources in pairs(by_dir) do
local dir_basename = dir:match("([^/\\]+)$") or dir local dir_basename = dir:match("([^/\\]+)$") or dir
local dir_atoms = 0 local dir_atoms = 0
local dir_errors = {} local dir_errors = {}
local dir_warnings = {} local dir_warnings = {}
-- Per-source validate() results, cached for the report pass (it reads from this instead of re-validating each source).
ctx.flags = ctx.flags or {}
ctx.flags._annot_source_results = ctx.flags._annot_source_results or {}
for _, src in ipairs(dir_sources) do for _, src in ipairs(dir_sources) do
local result = validate(ctx, src) local result = validate(ctx, src, corpus_pipe_ctx)
result.source = src.path -- tag for downstream rendering result.source = src.path -- tag for downstream rendering
ctx.flags._annot_source_results[src.path] = result -- stash so report.lua reads from cache instead of re-running validate()
dir_atoms = dir_atoms + #result.atoms dir_atoms = dir_atoms + #result.atoms
for _, e in ipairs(result.errors) do for _, e in ipairs(result.errors) do
dir_errors[#dir_errors + 1] = { line = e.line, msg = e.msg, source = src.path } dir_errors[#dir_errors + 1] = { line = e.line, msg = e.msg, source = src.path }
@@ -677,13 +642,6 @@ function M.run(ctx)
warnings [#warnings + 1] = { line = w.line, msg = w.msg } warnings [#warnings + 1] = { line = w.line, msg = w.msg }
end end
end end
local err_path = emit_module_errors_h(ctx, dir_basename, dir_atoms, dir_errors, dir_sources)
if err_path then
table.insert(outputs, { errors_h = err_path })
end
emit_module_annotations_stub(ctx, dir, dir_basename, dir_atoms)
end end
return { outputs = outputs, errors = errors, warnings = warnings } return { outputs = outputs, errors = errors, warnings = warnings }
+200 -377
View File
@@ -1,26 +1,21 @@
--- passes/atoms_source_map.lua — Per-.word source-line map emitter for tape atoms. --- passes/atoms_source_map.lua — Per-.word source-line map emitter for tape atoms.
--- ---
--- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`) --- Writer: this pass, given `atom.paths` (the per-atom mutable surface owned by `emission_model`). Readers:
--- for `MipsAtom_(name)` (kind="atom"), `MipsAtomComp_` / `MipsAtomComp_Proc_` (kind="comp_*"), --- `passes/dwarf_injection.lua` (synthesizes DW_TAG_inlined_subroutine + per-word line program rows) and the gdb-runtime
--- and `MipsCode code_<name>` (kind="raw_atom") declarations. --- wrapper at `scripts/gdb/gdb_tape_atoms.gdb` (loads the source map via `source <path>`).
--- Walks each atom's pre-tokenized body (`{{tok=string, rel=integer}, ...}` from `duffle.tokenize_body`),
--- counts per-token word contributions via `ctx.shared.word_counts`, and emits one
--- `WORD N LINE L TEXT T` line per `.word` to `<out_root>/<basename>.atoms.sourcemap.txt`.
--- ---
--- **Two output forms** (per the workspace's per-emission-form pattern from --- Inputs from `atom.paths`: the ordered `items` stream, dense `word_events`, `invocations` views. Outputs:
--- `guide_metaprogram_ssdl.md`): --- one `WORD N LINE L TEXT T` line per emitted `.word`, plus the per-word provenance form that DWARF synthesis consumes.
--- 1. **Canonical text form** — `<out_root>/<basename>.atoms.sourcemap.txt`. ---
--- Format-version-tagged for forward-compat. --- Two output forms:
--- Lives in `<out_root>/` (build/gen). --- 1. Markdown form: Handled by `passes/report.lua` (writes `<module>.atoms.md`).
--- Matches the convention used by `annotation.lua` (`<out_root>/<basename>.errors.h`) + `static_analysis.lua` (`<out_root>/<basename>.static_analysis.txt`). --- The render functions `render_source_map` + `render_provenance` are exported for `report.lua` to call directly.
--- Compile artifacts (`*.macs.h`, `*.offsets.h`) stay in `<source_dir>/gen/`. --- Compile artifacts (`*.macs.h`, `*.offsets.h`) stay in `<source_dir>/gen/`.
--- 2. **gdb-runtime form** — `<ctx.out_root>/gdb_tape_atoms_runtime.gdb` --- 2. `gdb_tape_atoms_runtime.gdb`: Post-link opt-in (`ctx.flags.gdb_runtime`),
--- (pure gdb command script; addresses pre-computed via `nm`; the 9 user commands defined as `define ... end` blocks). --- so the gdb wrapper script + the generated runtime script share the same canonical location.
--- Emitted ONLY when `ctx.flags.gdb_runtime` is true AND `ctx.flags.elf_path` points to an existing ELF. --- Triggered by `--post-link` or `--gdb-runtime`.
--- The gdb runtime form lets `gdb-multiarch --without-python` users (the common case on Windows MinGW builds)
--- load the source-map data via `source <path>` — no Python/Tcl/Guile required.
--- ---
--- **Output format** (canonical text form): --- Output forma (sourcemap.txt form):
--- ``` --- ```
--- # FORMAT_VERSION 1 --- # FORMAT_VERSION 1
--- # auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT --- # auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT
@@ -34,13 +29,7 @@
--- ENDATOM --- ENDATOM
--- ``` --- ```
--- ---
--- Marker calls (`atom_label(...)`, `atom_offset(...)`) emit 0 `.word`s. --- Marker records are zero-width in `atom.paths.items`, so they emit no WORD rows in the dense word view.
--- They share the same walking convention as `passes/offsets.lua :: scan_atom_body`:
--- Markers do NOT advance the word-offset counter, but if a marker is bundled on the same token with a trailing instruction
--- (e.g. `atom_label(foo) load_half_u(...)`), the trailing instruction's word count is added. This matches `offsets.lua :: count_marker_rest`.
---
--- **Conventions:** tabs (1/level), EmmyLua annotations, no regex,
--- Lua 5.3 compatible.
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Module-scope requires + package.path setup -- Module-scope requires + package.path setup
@@ -50,10 +39,8 @@
-- (works both standalone + when require'd). `duffle_paths.lua` sets package.path then returns `require("duffle")` -- (works both standalone + when require'd). `duffle_paths.lua` sets package.path then returns `require("duffle")`
-- at the bottom, so the dofile value IS the duffle module. -- at the bottom, so the dofile value IS the duffle module.
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
local elf_dwarf = require("elf_dwarf") local elf_dwarf = require("elf_dwarf")
local word_count_eval = require("word_count_eval")
local count_token_words = word_count_eval.count_token_words
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Constants -- Constants
@@ -63,199 +50,88 @@ local count_token_words = word_count_eval.count_token_words
-- the gdb runtime loader rejects mismatches (E2). -- the gdb runtime loader rejects mismatches (E2).
local FORMAT_VERSION = 1 local FORMAT_VERSION = 1
-- Marker-call identifiers (mirrors offsets.lua:33-34).
local LABEL_MARKER = "atom_label"
local OFFSET_MARKER = "atom_offset"
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Type declarations -- Type declarations
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- @class AtomSourceMapCtx --- @class AtomSourceMapCtx
--- @field sources table[] -- SourceScan payload per source (from `ctx.sources`)
--- @field shared table -- `ctx.shared` --- @field shared table -- `ctx.shared`
--- @field shared.word_counts table -- macro name -> word count (populated by word-counts + components passes) --- @field shared.corpus table -- source-order registry; single writer is build_ctx
--- @field shared.word_counts table
--- @field out_root string -- output root (e.g. "build/gen") --- @field out_root string -- output root (e.g. "build/gen")
--- @field dry_run boolean -- if true, compute but don't write
--- @field flags table -- `ctx.flags`; reads `flags.gdb_runtime` + `flags.elf_path` --- @field flags table -- `ctx.flags`; reads `flags.gdb_runtime` + `flags.elf_path`
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Helpers -- Atom-path renderers
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- ════════════════════════════════════════════════════════════════════════════ --- Join word boundaries (from `items`) to per-word call text + source lines (from `word_events`).
-- Provenance emission
-- ════════════════════════════════════════════════════════════════════════════
-- Component-macro invocation prefix (mirrors components.lua's MAC_PREFIX).
local MAC_PREFIX = "mac_"
local MAC_PREFIX_LEN = 4
--- Strip the `mac_` prefix from a token's leading identifier.
--- Returns nil if the identifier doesn't start with `mac_`
--- (so non-component tokens like `load_half_u`, `nop2`, `gte_cmdw_*` fall through cleanly).
--- @param tok string
--- @return string|nil
local function strip_mac_prefix_from_token(tok)
local leading = duffle.read_ident(tok, 1)
if not leading then return nil end
if leading:sub(1, MAC_PREFIX_LEN) == MAC_PREFIX then
return leading:sub(MAC_PREFIX_LEN + 1)
end
return nil
end
--- Fetch the per-word body lines for a `mac_X(...)` invocation.
--- Walks the component's pre-tokenized body in lockstep with `count_token_words` and attributes each emitted `.word`
--- to a source line via `idx.line_of(...)`.
--- Atom labels (`atom_label(...)`) emit 0 `.word`s and are skipped.
--- @param bare string|nil -- the bare component name (e.g. `gte_load_tri_verts`)
--- @param comp_body_index table
--- @param wc table
--- @return table|nil -- list of source lines, 1-based by word position
local function fetch_body_lines(bare, comp_body_index, wc)
if not (bare and comp_body_index) then return nil end
local idx = comp_body_index[bare]
if not (idx and idx.body_tokens and idx.line_of) then return nil end
local lines = {}
for _, bt in ipairs(idx.body_tokens) do
local bt_tok = duffle.trim(bt.tok or "")
if bt_tok ~= "" then
local leading = duffle.read_ident(bt_tok, 1)
local bt_words
if leading == "atom_label" or leading == "atom_offset" then
bt_words = 0
else
bt_words = count_token_words(bt_tok, wc)
end
if bt_words > 0 then
local body_line = idx.line_of(idx.body_off + bt.rel)
for _ = 1, bt_words do lines[#lines + 1] = body_line end
end
end
end
return lines
end
--- Unified per-word entry walker. `mode` is "sourcemap" (3 fields) or "provenance" (8 fields including component + body-line lookup).
--- Returns (entries, total_words). Markers contribute 0 entries.
--- @param atom table --- @param atom table
--- @param src table
--- @param wc table
--- @param mode string -- "sourcemap" | "provenance"
--- @param comp table|nil -- shared.components map (provenance only)
--- @param comp_body_index table|nil -- per-source body index (provenance only)
--- @return table[], integer --- @return table[], integer
local function compute_word_entries(atom, src, wc, mode, comp, comp_body_index) local function canonical_word_entries(atom)
local entries = {} local paths = atom.paths or {}
local pos = 0 local events = paths.word_events or {}
for _, t in ipairs(atom.body_tokens) do local word_items = {}
local tok = t.tok for _, item in ipairs(paths.items or {}) do
local rel = t.rel if item.kind == "word" then word_items[#word_items + 1] = item end
local words
if duffle.is_marker_token(tok) then
words = duffle.count_marker_rest(tok, wc, count_token_words)
else
words = count_token_words(tok, wc)
end
-- Provenance-only: resolve component + body_lines (one fetch per token).
local comp_name, comp_line, comp_path, comp_kind
local body_lines
if mode == "provenance" then
local bare = strip_mac_prefix_from_token(tok)
if bare and comp and comp[bare] then
comp_name = bare
comp_line = comp[bare].line
comp_path = comp[bare].path
comp_kind = comp[bare].kind
end
if comp_name then body_lines = fetch_body_lines(bare, comp_body_index, wc) end
end
if words > 0 then
local line = src.scan.line_of(atom.body_off + rel)
local text = duffle.trim(tok):gsub("[\t\r\n]+", " ")
for i = 1, words do
local entry
if mode == "provenance" then
entry = {
pos = pos,
line = line,
text = text,
comp_name = comp_name,
comp_line = comp_line,
comp_path = comp_path,
comp_kind = comp_kind,
body_line = body_lines and body_lines[i],
}
else -- "sourcemap" (default)
entry = { pos = pos, line = line, text = text }
end
entries[#entries + 1] = entry
pos = pos + 1
end
end
end end
return entries, pos
local entries = {}
for index, event in ipairs(events) do
local item = word_items[index] or {}
entries[#entries + 1] = {
pos = event.i or (index - 1),
line = event.call_line or item.line or 0,
text = event.call_text or item.call_text or "",
body_line = event.body_line or item.body_line or item.line or 0,
invocation = (event.outermost_invocation_id
and paths.invocations
and paths.invocations[event.outermost_invocation_id]) or nil,
}
end
return entries, #events
end end
--- Render one atom's provenance stanza. Format: --- Render one atom's provenance stanza. Format 1 line shapes:
--- `WORD N CALL <src-path>:<src-line> MACRO <name> "<def-path>:<def-line>" [BODY <line>]` (for component words) --- `WORD N CALL <src-path>:<src-line> MACRO <name> "<def-path>:<def-line>" BODY <line>` (component invocation)
--- `WORD N CALL <src-path>:<src-line> RAW` (for direct instructions) --- `WORD N CALL <src-path>:<src-line> RAW` (raw `.word` outside any mac_* component)
--- `BODY <line>` is the source line of THIS specific word within the macro body --- Component identity comes from the outermost invocation record; the count-table lookup confirms the component was
--- (lottes_tape.h:N where N is the per-word body line). --- declared in `corpus.word_counts` (populated by word_count_eval + components passes).
--- Absent for RAW rows and for component rows whose component declaration could not be indexed (older pass combinations / external macros). --- @param src table
--- Downstream consumers (dwarf_injection, tests) fall back to DefLine / comp_line when BODY is absent. --- @param atom table
--- Returns (lines, total_words). --- @param wc table -- identity alias of corpus.word_counts
--- @param src table
--- @param atom table
--- @param wc table
--- @param comp table -- shared.components map
--- @param comp_body_index table -- per-source component body index: bare_name -> {body_off, body_tokens, line_of}
--- @return string[], integer --- @return string[], integer
local function emit_provenance_stanza(src, atom, wc, comp, comp_body_index) local function emit_provenance_stanza(src, atom, wc)
local lines = {} local lines = {}
local rel_path = src.path:gsub("\\", "/") local rel_path = src.path:gsub("\\\\", "/")
local entries, total = compute_word_entries(atom, src, wc, "provenance", comp, comp_body_index) local entries, total = canonical_word_entries(atom)
-- ATOM header line with placeholder total (patched after we know it).
lines[#lines + 1] = string.format('ATOM %s "%s" 0', atom.raw_name or atom.name, rel_path) lines[#lines + 1] = string.format('ATOM %s "%s" 0', atom.raw_name or atom.name, rel_path)
for _, pe in ipairs(entries) do for _, entry in ipairs(entries) do
if pe.comp_name then local inv = entry.invocation
local body_suffix = "" local macro_count = inv and wc["mac_" .. inv.component_name]
if pe.body_line then if inv and macro_count ~= nil then
body_suffix = " BODY " .. tostring(pe.body_line) lines[#lines + 1] = string.format(
end 'WORD %d CALL %s:%d MACRO %s "%s:%d" BODY %d',
lines[#lines + 1] = string.format('WORD %d CALL %s:%d MACRO %s "%s:%d"%s', entry.pos, rel_path, entry.line, inv.component_name,
pe.pos, rel_path, pe.line, pe.comp_name, pe.comp_path, pe.comp_line, body_suffix) inv.def_path or "", inv.def_line or 0, entry.body_line)
else else
lines[#lines + 1] = string.format("WORD %d CALL %s:%d RAW", pe.pos, rel_path, pe.line) lines[#lines + 1] = string.format(
"WORD %d CALL %s:%d RAW", entry.pos, rel_path, entry.line)
end end
end end
-- Patch the placeholder total in the ATOM header line.
lines[1] = lines[1]:gsub(" 0$", " " .. tostring(total)) lines[1] = lines[1]:gsub(" 0$", " " .. tostring(total))
lines[#lines + 1] = "ENDATOM" lines[#lines + 1] = "ENDATOM"
return lines, total return lines, total
end end
--- Build a per-source component body index keyed by the bare component name (e.g. `gte_load_tri_verts`). --- Render the full provenance file content for one source.
--- Each entry holds the data we need to map each emitted `.word` to its actual source line within the macro body:
--- body_off -- byte offset of the `{` (start of body) in the component's source file.
--- body_tokens -- list of {tok, rel} pairs; `rel` is the byte offset within the body.
--- line_of -- closure resolving byte offsets in the component's source file to lines.
--- Only `comp_bare` + `comp_proc` declarations contribute (a macro invocation can only resolve to one of those).
--- First declaration wins (subsequent redeclarations would collide; today's sources declare each component exactly once).
--- Render the full provenance file content for one source (one `.atoms.provenance.txt` per source).
--- @param src table --- @param src table
--- @param wc table --- @param wc table
--- @param comp table -- shared.components map
--- @param comp_body_index table -- cross-source component body index (built once in M.run; may be empty)
--- @return string --- @return string
local function render_provenance(src, wc, comp, comp_body_index) local function render_provenance(src, wc)
local lines = {} local lines = {}
lines[#lines + 1] = "# FORMAT_VERSION 1" lines[#lines + 1] = "# FORMAT_VERSION 1"
lines[#lines + 1] = "# auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT" lines[#lines + 1] = "# auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT"
@@ -265,62 +141,60 @@ local function render_provenance(src, wc, comp, comp_body_index)
lines[#lines + 1] = "# dwarf_injection to synthesize DW_TAG_inlined_subroutine instances + per-word" lines[#lines + 1] = "# dwarf_injection to synthesize DW_TAG_inlined_subroutine instances + per-word"
lines[#lines + 1] = "# line program rows for native source-level step into component bodies." lines[#lines + 1] = "# line program rows for native source-level step into component bodies."
-- The cross-source component body index is passed in from M.run (one global lookup shared across every source's provenance file). local function append(atom)
-- A per-source lookup would miss every component whose declaration is in another source (e.g. `gte_load_tri_verts` is declared in `lottes_tape.h` but invoked from `hello_gte_tape.c`). local stanza = emit_provenance_stanza(src, atom, wc)
for _, atom in ipairs(src.scan.atoms or {}) do
local stanza = emit_provenance_stanza(src, atom, wc, comp, comp_body_index)
for _, line in ipairs(stanza) do lines[#lines + 1] = line end for _, line in ipairs(stanza) do lines[#lines + 1] = line end
end end
for _, atom in ipairs(src.scan.atoms or {}) do
if atom.paths then append(atom) end
end
for _, atom in ipairs(src.scan.raw_atoms or {}) do for _, atom in ipairs(src.scan.raw_atoms or {}) do
local stanza = emit_provenance_stanza(src, atom, wc, comp, comp_body_index) if atom.paths then append(atom) end
for _, line in ipairs(stanza) do lines[#lines + 1] = line end
end end
return table.concat(lines, "\n") .. "\n" return table.concat(lines, "\n") .. "\n"
end end
--- Render one atom's stanza for the canonical text form (ATOM header line, N WORD lines, ENDATOM marker). --- Render one atom's stanza for the sourcemap.txt form (ATOM header line, N WORD lines, ENDATOM marker).
--- Returns (lines, total_words). --- Returns (lines, total_words).
--- @param src table --- @param src table
--- @param atom table --- @param atom table
--- @param wc table --- @param wc table
--- @return string[], integer --- @return string[], integer
local function emit_atom_stanza(src, atom, wc) local function emit_atom_stanza(src, atom)
local lines = {} local lines = {}
local rel_path = src.path:gsub("\\", "/") local rel_path = src.path:gsub("\\\\", "/")
local entries, total = compute_word_entries(atom, src, wc) local entries, total = canonical_word_entries(atom)
-- ATOM header line with placeholder total (patched after we know it).
lines[#lines + 1] = string.format('ATOM %s "%s" 0', atom.raw_name or atom.name, rel_path) lines[#lines + 1] = string.format('ATOM %s "%s" 0', atom.raw_name or atom.name, rel_path)
for _, we in ipairs(entries) do for _, entry in ipairs(entries) do
lines[#lines + 1] = string.format("WORD %d LINE %d TEXT %s", lines[#lines + 1] = string.format("WORD %d LINE %d TEXT %s",
we.pos, we.line, we.text) entry.pos, entry.line, entry.text)
end end
-- Patch the placeholder total in the ATOM header line.
lines[1] = lines[1]:gsub(" 0$", " " .. tostring(total)) lines[1] = lines[1]:gsub(" 0$", " " .. tostring(total))
lines[#lines + 1] = "ENDATOM" lines[#lines + 1] = "ENDATOM"
return lines, total return lines, total
end end
--- Render the full source map file content for one source (one .atoms.sourcemap.txt per source). --- Render the full source map file content for one source (one .atoms.sourcemap.txt per source). Mirrors offsets.lua's
--- Mirrors offsets.lua's `project_atoms` shape: scan.atoms + scan.raw_atoms, no kind filter. --- `project_atoms` shape: scan.atoms + scan.raw_atoms, no kind filter.
--- @param src table --- @param src table
--- @param wc table --- @param wc table
--- @return string --- @return string
local function render_source_map(src, wc) local function render_source_map(src)
local lines = {} local lines = {}
lines[#lines + 1] = "# FORMAT_VERSION " .. FORMAT_VERSION lines[#lines + 1] = "# FORMAT_VERSION " .. FORMAT_VERSION
lines[#lines + 1] = "# auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT" lines[#lines + 1] = "# auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT"
for _, atom in ipairs(src.scan.atoms or {}) do local function append(atom)
local stanza = emit_atom_stanza(src, atom, wc) local stanza = emit_atom_stanza(src, atom)
for _, line in ipairs(stanza) do lines[#lines + 1] = line end for _, line in ipairs(stanza) do lines[#lines + 1] = line end
end end
for _, atom in ipairs(src.scan.raw_atoms or {}) do for _, atom in ipairs(src.scan.atoms or {}) do
local stanza = emit_atom_stanza(src, atom, wc) if atom.paths then append(atom) end
for _, line in ipairs(stanza) do lines[#lines + 1] = line end end
for _, atom in ipairs(src.scan.raw_atoms or {}) do
if atom.paths then append(atom) end
end end
return table.concat(lines, "\n") .. "\n" return table.concat(lines, "\n") .. "\n"
@@ -338,69 +212,48 @@ local function gdb_escape(s)
return (s:gsub("\\", "\\\\"):gsub('"', '\\"')) return (s:gsub("\\", "\\\\"):gsub('"', '\\"'))
end end
--- Build the list of atoms with addresses + word entries. --- Build the list of atoms with addresses + word entries. Shared helper for the gdb-runtime file emission.
--- Shared helper for the gdb-runtime file emission.
--- @param ctx PassCtx --- @param ctx PassCtx
--- @return table[] -- list of {idx, name, src_path, file_base, addr, size_bytes, words, entries} --- @return table[] -- list of {idx, name, src_path, file_base, addr, size_bytes, words, entries}
local function build_atom_table(ctx) local function build_atom_table(ctx)
local wc = (ctx.shared and ctx.shared.word_counts) or {}
local addrs = elf_dwarf.read_nm(ctx.flags.elf_path) local addrs = elf_dwarf.read_nm(ctx.flags.elf_path)
local corpus = ctx.shared and ctx.shared.corpus
local matched = {} local matched = {}
for _, src in ipairs(ctx.sources) do
if src.scan then for _, src in ipairs(corpus.source_order or {}) do
local file_base = src.path:match("([^/\\]+)$") or src.path local file_base = src.path:match("([^/\\\\]+)$") or src.path
for _, atom in ipairs(src.scan.atoms or {}) do local function append(atom)
if atom.kind == nil or atom.kind == "atom" then if not atom.paths then return end
local name = atom.raw_name or atom.name local name = atom.raw_name or atom.name
local info = addrs[name] local info = addrs[name]
if info then if not info then return end
local entries, total = compute_word_entries(atom, src, wc) local entries, total = canonical_word_entries(atom)
matched[#matched + 1] = { matched[#matched + 1] = {
name = name, name = name,
src_path = src.path, src_path = src.path,
file_base = file_base, file_base = file_base,
addr = info[1], addr = info[1],
size_bytes = info[2], size_bytes = info[2],
words = total, words = total,
entries = entries, entries = entries,
} }
end
end
end
for _, atom in ipairs(src.scan.raw_atoms or {}) do
local name = atom.name
local info = addrs[name]
if info then
local entries, total = compute_word_entries(atom, src, wc)
matched[#matched + 1] = {
name = name,
src_path = src.path,
file_base = file_base,
addr = info[1],
size_bytes = info[2],
words = total,
entries = entries,
}
end
end
end end
for _, atom in ipairs((src.scan or {}).atoms or {}) do append(atom) end
for _, atom in ipairs((src.scan or {}).raw_atoms or {}) do append(atom) end
end end
-- Deterministic order: sort by address (matches `nm` output ordering). -- Deterministic order: sort by address (matches `nm` output ordering).
table.sort(matched, function(a, b) return a.addr < b.addr end) table.sort(matched, function(a, b) return a.addr < b.addr end)
for i, a in ipairs(matched) do for i, a in ipairs(matched) do a.idx = i - 1 end
a.idx = i - 1
end
return matched return matched
end end
--- Append the 9 gdb command definitions to `lines`. Pure gdb scripting no Python, no Tcl, no Guile required. --- Append the 9 gdb command definitions to `lines`. Pure gdb scripting — addresses come from `nm`, the convenience
--- **Fully hardcoded per-atom** because gdb doesn't do nested `$` substitution in var names --- vars set in `emit_gdb_runtime` provide printf args, and each command is a static sequence of `printf` / `tbreak` /
--- `$__atom_name_$__i` inside a `while` loop is treated as one literal identifier, not a concat. --- `if ... end` blocks. The Lua pass emits N atoms' worth of lines; runtime iteration is gdb's job.
--- ---
--- Each command is a static sequence of `printf` / `tbreak` / `if ... end` blocks. --- Why hardcoded per-atom: gdb's `$` substitution doesn't concat inside var names — `$__atom_name_$__i` in a `while`
--- The Lua pass emits N atoms' worth of lines — no runtime iteration. --- loop resolves to one literal identifier, not `name_i`. Compile-time emission is the only path.
--- @param lines table -- output line buffer (mutated in place) --- @param lines table -- output line buffer (mutated in place)
--- @param matched table -- list of atom records from `build_atom_table` --- @param matched table -- list of atom records from `build_atom_table`
local function append_gdb_commands(lines, matched) local function append_gdb_commands(lines, matched)
@@ -529,31 +382,6 @@ local function append_gdb_commands(lines, matched)
lines[#lines + 1] = "end" lines[#lines + 1] = "end"
lines[#lines + 1] = "" lines[#lines + 1] = ""
-- ── show_c2 ──
-- GTE data regs (COP2). pcsx-redux's gdb stub doesn't expose COP2 (only 72 regs: 32 GPR + COP0 + FPR).
-- curl http://localhost:8080/api/v1/lua/gte
-- We keep the command definition as a stub that points the user at the plugin.
lines[#lines + 1] = "define show_c2"
lines[#lines + 1] = ' echo "[gdb_tape_atoms] show_c2: gdb stub does not expose COP2 in this build."'
lines[#lines + 1] = ' echo "[gdb_tape_atoms] Use scripts/pcsx_debug_helper.zip + curl http://localhost:8080/api/v1/lua/gte"'
lines[#lines + 1] = ' echo "[gdb_tape_atoms] (or pcsx-redux Debug > Registers window for a native view)"'
lines[#lines + 1] = "end"
lines[#lines + 1] = "document show_c2"
lines[#lines + 1] = " Stub. The gdb stub in this pcsx-redux build does not expose COP2 regs."
lines[#lines + 1] = " For GTE data + control state, use the pcsx_debug_helper Lua plugin or the"
lines[#lines + 1] = " pcsx-redux Debug > Registers window."
lines[#lines + 1] = "end"
lines[#lines + 1] = ""
-- ── show_c2ctl ──
lines[#lines + 1] = "define show_c2ctl"
lines[#lines + 1] = ' echo "[gdb_tape_atoms] show_c2ctl: see show_c2 for the same workaround."'
lines[#lines + 1] = "end"
lines[#lines + 1] = "document show_c2ctl"
lines[#lines + 1] = " Stub. Same workaround as show_c2."
lines[#lines + 1] = "end"
lines[#lines + 1] = ""
-- ── wave_ctx ── -- ── wave_ctx ──
lines[#lines + 1] = "define wave_ctx" lines[#lines + 1] = "define wave_ctx"
lines[#lines + 1] = ' printf "$t4 = R_FaceCursor 0x%08x\\n", $t4' lines[#lines + 1] = ' printf "$t4 = R_FaceCursor 0x%08x\\n", $t4'
@@ -566,9 +394,9 @@ local function append_gdb_commands(lines, matched)
lines[#lines + 1] = "end" lines[#lines + 1] = "end"
end end
--- Emit the gdb-runtime file (post-link). Pure gdb scripting — no Python. --- Emit the gdb-runtime file (post-link). Pure gdb scripting — addresses come from `mipsel-none-elf-nm -S`, get embedded
--- Reads ELF addresses via `mipsel-none-elf-nm -S`, embeds them in `<ctx.out_root>/gdb_tape_atoms_runtime.gdb` --- in `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`, and load via `set $var = ...` + `define ... end` blocks at gdb
--- so gdb loads the data via `set $var = ...` + `define ... end` blocks at source-time. --- source-time.
--- @param ctx PassCtx --- @param ctx PassCtx
local function emit_gdb_runtime(ctx) local function emit_gdb_runtime(ctx)
if not (ctx.flags and ctx.flags.gdb_runtime) then return end if not (ctx.flags and ctx.flags.gdb_runtime) then return end
@@ -626,13 +454,25 @@ local function emit_gdb_runtime(ctx)
-- Confirmation line for the source operator. -- Confirmation line for the source operator.
lines[#lines + 1] = 'printf "[gdb_tape_atoms] runtime loaded %d atoms from %s\\n", $__atom_count, $__elf_path' lines[#lines + 1] = 'printf "[gdb_tape_atoms] runtime loaded %d atoms from %s\\n", $__atom_count, $__elf_path'
local out_path = ctx.out_root .. "/gdb_tape_atoms_runtime.gdb" local out_path
if not ctx.dry_run then -- Move out of `<out_root>/gdb_tape_atoms_runtime.gdb` to `<out_root>/../gdb_tape_atoms_runtime.gdb` when the conventional `<out_root>` is `<build>/gen`
duffle.ensure_dir(duffle.dirname(out_path)) -- (any equivalent spelling — relative, absolute backslash, absolute forward-slash, trailing-separator variants).
duffle.write_file_lf(out_path, table.concat(lines, "\n") .. "\n") -- This puts the gdb runtime alongside the ELF at `build/` rather than under the report subdir.
local function ends_with_gen_dir(p)
if type(p) ~= "string" then return false end
return p:match("[/\\]gen[/\\]?$") ~= nil or p == "build/gen" or p == "build\\gen"
end end
io.stderr:write(string.format( if ends_with_gen_dir(ctx.out_root) then
"[atoms_source_map] wrote %s (%d atoms)\n", out_path, #matched)) -- Strip the trailing `/gen` segment, then write the runtime script under `build/`.
-- e.g. "C:/projects/Pikuma/ps1/build/gen" -> "C:/projects/Pikuma/ps1/build".
local parent = ctx.out_root:gsub("[/\\]gen[/\\]?$", "")
out_path = parent .. "/gdb_tape_atoms_runtime.gdb"
else
out_path = ctx.out_root .. "/gdb_tape_atoms_runtime.gdb"
end
duffle.ensure_dir(duffle.dirname(out_path))
duffle.write_file_lf(out_path, table.concat(lines, "\n") .. "\n")
-- io.stderr:write(string.format("[atoms_source_map] wrote %s (%d atoms)\n", out_path, #matched))
end end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -641,46 +481,62 @@ end
local M = {} local M = {}
--- Build the cross-source component body index used by `render_provenance` to attribute each emitted `.word` to its actual line within the macro body. -- Expose the pure render functions so `report.lua` and the focused tests can call them directly without triggering the file-emit path.
--- M.render_source_map = render_source_map
--- Components are declared in one source (the header that contains `MipsAtomComp_(ac_X)` / `MipsAtomComp_Proc_(ac_X, ...)`) M.render_provenance = render_provenance
--- but invoked from many source files (every atom body that calls `mac_X(...)`).
--- The body_offset + body_tokens + line_of live with the declaration source, so a per-source index would miss invocations from other sources. --- Render ONE atom's sourcemap stanza.
--- --- @param atom table -- atom record (must have `atom.paths` populated)
--- The cross-source index is keyed by the bare component name (`gte_load_tri_verts`, NOT `ac_gte_load_tri_verts`) --- @return string
--- `strip_mac_prefix_from_token` strips the `mac_` prefix from call-site identifiers and yields that exact bare name; function M.render_atom_source_map(atom)
--- matching it here keeps the lookup aligned with the `ctx.shared.components` map's keying convention. assert(type(atom) == "table", "render_atom_source_map: atom must be a table")
--- First declaration wins (subsequent redeclarations would collide; today's sources declare each component exactly once). assert(type(atom.paths) == "table", "render_atom_source_map: atom.paths must be a table")
--- @param ctx PassCtx local entries, total = canonical_word_entries(atom)
--- @return table<string, table> -- {[comp_name] = {body_off, body_tokens, line_of}} local lines = {}
local function build_cross_source_component_body_index(ctx) lines[#lines + 1] = string.format("ATOM %s %d", (atom.raw_name or atom.name), total)
local index = {} for _, entry in ipairs(entries) do
for _, src in ipairs(ctx.sources or {}) do lines[#lines + 1] = string.format("WORD %d LINE %d TEXT %s",
if src.scan and src.scan.atoms then entry.pos, entry.line, entry.text)
local line_of = src.scan.line_of
for _, atom in ipairs(src.scan.atoms) do
if atom.kind == "comp_bare" or atom.kind == "comp_proc" then
-- Prefer `atom.name` (stripped of `ac_` prefix); fall back to `raw_name`
-- only if the stripped name is absent (defensive — current scan-source always sets both).
local name = atom.name or atom.raw_name
if name and not index[name] then
index[name] = {
body_off = atom.body_off,
body_tokens = atom.body_tokens,
line_of = line_of,
}
end
end
end
end
end end
return index lines[#lines + 1] = "ENDATOM"
return table.concat(lines, "\n") .. "\n"
end end
--- Pass entry: emit one `<out_root>/<basename>.atoms.sourcemap.txt` per source file that contains at least one `MipsAtom_(name)` / `MipsCode code_<name>` declaration. --- Render ONE atom's provenance stanza — no per-file format header, no enumeration of other atoms.
--- Also emits `<out_root>/<basename>.atoms.provenance.txt`: ---
--- per-.word provenance with `mac_X(...)` component resolution back to the component's definition file:line + the per-word body line. --- `rel_path` is the source path (forward-slashes) embedded in every `CALL` line.
--- Optionally also emit `<ctx.out_root>/gdb_tape_atoms_runtime.gdb` when `ctx.flags.gdb_runtime` is true. --- The .md caller (report.lua) is expected to derive this once per `## <source>` heading and pass it down for each atom in that source.
--- @param atom table -- atom record (must have `atom.paths` populated)
--- @param wc table -- identity alias of `corpus.word_counts`
--- @param rel_path string -- source path (forward-slashes) for `CALL` fields
--- @return string
function M.render_atom_provenance(atom, wc, rel_path)
assert(type(atom) == "table", "render_atom_provenance: atom must be a table")
assert(type(atom.paths) == "table", "render_atom_provenance: atom.paths must be a table")
assert(type(rel_path) == "string", "render_atom_provenance: rel_path must be a string")
local entries, total = canonical_word_entries(atom)
local lines = {}
lines[#lines + 1] = string.format("ATOM %s %d", (atom.raw_name or atom.name), total)
for _, entry in ipairs(entries) do
local inv = entry.invocation
local macro_count = inv and wc and wc["mac_" .. inv.component_name]
if inv and macro_count ~= nil then
lines[#lines + 1] = string.format(
'WORD %d CALL %s:%d MACRO %s "%s:%d" BODY %d',
entry.pos, rel_path, entry.line, inv.component_name,
inv.def_path or "", inv.def_line or 0, entry.body_line)
else
lines[#lines + 1] = string.format(
"WORD %d CALL %s:%d RAW", entry.pos, rel_path, entry.line)
end
end
return table.concat(lines, "\n") .. "\n"
end
--- Pass entry. For each source that declares at least one `MipsAtom_(name)` / `MipsCode code_<name>`,
--- emit two files in `<out_root>/`: `<basename>.atoms.sourcemap.txt` (per-word call-site map) and `<basename>.atoms.provenance.txt`
--- (per-word definition + body line, resolved via the outermost `mac_X(...)` invocation).
--- When `ctx.flags.gdb_runtime` is true and `ctx.flags.elf_path` exists, also emit the post-link gdb script `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`.
--- @param ctx PassCtx --- @param ctx PassCtx
--- @return PassResult --- @return PassResult
function M.run(ctx) function M.run(ctx)
@@ -688,55 +544,22 @@ function M.run(ctx)
local errors = {} local errors = {}
local warnings = {} local warnings = {}
-- word-counts + components passes must have populated shared.word_counts. local corpus = ctx.shared and ctx.shared.corpus
-- If absent, the orchestrator wired the deps wrong — fail loud. if type(corpus) ~= "table" or type(corpus.source_order) ~= "table" then
local wc = (ctx.shared and ctx.shared.word_counts) or {} error("atoms_source_map.run requires ctx.shared.corpus.source_order (canonical corpus).", 0)
if not wc or not next(wc) then end
-- Word counts come from `corpus.word_counts` (populated by word_count_eval + components passes).
local wc = corpus.word_counts or {}
if not next(wc) then
warnings[#warnings + 1] = { warnings[#warnings + 1] = {
line = 0, line = 0,
msg = "atoms_source_map: ctx.shared.word_counts is empty; the word-counts + components passes may not have populated it. Check the PASSES dep edges.", msg = "atoms_source_map: corpus.word_counts is empty; the word-counts + components passes may not have populated it. Check the PASSES dep edges.",
} }
end end
-- shared.components map is populated by `passes/components.lua`. -- atoms.sourcemap.txt + atoms.provenance.txt content moved to report.lua via `<module>.atoms.md` markdown file.
-- Used to attribute each emitted `.word` to either a component macro or the enclosing atom body. -- This pass emits only the post-link gdb_runtime artifact (see emit_gdb_runtime below).
-- If absent, all words fall through as RAW (correct behavior — provenance is additive).
local comp = (ctx.shared and ctx.shared.components) or {}
-- Cross-source component body index.
-- Built ONCE so every source's provenance writer can resolve `mac_X(...)` invocations back to the macro's body tokens (regardless of which source declared the component).
-- Per-source copies were insufficient — the atom file (`hello_gte_tape.c`) does not contain the `MipsAtomComp_(...)` declarations,
-- so the body data would be missing for every component invocation the atom file emitted.
local comp_body_index = build_cross_source_component_body_index(ctx)
-- Always emit the canonical text form (per-source).
for _, src in ipairs(ctx.sources) do
if src.scan then
local n_atoms = src.scan.atoms and #src.scan.atoms or 0
local n_raw_atoms = src.scan.raw_atoms and #src.scan.raw_atoms or 0
if n_atoms + n_raw_atoms > 0 then
local basename = duffle.basename_no_ext(src.path)
-- (1) atoms.sourcemap.txt — per-.word line map (unchanged contract).
local sourcemap_path = ctx.out_root .. "/" .. basename .. ".atoms.sourcemap.txt"
local sourcemap_body = render_source_map(src, wc)
-- (2) atoms.provenance.txt — per-.word provenance with `mac_X(...)` component resolution back to the component's definition file:line.
-- Consumed by `passes/dwarf_injection.lua` to synthesize `DW_TAG_inlined_subroutine` instances for source-level Step Into on component invocations.
local prov_path = ctx.out_root .. "/" .. basename .. ".atoms.provenance.txt"
local prov_body = render_provenance(src, wc, comp, comp_body_index)
if not ctx.dry_run then
duffle.ensure_dir(duffle.dirname(sourcemap_path))
duffle.write_file_lf(sourcemap_path, sourcemap_body)
duffle.write_file_lf(prov_path, prov_body)
end
outputs[#outputs + 1] = { kind = "report", path = sourcemap_path }
outputs[#outputs + 1] = { kind = "report", path = prov_path }
end
end
end
-- Optionally emit the gdb-runtime form (post-link, one file per build). -- Optionally emit the gdb-runtime form (post-link, one file per build).
if ctx.flags and ctx.flags.gdb_runtime then if ctx.flags and ctx.flags.gdb_runtime then
+179 -181
View File
@@ -1,25 +1,16 @@
--- passes/components.lua — Component-macro header generator. --- passes/components.lua — Component-macro header generator.
--- ---
--- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`) --- Ownership: `corpus.word_counts`, `corpus.components`, and `corpus.component_body_index`.
--- for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations, then does per-source backward lookups --- Scanner owns `declaration_comment` and `debug_skip` on each declaration record; this pass projects both forward.
--- for the function-args string (from the preceding `FI_ MipsAtom ac_X(...)` function declaration)
--- and the preceding comment block (for LSP/IntelliSense signature docs).
--- ---
--- Emits a per-directory `<dir_basename>.macs.h` containing one `#define mac_X(sig) \` macro per component + `WORD_COUNT(mac_X, N)` --- Reads the pre-scanned SourceScan payload from `duffle.scan_source` for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations,
--- entries for downstream offset computation. --- then resolves the function-args string from the preceding `FI_ MipsAtom ac_X(...)` declaration via a backward walk.
---
--- Emits one `<dir_basename>.macs.h` per source with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
--- ---
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex, --- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
--- Lua 5.3 compatible. --- Lua 5.3 compatible.
--- @class Component
--- @field name string
--- @field body string
--- @field args string|nil
--- @field line integer
--- @field comment string|nil
--- @class M
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Module-scope requires + package.path setup -- Module-scope requires + package.path setup
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -31,7 +22,6 @@
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module. -- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
local word_count_eval = require("word_count_eval")
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Constants -- Constants
@@ -69,12 +59,10 @@ local GEN_SUBDIR = "gen"
--- @field sources SourceFile[] -- all source files in the build --- @field sources SourceFile[] -- all source files in the build
--- @field metadata_path string -- path to word_count.metadata.h --- @field metadata_path string -- path to word_count.metadata.h
--- @field shared table -- cross-pass shared state --- @field shared table -- cross-pass shared state
--- @field shared.word_counts table<string, integer> -- populated by word-counts + components
--- @field out_root string -- output root (e.g. "build/gen") --- @field out_root string -- output root (e.g. "build/gen")
--- @field project_root string -- project root (e.g. "code/") --- @field project_root string -- project root (e.g. "code/")
--- @field upstream table<string, table> -- per-pass upstream outputs --- @field upstream table<string, table> -- per-pass upstream outputs
--- @field flags table -- CLI flags --- @field flags table -- CLI flags
--- @field dry_run boolean -- if true, compute but don't write
--- @field verbose boolean -- log diagnostic info --- @field verbose boolean -- log diagnostic info
--- @class PassResult --- @class PassResult
@@ -83,11 +71,13 @@ local GEN_SUBDIR = "gen"
--- @field warnings table[] -- {line=, msg=} entries; build-succeeds --- @field warnings table[] -- {line=, msg=} entries; build-succeeds
--- @class Component --- @class Component
--- @field name string -- atom name (without `ac_` prefix) --- @field name string -- atom name (without `ac_` prefix)
--- @field body string -- brace-delimited body (without the braces) --- @field body string -- brace-delimited body (without the braces)
--- @field args string|nil -- function-args string (function form only) --- @field args string|nil -- function-args string (function form only)
--- @field line integer -- source line of the declaration --- @field line integer -- source line of the declaration
--- @field comment string|nil -- preceding `/* */` or `//` comment block (signature doc) --- @field comment string|nil -- scanner-owned `declaration_comment`; the components pass reads it from the scanner record
--- @field kind string -- "comp_bare" | "comp_proc"
--- @field debug_skip boolean -- mirror of `a.debug_skip` (scanner-owned); true iff a bare `atom_dbg_skip` marker immediately preceded the declaration
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Local helpers (file I/O + path normalization) -- Local helpers (file I/O + path normalization)
@@ -96,7 +86,11 @@ local GEN_SUBDIR = "gen"
local M = {} local M = {}
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Back-walk helpers (composed into the 2 entry points below: find_function_args_for + preceding_comment_block) -- Back-walk helpers (composed into the entry point below: find_function_args_for)
--
-- Only the function-args lookup for proc components occurs here.
-- The preceding-comment walk occur in `scan_source.lua` — `a.declaration_comment` carries the resolved comment,
-- so this file reads it forward rather than re-walking the source.
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- Find the args of the function declaration that immediately precedes a `MipsAtomComp_Proc_` invocation of the given name. --- Find the args of the function declaration that immediately precedes a `MipsAtomComp_Proc_` invocation of the given name.
@@ -108,8 +102,8 @@ local M = {}
--- We then verify the preceding context ends with `MipsAtom` --- We then verify the preceding context ends with `MipsAtom`
--- (the function-decl keyword with possible qualifiers between). --- (the function-decl keyword with possible qualifiers between).
--- ---
--- @param source string --- @param source string
--- @param name string --- @param name string
--- @param before_pos integer --- @param before_pos integer
--- @return string|nil --- @return string|nil
local function find_function_args_for(source, name, before_pos) local function find_function_args_for(source, name, before_pos)
@@ -143,79 +137,6 @@ local function find_function_args_for(source, name, before_pos)
return inner return inner
end end
--- Find the contiguous comment block immediately preceding `pos` in `source`.
--- Returns the comment text (with the `/* */` or `//` markers preserved) or an empty string if no comment is adjacent.
---
--- Used to copy signature comments from the source declaration (`MipsAtomComp_` / `MipsAtomComp_Proc_` / function decl)
--- over to the generated `mac_X` macro, so LSP/IntelliSense displays the args doc.
--- @param source string
--- @param pos integer
--- @return string
local function preceding_comment_block(source, pos)
local scan_pos = pos
local pieces = {}
while true do
-- Skip whitespace (space/tab/newline/CR) backward from `scan_pos`,
-- returning the position of the first non-whitespace char.
local non_ws = scan_pos - 1
while non_ws > 0 do
local ch = source:sub(non_ws, non_ws)
if ch == " " or ch == "\t" or ch == "\n" or ch == "\r" then
non_ws = non_ws - 1
else
break
end
end
if non_ws == 0 then break end
local is_block_close = non_ws >= 2 and source:sub(non_ws - 1, non_ws) == "*/"
local is_line_end = source:sub(non_ws, non_ws) == "\n" or source:sub(non_ws, non_ws) == "\r"
if is_block_close then
-- Find the opening `/*` for a block comment whose `*/` ends at `non_ws`.
-- Walk back from `non_ws` over `/*` candidates.
local prefix = source:sub(1, non_ws - 1)
local open_at = nil
for scan = #prefix - 1, 1, -1 do
if prefix:sub(scan, scan + 1) == "/*" then
open_at = scan
break
end
end
if not open_at then break end
-- Walk back from `open_at` over leading spaces + tabs to include the indentation before the `/*`.
local block_start = open_at
while block_start > 1 do
local ch = source:sub(block_start - 1, block_start - 1)
if ch == " " or ch == "\t" then
block_start = block_start - 1
else
break
end
end
table.insert(pieces, 1, source:sub(block_start, non_ws))
scan_pos = block_start
elseif is_line_end then
-- Walk back from `non_ws` to the start of the source line (the most recent `\n` or position 1).
local line_start = non_ws
while line_start > 1 and source:sub(line_start - 1, line_start - 1) ~= "\n" do
line_start = line_start - 1
end
local line = source:sub(line_start, non_ws)
if line:sub(1, 2) == "//" then
table.insert(pieces, 1, line)
scan_pos = line_start - 1
else
break
end
else
break
end
end
if #pieces == 0 then return "" end
return table.concat(pieces, "\n")
end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Argument-name extraction -- Argument-name extraction
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -235,7 +156,6 @@ local function extract_arg_names(args_str)
if trimmed ~= "" then if trimmed ~= "" then
-- Find the identifier at the end: walk back over trailers (whitespace + `*` + `[]`), -- Find the identifier at the end: walk back over trailers (whitespace + `*` + `[]`),
-- then walk back over the identifier chars (alnum + `_`). -- then walk back over the identifier chars (alnum + `_`).
-- Plex: inlined the 2 single-caller helpers (no 2-caller rule met).
local ident_end = #trimmed local ident_end = #trimmed
while ident_end > 0 do while ident_end > 0 do
local ch = trimmed:sub(ident_end, ident_end) local ch = trimmed:sub(ident_end, ident_end)
@@ -248,7 +168,7 @@ local function extract_arg_names(args_str)
local ident_start = ident_end local ident_start = ident_end
while ident_start > 0 do while ident_start > 0 do
local ch = trimmed:sub(ident_start, ident_start) local ch = trimmed:sub(ident_start, ident_start)
if duffle.is_alnum(ch) or ch == "_" then if duffle.is_alnum_byte(string.byte(ch)) or ch == "_" then
ident_start = ident_start - 1 ident_start = ident_start - 1
else else
break break
@@ -267,26 +187,34 @@ end
-- Component projection (read from pre-scanned SourceScan) -- Component projection (read from pre-scanned SourceScan)
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Project pre-scanned MipsAtomComp_ / MipsAtomComp_Proc_ entries into Component shape. --- Project pre-scanned MipsAtomComp_ / MipsAtomComp_Proc_ entries into Component shape.
-- Does per-source backward lookups for args (preceding function decl) and comment (preceding comment block). --- Reads the scanner-owned `declaration_comment` (resolved by scan_source.lua, skipping backward across an associated bare `atom_dbg_skip` marker when present).
-- Carries `body_tokens` forward from scan-source so word_count_rec reads from the precomputed table instead of calling duffle.tokenize_body again. --- Per-source backward lookups remain in place only for the function `args` of proc components.
-- @param source string -- the full source text (needed for backward lookups) --- That lookup is unique to components.lua and stays separate from the declaration-comment walk.
-- @param scan table -- SourceScan from duffle.scan_source --- Carries `body_tokens` forward from scan-source so word_count_rec reads from the precomputed table instead of calling duffle.tokenize_body again.
-- @return Component[] --- Carries the scanner-owned `debug_skip` flag forward so the generated projection can emit `/* atom_dbg_skip */`
--- before the authored comment and so `update_canonical_components` can mirror the same field onto `corpus.components[name]`.
--- @param source string -- the full source text (needed for backward lookups)
--- @param scan table -- SourceScan from duffle.scan_source
--- @return Component[]
local function project_components(source, scan) local function project_components(source, scan)
local out = {} local out = {}
for _, a in ipairs(scan.atoms) do for _, a in ipairs(scan.atoms) do
if a.kind == "comp_bare" or a.kind == "comp_proc" then if a.kind == "comp_bare" or a.kind == "comp_proc" then
local args = find_function_args_for(source, a.raw_name, a.ident_pos) local args = find_function_args_for(source, a.raw_name, a.ident_pos)
local comment = preceding_comment_block(source, a.ident_pos) -- Comment ownership: scan_source.lua stamps `declaration_comment` on the record by walking backward past any associated bare marker.
-- The pass reads `declaration_comment` directly.
local comment = a.declaration_comment or ""
out[#out + 1] = { out[#out + 1] = {
line = a.line, line = a.line,
name = a.name, name = a.name,
body = a.body, body = a.body,
body_off = a.body_off,
body_tokens = a.body_tokens, body_tokens = a.body_tokens,
args = args, args = args,
comment = comment, comment = comment,
kind = a.kind, -- "comp_bare" | "comp_proc"; provenance emitter reads this. kind = a.kind, -- "comp_bare" | "comp_proc"; provenance emitter reads this.
debug_skip = a.debug_skip == true,
} }
end end
end end
@@ -340,11 +268,11 @@ end
-- Word-count computation (memoized recursive lookup) -- Word-count computation (memoized recursive lookup)
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Strip the `mac_` prefix from a component-call ident so we can look it up against the components-by-name table. --- Strip the `mac_` prefix from a component-call ident so we can look it up against the components-by-name table.
-- Returns the ident unchanged if it doesn't start with the prefix --- Returns the ident unchanged if it doesn't start with the prefix
-- (so a non-component ident like `mask_upper` falls through to the wc-table branch). --- (so a non-component ident like `mask_upper` falls through to the wc-table branch).
-- @param ident string|nil --- @param ident string|nil
-- @return string|nil --- @return string|nil
local function strip_mac_prefix(ident) local function strip_mac_prefix(ident)
if not ident then return nil end if not ident then return nil end
if ident:sub(1, MAC_PREFIX_LEN) == MAC_PREFIX then if ident:sub(1, MAC_PREFIX_LEN) == MAC_PREFIX then
@@ -353,13 +281,13 @@ local function strip_mac_prefix(ident)
return ident return ident
end end
-- (internal) Recursive word-count lookup. `cache` is the memoization table shared across all components --- (internal) Recursive word-count lookup. `cache` is the memoization table shared across all components
-- in a single source's `count_all_components` pass; the in-progress -1 sentinel detects cycles (A -> B -> A). --- in a single source's `count_all_components` pass; the in-progress -1 sentinel detects cycles (A -> B -> A).
-- @param name string -- the component name (without `mac_`) --- @param name string -- the component name (without `mac_`)
-- @param comp_by_name table<string, Component> --- @param comp_by_name table<string, Component>
-- @param wc table<string, integer> --- @param wc table<string, integer>
-- @param cache table<string, integer> --- @param cache table<string, integer>
-- @return integer --- @return integer
local function word_count_rec(name, comp_by_name, wc, cache) local function word_count_rec(name, comp_by_name, wc, cache)
if cache[name] ~= nil then return cache[name] end if cache[name] ~= nil then return cache[name] end
cache[name] = -1 -- mark in-progress (cycle detection) cache[name] = -1 -- mark in-progress (cycle detection)
@@ -398,7 +326,7 @@ end
--- references hit memoized values instead of re-walking the body. --- references hit memoized values instead of re-walking the body.
--- Cycle detection (A -> B -> A) is preserved via the in-progress `-1` sentinel in `cache`. --- Cycle detection (A -> B -> A) is preserved via the in-progress `-1` sentinel in `cache`.
--- @param components Component[] --- @param components Component[]
--- @param wc table<string, integer> --- @param wc table<string, integer>
--- @return table<string, integer> -- map of component name (without `mac_`) -> word count --- @return table<string, integer> -- map of component name (without `mac_`) -> word count
local function count_all_components(components, wc) local function count_all_components(components, wc)
local comp_by_name = {} local comp_by_name = {}
@@ -471,13 +399,23 @@ end
--- Build the list of lines for one component --- Build the list of lines for one component
--- (signature comment, `#define mac_X(...)` line with backslash-continued tokens, then `WORD_COUNT(mac_X, N)` entry). --- (signature comment, `#define mac_X(...)` line with backslash-continued tokens, then `WORD_COUNT(mac_X, N)` entry).
--- @param c Component --- For skipped components, a `/* atom_dbg_skip */` marker comment is emitted immediately before the authored comment block.
--- The marker is a single line, the comment comes next, and the `#define` line follows. The `debug_skip` stamp is scanner-owned
--- (`a.debug_skip == true` on the declaration record); the components pass projects it directly.
--- @param c Component
--- @param components Component[] --- @param components Component[]
--- @param wc table<string, integer> --- @param wc table<string, integer>
--- @return string[] -- list of lines for this component --- @return string[] -- list of lines for this component
local function build_component_lines(c, counts) local function build_component_lines(c, counts)
local lines = {} local lines = {}
-- Marker comment: emitted once for every skipped component.
-- The marker is scanner-owned (declared by `atom_dbg_skip` immediately before the declaration in the source);
-- the components pass projects `c.debug_skip` and emits the marker as a generated comment.
if c.debug_skip then
lines[#lines + 1] = "/* atom_dbg_skip */"
end
if c.comment and c.comment ~= "" then if c.comment and c.comment ~= "" then
for _, line in ipairs(split_comment_lines(c.comment)) do for _, line in ipairs(split_comment_lines(c.comment)) do
lines[#lines + 1] = line lines[#lines + 1] = line
@@ -505,10 +443,10 @@ end
-- Per-source emit logic -- Per-source emit logic
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Build the boilerplate header lines (the `#ifdef INTELLISENSE_DIRECTIVES` block, --- Build the boilerplate header lines (the `#ifdef INTELLISENSE_DIRECTIVES` block,
-- the `// Auto-generated` comment, the `// Source:` line, and the self-contained `WORD_COUNT` macro definition). --- the `// Auto-generated` comment, the `// Source:` line, and the self-contained `WORD_COUNT` macro definition).
-- @param src SourceFile --- @param src SourceFile
-- @return string[] --- @return string[]
local function header_boilerplate(src) local function header_boilerplate(src)
return { return {
-- #pragma once wrapped in #ifdef INTELLISENSE_DIRECTIVES, matching the convention in lottes_tape.h. -- #pragma once wrapped in #ifdef INTELLISENSE_DIRECTIVES, matching the convention in lottes_tape.h.
@@ -530,13 +468,13 @@ local function header_boilerplate(src)
} }
end end
-- Compute the output path for one source's `.macs.h` file. --- Compute the output path for one source's `.macs.h` file.
-- The pre-rework convention uses the *directory* basename --- The pre-rework convention uses the *directory* basename (not the source file basename)
-- (not the source file basename) e.g. `code/duffle/lottes_tape.h` produces `code/duffle/gen/duffle.macs.h`. --- e.g. `code/duffle/lottes_tape.h` produces `code/duffle/gen/duffle.macs.h`.
-- This matches what the C codebase #includes. --- This matches what the C codebase #includes.
-- @param src SourceFile --- @param src SourceFile
-- @return string -- the output directory --- @return string -- the output directory
-- @return string -- the full output path --- @return string -- the full output path
local function compute_macs_h_path(src) local function compute_macs_h_path(src)
local out_dir = src.dir .. "/" .. GEN_SUBDIR local out_dir = src.dir .. "/" .. GEN_SUBDIR
local out_path = out_dir .. "/" .. duffle.basename_no_ext(src.dir) .. ".macs.h" local out_path = out_dir .. "/" .. duffle.basename_no_ext(src.dir) .. ".macs.h"
@@ -545,11 +483,10 @@ end
--- Emit a per-source `.macs.h` header with the `mac_X` macros + `WORD_COUNT` entries. --- Emit a per-source `.macs.h` header with the `mac_X` macros + `WORD_COUNT` entries.
--- Writes in BINARY mode so LF line endings are preserved (the git blob is LF; Windows text-mode would emit CRLF and break the byte-identical diff). --- Writes in BINARY mode so LF line endings are preserved (the git blob is LF; Windows text-mode would emit CRLF and break the byte-identical diff).
--- Honors `ctx.dry_run`: prints the intended path but does not write the file. --- @param ctx PassCtx
--- @param ctx PassCtx --- @param src SourceFile
--- @param src SourceFile
--- @param components Component[] --- @param components Component[]
--- @param counts table<string, integer> -- precomputed word counts (from count_all_components) --- @param counts table<string, integer> -- precomputed word counts (from count_all_components)
--- @return string|nil -- path to the written file (nil if no components) --- @return string|nil -- path to the written file (nil if no components)
local function emit_component_macros_h(ctx, src, components, counts) local function emit_component_macros_h(ctx, src, components, counts)
if #components == 0 then return nil end if #components == 0 then return nil end
@@ -563,11 +500,6 @@ local function emit_component_macros_h(ctx, src, components, counts)
end end
local content = table.concat(lines, "\n") .. "\n" local content = table.concat(lines, "\n") .. "\n"
if ctx.dry_run then
print(string.format(" -> %s (dry-run)", out_path))
return out_path
end
duffle.ensure_dir(out_dir) duffle.ensure_dir(out_dir)
duffle.write_file_lf(out_path, content) duffle.write_file_lf(out_path, content)
print(string.format(" -> %s", out_path)) print(string.format(" -> %s", out_path))
@@ -578,41 +510,90 @@ end
-- Pass entry -- Pass entry
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- (internal) Extend `ctx.shared.word_counts` with this source's component macros so offsets sees them without re-reading the file. --- (internal) Extend `corpus.word_counts` with this source's component macros so offsets sees them without re-reading the file.
-- @param ctx PassCtx --- First declaration wins: a later caller's count is dropped (the existing entry from the first source is preserved).
-- @param components Component[] --- @param corpus table -- the corpus
-- @param counts table<string, integer> -- precomputed word counts (from count_all_components) --- @param components Component[]
local function update_shared_word_counts(ctx, components, counts) --- @param counts table<string, integer> -- precomputed word counts (from count_all_components)
local wc = ctx.shared.word_counts local function update_canonical_word_counts(corpus, components, counts)
local wc = corpus.word_counts
for _, c in ipairs(components) do for _, c in ipairs(components) do
wc["mac_" .. c.name] = counts[c.name] local key = "mac_" .. c.name
if wc[key] == nil then
wc[key] = counts[c.name]
end
end end
end end
--- @class ComponentDef --- @class ComponentDef
--- @field name string -- bare name (without ac_/mac_ prefix) --- @field name string -- bare name (without ac_/mac_ prefix)
--- @field line integer -- definition source line (line of `MipsAtomComp_(ac_X)` / `MipsAtomComp_Proc_(ac_X, ...)`) --- @field line integer -- definition source line (line of `MipsAtomComp_(ac_X)` / `MipsAtomComp_Proc_(ac_X, ...)`)
--- @field path string -- absolute source path of the definition --- @field path string -- absolute source path of the definition
--- @field kind string -- "comp_bare" | "comp_proc" --- @field kind string -- "comp_bare" | "comp_proc"
--- @field debug_skip boolean -- mirror of the scanner-owned `a.debug_skip`; consumers read this directly
--- (internal) Extend `ctx.shared.components` with this source's components-by-name map so downstream passes --- (internal) Populate `corpus.components` with this source's components-by-name map.
--- (atoms_source_map, dwarf_injection) can resolve `mac_X(...)` invocations back to their component definition file:line. --- First declaration wins; later declarations of the same bare name are dropped and recorded as a collision via `corpus.collisions` (kind = "component").
--- provenance emission uses this to attribute each emitted `.word` to either a component macro or the enclosing atom body. --- The pass does NOT write to `ctx.shared.components` (ownership follows the canonical contract).
-- @param ctx PassCtx --- The `debug_skip` field mirrors the scanner-owned declaration record (`c.debug_skip`).
-- @param src SourceFile --- No parallel skip map is built here; consumers that need the per-component skip state read `corpus.components[name].debug_skip` directly.
-- @param components Component[] --- @param corpus table -- the corpus
local function update_shared_components(ctx, src, components) --- @param src SourceFile
ctx.shared.components = ctx.shared.components or {} --- @param components Component[]
local function update_canonical_components(corpus, src, components)
local rel_path = src.path:gsub("\\", "/") local rel_path = src.path:gsub("\\", "/")
for _, c in ipairs(components) do for _, c in ipairs(components) do
-- Keyed by bare name (e.g. `yield`, `load_tri_indices`). -- Keyed by bare name (e.g. `yield`, `load_tri_indices`).
-- The atoms_source_map pass strips the `mac_` prefix from the call site identifier before lookup. -- The atoms_source_map pass looks up components by bare name from the corpus;
ctx.shared.components[c.name] = { -- `mac_` prefix lives at the call-site identifier and is stripped before lookup.
name = c.name, if corpus.components[c.name] == nil then
line = c.line, corpus.components[c.name] = {
path = rel_path, name = c.name,
kind = c.kind or "comp_bare", line = c.line,
} path = rel_path,
kind = c.kind or "comp_bare",
debug_skip = c.debug_skip == true,
}
else
-- A second declaration of the same bare name: record a typed collision so static-analysis + the report can surface it.
-- Identical-shape declarations (same path + line) reuse the first-wins entry without a collision record.
local existing = corpus.components[c.name]
if existing.path ~= rel_path or existing.line ~= c.line then
local kind = c.kind or "comp_bare"
local first_kind = existing.kind or "comp_bare"
corpus.collisions[#corpus.collisions + 1] = {
kind = "component",
name = c.name,
first_site = { path = existing.path, line = existing.line },
conflicting_site = { path = rel_path, line = c.line },
first_shape = "kind=" .. first_kind,
conflicting_shape = "kind=" .. kind,
}
end
end
end
end
--- (internal) Populate `corpus.component_body_index` with this source's body index entries.
--- First declaration wins; later declarations are dropped (no separate collision record: the components collision is already surfaced by `update_canonical_components`).
--- The pass writes to `corpus.component_body_index` only (the corpus owns this projection).
--- @param corpus table -- the corpus
--- @param src SourceFile
--- @param components Component[]
--- @param scan table -- the SourceScan payload (for line_of)
local function update_canonical_component_body_index(corpus, src, components, scan)
local line_of = scan and scan.line_of
for _, c in ipairs(components) do
if corpus.component_body_index[c.name] == nil then
corpus.component_body_index[c.name] = {
body_tokens = c.body_tokens,
body_off = c.body_off,
line_of = line_of,
source = src.path,
declaration = c.line,
kind = c.kind,
}
end
end end
end end
@@ -623,24 +604,41 @@ function M.run(ctx)
local errors = {} local errors = {}
local warnings = {} local warnings = {}
-- Initialize shared component map. -- Corpus ownership gate.
-- The atoms_source_map and dwarf_injection passes consume `ctx.shared.components` to resolve `mac_X(...)` local corpus = ctx.shared and ctx.shared.corpus
-- invocations back to the component's definition file:line. if type(corpus) ~= "table" then
ctx.shared.components = ctx.shared.components or {} error("components.run requires ctx.shared.corpus.", 0)
end
if type(corpus.source_order) ~= "table" then
error("components.run requires ctx.shared.corpus.source_order.", 0)
end
if type(corpus.word_counts) ~= "table" then
error("components.run requires ctx.shared.corpus.word_counts; "
.. "word_count_eval.run must run before components.run "
.. "(see PASSES deps).", 0)
end
for _, src in ipairs(ctx.sources) do -- Projection ownership:
-- * `corpus.word_counts["mac_"..name]` — current component count
-- * `corpus.components[name]` — bare-name component definition
-- * `corpus.component_body_index[name]` — body / line_of / source index
-- The pass writes to the corpus only; consumers read from the corpus directly.
for _, src in ipairs(corpus.source_order) do
-- project_components reads from src.scan + does backward lookups on src.text -- project_components reads from src.scan + does backward lookups on src.text
local components = project_components(src.text, src.scan) local components = project_components(src.text, src.scan)
if #components > 0 then if #components > 0 then
-- Compute word counts for ALL components once (was: rebuilt per call inside the helpers). -- Compute all component word counts once per source.
local counts = count_all_components(components, ctx.shared.word_counts) -- Use `corpus.word_counts` so the recursive lookup sees both authored-metadata entries
-- (loaded by word_count_eval.run) AND same-source component entries (populated earlier in this loop by `update_canonical_word_counts`).
local counts = count_all_components(components, corpus.word_counts)
local macs_path = emit_component_macros_h(ctx, src, components, counts) local macs_path = emit_component_macros_h(ctx, src, components, counts)
if macs_path then if macs_path then
outputs[#outputs + 1] = { macs_h = macs_path } outputs[#outputs + 1] = { macs_h = macs_path }
update_shared_word_counts(ctx, components, counts) -- Populate the projections AFTER disk emission (so the byte-identical `.macs.h` contract is preserved before any current-count mutation).
-- share component definitions with downstream passes. update_canonical_word_counts(corpus, components, counts)
-- `mac_X(...)` invocations in atom bodies resolve back to (path, line) via this map. update_canonical_components(corpus, src, components)
update_shared_components(ctx, src, components) update_canonical_component_body_index(corpus, src, components, src.scan)
end end
end end
end end
File diff suppressed because it is too large Load Diff
+239
View File
@@ -0,0 +1,239 @@
--- passes/emission_model.lua: Per-atom emission projection.
---
--- The `emission-model` pass owns `atom.paths`, the canonical per-atom mutable surface for atoms and raw atoms with bodies in `ctx.shared.corpus.source_order`.
--- For each atom, the pass invokes `duffle.project_emission(body_text, component_index, word_counts, components)`.
--- It stores the ordered `items` stream plus the dense `word_events` / `markers` / `invocations` views on `atom.paths`.
---
--- Public boundary:
--- * `M.run(ctx)` is the only entry point.
--- * The pass returns `{outputs = {}, errors = ..., warnings = ...}`.
--- Pass kind = `validation` → `PASS_KIND_STOP_ON_ERROR.validation` preserves the existing build-stopping policy.
---
--- Source-order discipline:
--- * `corpus.source_order` sets the source-record order.
--- * Within each source, the pass visits `src.scan.atoms` and `src.scan.raw_atoms` in declaration order.
---
--- Per-atom projection fields on `atom.paths`:
--- `tokens`, `line_in_body`, `items`, `word_events`, `markers`, `invocations`, `errors`, `warnings`.
--- The construction walk appends `items` and derives each dense view from that ordered stream.
---
--- Component expansion and construction validation:
--- * known `mac_X(...)` calls recursively expand component bodies;
--- * invocation records retain monotonic IDs, parent IDs, immediate call text, and the immutable outermost root call text;
--- * invocation construction stamps `debug_skip` from `corpus.components[name].debug_skip` at the construction site (no second pass, no source parse, no parallel lookup);
--- * component cycles close balanced invocation boundaries and emit a `cycle` construction error at the recursive edge;
--- * declared-vs-measured component word counts emit `count_mismatch` construction errors; opaque uncounted macros emit warnings.
---
--- `passes.scan_source` strips its private `_code_macros` / `_code_macro_bodies` tables before this pass runs.
local M = {}
-- ─────────────────────────────────────────────────────────────────────────
-- Bootstrap: load `duffle_paths.lua` via debug.getinfo so the module works standalone (run as `luajit passes/emission_model.lua`) and when require'd from the orchestrator.
-- ─────────────────────────────────────────────────────────────────────────
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
-- ─────────────────────────────────────────────────────────────────────────
-- Helpers
-- ─────────────────────────────────────────────────────────────────────────
-- Convert the recursive walk's body-relative line numbers into physical source lines once.
-- The walker builds `line_of` from `body_text` and stamps body-relative line numbers (1..N) into `item.line` and `invocation.call_line`.
-- This function converts those values to physical source lines at the close site with the forwarded source `line_of` closure.
--
-- `call_line` discipline:
-- * ROOT invocations (`inv.parent_id == 0`) receive body-relative `call_line` values directly from `M.LineIndex(body_text)` in the walker.
-- The source `line_of` closure supplies physical lines at the close site, so this function converts each root value exactly once.
-- * INNER invocations (`inv.parent_id ~= 0`) receive physical `call_line` values directly from the COMPONENT's `line_of` in the walker.
-- Recursive descent forwards that closure through `corpus.component_body_index[name].line_of`; those values arrive physical and remain unchanged.
--
-- After this function, every `inv.call_line` is physical. DWARF and provenance output read it directly.
-- The word-event loop forwards the already-physical `outer_inv.call_line` into `we.call_line` for words inside an invocation.
local function stamp_root_provenance(projection, atom_record, src, corpus)
local root_line_of = src.scan and src.scan.line_of
assert(type(root_line_of) == "function"
, "emission_model: src.scan.line_of is required (canonical LineIndex closure over the source text) to stamp physical provenance")
assert(type(atom_record.body_off) == "number"
, "emission_model: atom_record.body_off (byte offset of the body's first byte in source) is required to derive `root_body_line`. The scanner must populate body_off for every atom record.")
-- `root_body_line` is the physical source line of the ATOM HEADER byte containing the opening `{`; that byte is one byte BEFORE `atom_record.body_off`.
-- The walker assigns line 2 to the body's first content line because line 1 is the trailing `\n` after `{`. Body-text line k therefore maps to `root_body_line + (k - 1)`.
-- `body_off - 1` points at the opening `{`, whose line index identifies the header line. `body_off` points after `{` and would shift every word row forward by one line.
local root_body_line = root_line_of(atom_record.body_off - 1)
or atom_record.line or 0
local component_index = corpus.component_body_index or {}
local word_items = {}
for _, item in ipairs(projection.items) do
if item.kind == "word" then word_items[#word_items + 1] = item end
end
-- Resolve one word's physical body line, where the byte containing that word appears in source.
-- * Component expansions carry `invocation_ids`; the component's full-file `line_of` leaves `item.line` physical.
-- * Raw tokens in the root atom body carry an empty `invocation_ids` list and a body-relative `item.line`; convert them here.
local function body_line_for(event, item)
local ids = event.invocation_ids or {}
-- The innermost open invocation identifies which line index the walker used.
-- A component `line_of` makes `item.line` physical; the atom's `body_text` line index makes it body-relative.
if ids and #ids > 0 then
local inner_id = ids[#ids]
local inner_inv = inner_id and projection.invocations[inner_id]
if inner_inv then
local component = component_index[inner_inv.component_name]
if component and component.line_of then
-- Walker used `comp.line_of`, which is the source's physical LineIndex. item.line is already physical.
return item.line or 0
end
end
end
-- RAW root-body word: item.line is body-text's 1-based line number (the first content line is line 2 because line 1 is the trailing `\n` after `{`).
-- Convert body-text-relative → physical using `root_body_line + (item.line - 1)`.
return (root_body_line or 0) + (item.line or 1) - 1
end
-- Stamp the root source path onto invocation records whose `call_path` the walker left empty.
-- The walker passes `body_entry.source` to `emit_invoke_begin`; `M.project_emission` creates the root `body_entry` with source `""`, leaving its `call_path` empty.
-- This stamp gives every invocation a physical `call_path` matching `passes/atoms_source_map.lua`'s in-memory provenance projection.
local root_path = src.path or ""
for _, inv in ipairs(projection.invocations) do
if inv.call_path == nil or inv.call_path == "" then
inv.call_path = root_path
end
end
-- Normalize `inv.call_line` to a physical source line.
-- * ROOT invocations (`parent_id == 0`) carry body-relative `call_line` values from `M.LineIndex(body_text)`; convert them once with `root_body_line`.
-- * INNER invocations (`parent_id ~= 0`) carry physical `call_line` values from the component's `line_of`; retain them unchanged.
for _, inv in ipairs(projection.invocations) do
if inv.parent_id == 0 then
inv.call_line = (root_body_line or 0) + (inv.call_line or 1) - 1
end
end
-- Build `body_lines` for each invocation.
-- `atoms_source_map` and `dwarf_injection` read `inv.body_lines[k]` directly from the invocation record created here.
-- Component words already carry physical `item.line` values from the walker's COMPONENT line index, so `body_line_for` returns them unchanged.
for _, inv in ipairs(projection.invocations) do
local sw = inv.start_word
local ew = inv.end_word
local bls = {}
for i = sw, ew do
local it = projection.items and projection.items[i]
if it and it.kind == "word" then
local fake_event = { invocation_ids = { inv.id } }
bls[#bls + 1] = body_line_for(fake_event, it) or 0
end
end
inv.body_lines = bls
end
-- Resolve each `word_event`'s physical `body_line` and `call_line`.
-- For words inside an invocation, `we.call_line` identifies the OUTER atom source line containing the `mac_X(...)` token that triggered expansion.
-- The root-invocation conversion above makes every `inv.call_line` physical; forward it directly and use each raw word's `body_line` as the fallback.
for index, we in ipairs(projection.word_events) do
local item = word_items[index] or {}
local body_line = body_line_for(we, item)
item.line = body_line
we.body_line = body_line
local call_line = body_line
local outer_id = we.outermost_invocation_id or 0
local outer_inv = projection.invocations[outer_id]
if outer_inv then
-- `outer_inv.call_line` is physical after the conversion loop above, so use it directly.
call_line = outer_inv.call_line
end
we.call_line = call_line
if we.def_path == nil or we.def_path == "" then we.def_path = src.path or "" end
if we.def_line == nil or we.def_line == 0 then we.def_line = atom_record.line or 0 end
if we.call_path == nil or we.call_path == "" then we.call_path = src.path or "" end
end
end
-- Project one atom record into `atom.paths`.
-- Mutates the atom record in-place and returns the projection (for pass-level error/warning accumulation).
local function project_atom(atom_record, src, corpus)
local body = atom_record.body or ""
local wc = corpus.word_counts or {}
local cbi = corpus.component_body_index or {}
-- That construction site stamps `invocation.debug_skip` while appending each record to `proj.invocations`.
local proj = duffle.project_emission(body, cbi, wc, corpus.components)
local paths = {
tokens = atom_record.body_tokens or {},
line_in_body = duffle.build_body_line_index(body),
items = proj.items,
word_events = proj.word_events,
markers = proj.markers,
invocations = proj.invocations,
errors = proj.errors,
warnings = proj.warnings,
}
stamp_root_provenance(proj, atom_record, src, corpus)
atom_record.paths = paths
return proj
end
-- ─────────────────────────────────────────────────────────────────────────
-- Run the emission-model pass.
-- ─────────────────────────────────────────────────────────────────────────
--- @param ctx PassCtx -- { shared = { corpus = ... }, out_root, ... }
--- @return PassResult
function M.run(ctx)
local outputs = {}
local errors = {}
local warnings = {}
local corpus = ctx and ctx.shared and ctx.shared.corpus
if type(corpus) ~= "table" then error("emission_model: ctx.shared.corpus is required (canonical projection)", 0) end
if type(corpus.source_order) ~= "table" then error("emission_model: ctx.shared.corpus.source_order is required", 0) end
-- Project once, collect errors + warnings for one atom.
-- Kind must be one of: atom | raw_atom | comp_bare | comp_proc.
local function process_atom(atom, src)
if not (atom and atom.body) then return end
local kind = atom.kind
if kind ~= "atom" and kind ~= "raw_atom" and kind ~= "comp_bare" and kind ~= "comp_proc" then
return
end
local proj = project_atom(atom, src, corpus)
for _, e in ipairs(proj.errors) do
-- Preserve `kind` (cycle / count_mismatch / unbalanced) so readers dispatch on the diagnostic class and leave the message string as display text.
errors[#errors + 1] = {
kind = e.kind,
line = e.line,
msg = e.msg,
source = e.source or src.path,
}
end
for _, w in ipairs(proj.warnings) do
warnings[#warnings + 1] = {
kind = w.kind,
line = w.line,
msg = w.msg,
}
end
end
-- Walk `corpus.source_order`; within each source, visit atoms followed by raw_atoms.
-- Recognized kinds (atom | raw_atom | comp_bare | comp_proc) each receive the atom.paths projection via duffle.project_emission.
-- Components are macros inlined into atom bodies; focused tests and isolated component analyses consume atom.paths directly.
for _, src in ipairs(corpus.source_order) do
local scan = src.scan or {}
for _, atom in ipairs(scan.atoms or {}) do
process_atom(atom, src)
end
for _, atom in ipairs(scan.raw_atoms or {}) do
process_atom(atom, src)
end
end
return {
outputs = outputs,
errors = errors,
warnings = warnings,
}
end
return M
+88 -197
View File
@@ -21,18 +21,12 @@
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd). -- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module. -- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
local word_count_eval = require("word_count_eval")
local count_token_words = word_count_eval.count_token_words
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Constants -- Constants
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Marker-call identifiers inside atom bodies.
local LABEL_MARKER = "atom_label"
local OFFSET_MARKER = "atom_offset"
-- Offset macro/enum naming prefixes (the emitted header uses these). -- Offset macro/enum naming prefixes (the emitted header uses these).
local OFFSET_MACRO_PREFIX = "_atom_offset_" local OFFSET_MACRO_PREFIX = "_atom_offset_"
local OFFSET_ENUM_PREFIX = "atom_offset_" local OFFSET_ENUM_PREFIX = "atom_offset_"
@@ -52,16 +46,10 @@ local OFFSET_MACRO_COL = 44
--- @field scan table -- pre-scanned SourceScan payload (from duffle.scan_source) --- @field scan table -- pre-scanned SourceScan payload (from duffle.scan_source)
--- @class PassCtx --- @class PassCtx
--- @field sources SourceFile[] -- all source files in the build
--- @field metadata_path string -- path to word_count.metadata.h
--- @field shared table -- cross-pass shared state --- @field shared table -- cross-pass shared state
--- @field shared.word_counts table -- macro name -> word count --- @field shared.corpus table -- canonical corpus projection
--- @field shared.word_counts table
--- @field out_root string -- output root (e.g. "build/gen") --- @field out_root string -- output root (e.g. "build/gen")
--- @field project_root string -- project root (e.g. "code/")
--- @field upstream table<string, table> -- per-pass upstream outputs
--- @field flags table -- CLI flags
--- @field dry_run boolean -- if true, compute but don't write
--- @field verbose boolean -- log diagnostic info
--- @class PassResult --- @class PassResult
--- @field outputs table[] -- {kind=, path=} entries describing emit files --- @field outputs table[] -- {kind=, path=} entries describing emit files
@@ -69,10 +57,10 @@ local OFFSET_MACRO_COL = 44
--- @field warnings table[] -- {line=, msg=} entries; build-succeeds --- @field warnings table[] -- {line=, msg=} entries; build-succeeds
--- @class BranchOffset --- @class BranchOffset
--- @field tag string -- the marker tag (e.g. "F" in `atom_offset(F, T)`) --- @field tag string -- the marker tag (e.g. "F" in `atom_offset(F, T)`)
--- @field target string -- the target label name (e.g. "T" in `atom_offset(F, T)`) --- @field target string -- the target label name (e.g. "T" in `atom_offset(F, T)`)
--- @field pos integer -- the branch's word position within the atom body --- @field branch_word integer -- branch word position within the atom body
--- @field offset integer -- computed `target_word - branch_word - 1` --- @field offset integer -- computed `target_word - branch_word - 1`
--- @class AtomData --- @class AtomData
--- @field name string -- atom name --- @field name string -- atom name
@@ -80,169 +68,85 @@ local OFFSET_MACRO_COL = 44
--- @field offsets BranchOffset[] -- per-branch offset list --- @field offsets BranchOffset[] -- per-branch offset list
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Per-token marker-call helpers (atom_label / atom_offset inside bodies) -- Canonical marker projection
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Extract comma-separated identifier args from a parenthesized group after a function-like macro call. -- MARKER_PROJECTORS is the marker-kind data table.
-- Returns (args, after_paren) where `after_paren` is the position just past the closing `)`, or nil if `token` did not start with `(`. -- The emission-model pass already records marker word positions;
-- @param token string -- this pass only projects those records into the label/branch lookup shape needed by offset computation.
-- @param after_ident integer local MARKER_PROJECTORS = {
-- @return string[], integer|nil label = function(state, marker)
local function extract_ident_args(token, after_ident) state.labels[marker.name] = marker.word_index
local arg_start = duffle.skip_ws_and_cmt(token, after_ident) end,
if token:sub(arg_start, arg_start) ~= "(" then return {}, nil end offset = function(state, marker)
local inner, after_paren = duffle.read_parens(token, arg_start) state.branches[#state.branches + 1] = {
-- scan: <marker>(<args>) tag = marker.name,
target = marker.target,
local args = {} branch_word = marker.word_index,
local pos = 1 }
local inner_len = #inner end,
while pos <= inner_len do
pos = duffle.skip_ws_and_cmt(inner, pos)
if pos > inner_len then break end
local ident, after = duffle.read_ident(inner, pos)
if ident and ident ~= "" then
table.insert(args, ident)
pos = after
else
pos = pos + 1
end
pos = duffle.skip_ws_and_cmt(inner, pos)
if pos <= inner_len and inner:sub(pos, pos) == "," then pos = pos + 1 end
end
return args, after_paren
end
-- (internal) Record a `atom_label(name)` marker — `at_pos` is the branch-free word position within the atom body.
-- @param labels table<string, integer>
-- @param args string[]
-- @param at_pos integer
local function record_label_marker(labels, args, at_pos)
if #args >= 1 then labels[args[1]] = at_pos end
end
-- (internal) Record a `atom_offset(tag, target)` marker.
-- @param branches table[] -- list of {pos=, target=, tag=}
-- @param args string[]
-- @param at_pos integer
local function record_offset_marker(branches, args, at_pos)
if #args >= 2 then
table.insert(branches, { pos = at_pos, target = args[2], tag = args[1] })
end
end
-- MARKER_TO_HANDLER — data-driven marker dispatch (the plex pattern).
-- Maps the marker ident to its recorder function. Each handler takes (out_table, args, at_pos).
-- Adding a new marker type = 1 row + 1 recorder function.
local MARKER_TO_HANDLER = {
[LABEL_MARKER] = record_label_marker,
[OFFSET_MARKER] = record_offset_marker,
} }
--- Scan a single token for atom_label/atom_offset markers, walking through balanced groups transparently (so nested calls are found). --- Project canonical marker records into the two lookup tables used by the offset renderer.
--- @param token string --- No source text, body text, or body token is inspected.
--- @param at_pos integer -- the branch-free word position of this token in the body --- @param markers table[] -- atom.paths.markers
--- @param labels table<string, integer> --- @return table<string, integer>, table[]
--- @param branches table[] local function project_markers(markers)
local function scan_for_atom_markers(token, at_pos, labels, branches) local state = { labels = {}, branches = {} }
local pos = 1 for _, marker in ipairs(markers or {}) do
local tok_len = #token local project = MARKER_PROJECTORS[marker.kind]
while pos <= tok_len do if project then project(state, marker) end
pos = duffle.skip_ws_and_cmt(token, pos)
if pos > tok_len then break end
local ch = token:sub(pos, pos)
if duffle.is_alpha(ch) then
local ident, after = duffle.read_ident(token, pos)
local handler = MARKER_TO_HANDLER[ident]
if handler then
local args, after_paren = extract_ident_args(token, after)
-- Marker found — dispatch to its recorder. markers share labels and branches as
-- out-tables; the recorder picks which one(s) to write to based on its semantics.
-- (record_label_marker writes to labels; record_offset_marker writes to branches.)
handler(ident == LABEL_MARKER and labels or branches, args, at_pos)
pos = after_paren or after
else
pos = after
end
else
local nx = duffle.skip_str_or_cmt(token, pos)
pos = (nx > pos) and nx or (pos + 1)
end
end end
end return state.labels, state.branches
--- Scan an atom body for labels + branches, count total words.
--- Returns (labels, branches, total_words).
--- @param body string
--- @param word_counts table
--- @return table<string, integer>, table[], integer
-- scan_atom_body: walk pre-tokenized body for atom_label/atom_offset markers + word counts.
-- Uses `atom.body_tokens` from the SourceScan payload (pre-tokenized by scan-source pass).
-- @param body_tokens table[] -- {{tok=string, rel=integer}, ...} from duffle.tokenize_body
-- @param word_counts table
-- @return table, table, integer -- labels, branches, total_words
local function scan_atom_body(body_tokens, word_counts)
local pos = 0
local labels = {}
local branches = {}
for _, t in ipairs(body_tokens) do
local tok = t.tok
if duffle.is_marker_token(tok) then
-- Marker call: record at the current pos, do NOT advance pos.
scan_for_atom_markers(tok, pos, labels, branches)
pos = pos + duffle.count_marker_rest(tok, word_counts, count_token_words)
else
local words = count_token_words(tok, word_counts)
scan_for_atom_markers(tok, pos, labels, branches)
pos = pos + words
end
end
return labels, branches, pos
end end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Offset computation + header generation -- Offset computation + header generation
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Compute branch offsets as `target_word - branch_word - 1` (the standard MIPS branch-immediate encoding). --- Compute branch offsets as `target_word - branch_word - 1` (the standard MIPS branch-immediate encoding).
-- @param labels table<string, integer> --- @param labels table<string, integer>
-- @param branches table[] --- @param branches table[]
-- @return BranchOffset[] --- @return BranchOffset[]
local function compute_offsets(labels, branches) local function compute_offsets(labels, branches)
local results = {} local results = {}
for _, br in ipairs(branches) do for _, br in ipairs(branches) do
local target = labels[br.target] local target = labels[br.target]
if not target then if not target then
error("Branch target '" .. br.target .. "' has no atom_label (at word " .. br.pos .. ")") error("Branch target '" .. br.target .. "' has no atom_label (at word " .. br.branch_word .. ")")
end end
results[#results + 1] = { target = br.target, tag = br.tag, offset = target - br.pos - 1 } results[#results + 1] = {
target = br.target,
tag = br.tag,
branch_word = br.branch_word,
offset = target - br.branch_word - 1,
}
end end
return results return results
end end
-- Right-pad `s` with spaces to width `w`. If `s` is already `w` or wider, no padding is added. --- Right-pad `s` with spaces to width `w`. If `s` is already `w` or wider, no padding is added.
-- @param s string --- @param s string
-- @param w integer --- @param w integer
-- @return string --- @return string
local function pad_right(s, w) local function pad_right(s, w)
return s .. string.rep(" ", math.max(0, w - #s)) return s .. string.rep(" ", math.max(0, w - #s))
end end
-- (internal) Build a constant-table entry `{macro_name, enum_name, value}` from a BranchOffset. --- (internal) Build a constant-table entry `{macro_name, enum_name, value}` from a BranchOffset.
-- @param r BranchOffset --- @param bo BranchOffset
-- @return table --- @return table
local function make_offset_const(r) local function make_offset_const(bo)
return { return {
macro_name = OFFSET_MACRO_PREFIX .. r.tag .. "_" .. r.target, macro_name = OFFSET_MACRO_PREFIX .. bo.tag .. "_" .. bo.target,
enum_name = OFFSET_ENUM_PREFIX .. r.tag .. "_" .. r.target, enum_name = OFFSET_ENUM_PREFIX .. bo.tag .. "_" .. bo.target,
value = r.offset, value = bo.offset,
} }
end end
-- (internal) Emit one atom's offset constants + enum into the lines buffer. --- (internal) Emit one atom's offset constants + enum into the lines buffer.
-- @param add fun(s: string) --- @param add fun(s: string)
-- @param atom AtomData --- @param atom AtomData
local function emit_atom_offsets(add, atom) local function emit_atom_offsets(add, atom)
if #atom.offsets == 0 then return end if #atom.offsets == 0 then return end
add("// --- atom: " .. atom.name .. " (" .. atom.total_words .. " words) ---") add("// --- atom: " .. atom.name .. " (" .. atom.total_words .. " words) ---")
@@ -263,10 +167,10 @@ local function emit_atom_offsets(add, atom)
add("") add("")
end end
-- Generate the per-source .offsets.h header. --- Generate the per-source .offsets.h header.
-- @param source_path string --- @param source_path string
-- @param atoms_data AtomData[] --- @param atoms_data AtomData[]
-- @return string --- @return string
local function generate_header(source_path, atoms_data) local function generate_header(source_path, atoms_data)
local basename = duffle.basename_no_ext(source_path) local basename = duffle.basename_no_ext(source_path)
@@ -288,59 +192,41 @@ local function generate_header(source_path, atoms_data)
return table.concat(lines, "\n") .. "\n" return table.concat(lines, "\n") .. "\n"
end end
-- ════════════════════════════════════════════════════════════════════════════
-- M — module exports
-- ════════════════════════════════════════════════════════════════════════════
local M = {} local M = {}
-- Project the pre-scanned SourceScan entries into the {name, body, body_tokens} shape this pass needs. --- (internal) Process one source: render offsets from canonical atom paths.
-- MipsAtom_ entries have kind="atom"; MipsCode code_<name> entries have kind="raw_atom". --- Returns the offsets_h path if a header was written, or nil.
-- `body_tokens` is set by scan-source on every `scan.atoms[i]` / `scan.raw_atoms[i]`; we carry it forward --- @param ctx PassCtx
-- so `scan_atom_body` reads from the precomputed table directly (no per-atom tokenize_body fallback). --- @param src SourceFile
-- @param scan table -- SourceScan from duffle.scan_source --- @return string|nil -- the offsets_h path
-- @return table[] -- list of {name=, body=, body_tokens=}
local function project_atoms(scan)
local out = {}
for _, a in ipairs(scan.atoms) do
out[#out + 1] = { name = a.raw_name, body = a.body, body_tokens = a.body_tokens }
end
for _, a in ipairs(scan.raw_atoms) do
out[#out + 1] = { name = a.name, body = a.body, body_tokens = a.body_tokens }
end
return out
end
-- (internal) Process one source: project atoms from scan, scan bodies, write header.
-- Returns the offsets_h path if a header was written, or nil.
-- @param ctx PassCtx
-- @param src SourceFile
-- @return string|nil -- the offsets_h path
local function process_source(ctx, src) local function process_source(ctx, src)
local atoms = project_atoms(src.scan)
if #atoms == 0 then return nil end
local atoms_data = {} local atoms_data = {}
for _, atom in ipairs(atoms) do local scan = src.scan or {}
local labels, branches, total = scan_atom_body(atom.body_tokens, ctx.shared.word_counts)
local function append_atom(atom)
local paths = atom and atom.paths
if not paths then return end
local labels, branches = project_markers(paths.markers)
atoms_data[#atoms_data + 1] = { atoms_data[#atoms_data + 1] = {
name = atom.name, name = atom.raw_name or atom.name,
total_words = total, total_words = #(paths.word_events or {}),
offsets = compute_offsets(labels, branches), offsets = compute_offsets(labels, branches),
} }
end end
for _, atom in ipairs(scan.atoms or {}) do append_atom(atom) end
for _, atom in ipairs(scan.raw_atoms or {}) do append_atom(atom) end
if #atoms_data == 0 then return nil end
local out_path = src.dir .. "/gen/" .. duffle.basename_no_ext(src.dir) .. ".offsets.h" local out_path = src.dir .. "/gen/" .. duffle.basename_no_ext(src.dir) .. ".offsets.h"
if not ctx.dry_run then duffle.ensure_dir(duffle.dirname(out_path))
duffle.ensure_dir(duffle.dirname(out_path)) duffle.write_file(out_path, generate_header(src.path:gsub("/", "\\"), atoms_data))
duffle.write_file(out_path, generate_header(src.path, atoms_data))
end
return out_path return out_path
end end
--- Run the offsets pass. --- Run the offsets pass.
--- For each source, emits a per-module `<dir_basename>.offsets.h` containing `#define _atom_offset_F_T = N` constants --- For each canonical source, emits a per-module `<dir_basename>.offsets.h`
--- for every `atom_offset(F, T)` reference in the source's atoms. --- containing constants for every marker recorded in atom.paths.
--- @param ctx PassCtx --- @param ctx PassCtx
--- @return PassResult --- @return PassResult
function M.run(ctx) function M.run(ctx)
@@ -348,7 +234,12 @@ function M.run(ctx)
local errors = {} local errors = {}
local warnings = {} local warnings = {}
for _, src in ipairs(ctx.sources) do local corpus = ctx.shared and ctx.shared.corpus
if type(corpus) ~= "table" or type(corpus.source_order) ~= "table" then
error("offsets.run requires ctx.shared.corpus.source_order (canonical corpus).", 0)
end
for _, src in ipairs(corpus.source_order) do
local out_path = process_source(ctx, src) local out_path = process_source(ctx, src)
if out_path then if out_path then
outputs[#outputs + 1] = { offsets_h = out_path } outputs[#outputs + 1] = { offsets_h = out_path }
+413 -289
View File
@@ -5,11 +5,8 @@
--- - `build/gen/<dir_basename>.annotations.txt` — one per source-directory containing atoms; aggregates across all sources in the directory. --- - `build/gen/<dir_basename>.annotations.txt` — one per source-directory containing atoms; aggregates across all sources in the directory.
--- - `build/gen/annotation_validation.txt` — the project summary. --- - `build/gen/annotation_validation.txt` — the project summary.
--- ---
--- The annotation pass stashes per-MODULE summary entries in `ctx.flags._annot_results` (set by `passes/annotation.lua`). --- The annotation pass emits `errors.h` files per module and the canonical `corpus.sources_by_dir` projection groups sources by directory.
--- This pass re-validates each source via `annotation.validate()` to get the detailed per-source results needed for the report. --- This pass iterates the canonical dir projection directly and re-validates each source via `annotation.validate()` to get the detailed per-source results.
---
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
--- Lua 5.3 compatible.
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Module-scope requires + package.path setup -- Module-scope requires + package.path setup
@@ -18,12 +15,21 @@
-- Resolve `arg[0]` to an absolute-ish script directory so that `require("duffle")` resolves against `scripts/` regardless of CWD. -- Resolve `arg[0]` to an absolute-ish script directory so that `require("duffle")` resolves against `scripts/` regardless of CWD.
-- Bootstrap: see `ps1_meta.lua` for the rationale. -- Bootstrap: see `ps1_meta.lua` for the rationale.
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath). -- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
-- Uses `debug.getinfo` to find this file's own directory, so it works -- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator.
-- both standalone and when require'd from the orchestrator.
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd). -- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module. -- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
-- Load the annotation pass so we can re-validate each source against the canonical corpus projection.
-- The annotation pass exposes `M.validate`, which returns the per-source AnnotationResult (atoms / annots / macros / binds / errors / warnings)
-- that the report pass renders into the per-module `<dir_basename>.annotations.txt` output.
local annotation = dofile(_bootstrap_dir .. "annotation.lua")
-- Load atoms_source_map for the `render_source_map` / `render_provenance` module functions (used by `render_module_atoms_md` to produce `<module>.atoms.md` without re-walking source tokens).
-- The pass itself emits no per-source files anymore; we only consume the two pure renderers here.
-- Defined BEFORE the renderer functions below so their upvalues resolve to this local (not the global `atoms_source_map`, which is nil).
local atoms_source_map = dofile(_bootstrap_dir .. "atoms_source_map.lua")
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Constants -- Constants
@@ -67,8 +73,6 @@ local PASS_NAME = "report"
--- @field project_root string -- project root (e.g. "code/") --- @field project_root string -- project root (e.g. "code/")
--- @field upstream table<string, table> -- per-pass upstream outputs --- @field upstream table<string, table> -- per-pass upstream outputs
--- @field flags table -- CLI flags + per-pass stash --- @field flags table -- CLI flags + per-pass stash
--- @field flags._annot_results ModuleEntry[] -- stashed by annotation pass
--- @field dry_run boolean -- if true, compute but don't write
--- @field verbose boolean -- if true, log diagnostic info --- @field verbose boolean -- if true, log diagnostic info
--- @class PassResult --- @class PassResult
@@ -138,338 +142,458 @@ local PASS_NAME = "report"
-- Per-MODULE annotation report (aggregated across all sources in a dir) -- Per-MODULE annotation report (aggregated across all sources in a dir)
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Extract the basename (last path segment) of a forward- or back-slash separated path. Returns the input unchanged if no separator is found. --- Extract the basename (last path segment) of a forward- or back-slash separated path. Returns the input unchanged if no separator is found.
-- @param path string --- @param path string
-- @return string --- @return string
local function source_basename(path) local function source_basename(path)
return path:match(BASENAME_PATTERN) or path return path:match(BASENAME_PATTERN) or path
end end
-- (internal) Format a single annotation entry as one rendered line. -- ════════════════════════════════════════════════════════════════════════════
-- @param a AnnotEntry -- Markdown renderers (consolidated-report-files refactor, 2026-07-26)
-- @param src_name string -- ════════════════════════════════════════════════════════════════════════════
-- @return string
local function format_annot_line(a, src_name) --- Render the thin project-wide summary (`build/atom_meta_report.summary.md`).
if a.error then --- @param all_results { module:string, atoms:integer, annots:integer, binds:integer,
return string.format(" ✗ line %d %s [ERROR: %s] [%s]", a.line, a.macro or "?", a.error, src_name) --- macros:integer, findings:integer, errors:integer,
--- warnings:integer, info:integer }[]
--- @return string
local function render_project_summary(all_results)
local lines = {
"# Project summary",
"> Auto-generated by ps1_meta.lua (passes/report.lua).",
"",
"| module | atoms | annots | binds | macros | findings | errors | warnings | info |",
"|--------|-------|--------|-------|--------|----------|--------|----------|------|",
}
local totals = { atoms = 0, annots = 0, binds = 0, macros = 0,
findings = 0, errors = 0, warnings = 0, info = 0 }
for _, e in ipairs(all_results) do
lines[#lines + 1] = string.format(
"| %s | %d | %d | %d | %d | %d | %d | %d | %d |",
e.module, e.atoms, e.annots, e.binds, e.macros,
e.findings, e.errors, e.warnings, e.info)
totals.atoms = totals.atoms + e.atoms
totals.annots = totals.annots + e.annots
totals.binds = totals.binds + e.binds
totals.macros = totals.macros + e.macros
totals.findings = totals.findings + e.findings
totals.errors = totals.errors + e.errors
totals.warnings = totals.warnings + e.warnings
totals.info = totals.info + e.info
end end
local line = string.format(" ● line %d %s [%s]", a.line, a.name, src_name) lines[#lines + 1] = string.format(
if a.binds then line = line .. " binds=" .. a.binds end "| **TOTAL** | %d | %d | %d | %d | %d | %d | %d | %d |",
if #a.reads > 0 then line = line .. " reads={" .. table.concat(a.reads, ",") .. "}" end totals.atoms, totals.annots, totals.binds, totals.macros,
if #a.writes > 0 then line = line .. " writes={" .. table.concat(a.writes, ",") .. "}" end totals.findings, totals.errors, totals.warnings, totals.info)
return line return table.concat(lines, "\n") .. "\n"
end end
-- (internal) Tally totals across all results in a module. --- Render the per-module verbose source-map markdown (`build/<module>.atoms.md`).
-- @param results AnnotationResult[] --- Per-source sub-section, per-atom stanza with sourcemap + provenance rows.
-- @return integer, integer, integer, integer, integer, integer --- Pulls sourcemap + provenance from `atoms_source_map` (no second source walk).
local function tally_module_totals(results) --- @param dir string
local total_atoms, total_annots, total_binds, total_macros = 0, 0, 0, 0 --- @param dir_sources SourceFile[]
local total_errors, total_warnings = 0, 0 --- @param wc table<string, integer>
for _, r in ipairs(results) do --- @return string
total_atoms = total_atoms + #r.atoms local function render_module_atoms_md(dir, dir_sources, wc)
total_annots = total_annots + #r.annots local dir_basename = source_basename(dir)
total_binds = total_binds + #r.binds local lines = {
total_macros = total_macros + #r.macros "# " .. dir_basename .. " — atoms (verbose source map)",
total_errors = total_errors + #r.errors "> Per-word call-site + provenance. Auto-generated.",
total_warnings = total_warnings + #r.warnings "",
}
for _, src in ipairs(dir_sources) do
local src_name = source_basename(src.path)
lines[#lines + 1] = "## " .. src_name
lines[#lines + 1] = ""
-- For each atom with a projection, render its sourcemap + provenance.
local atoms_list = {}
for _, atom in ipairs((src.scan or {}).atoms or {}) do
if atom.paths then atoms_list[#atoms_list + 1] = atom end
end
for _, atom in ipairs((src.scan or {}).raw_atoms or {}) do
if atom.paths then atoms_list[#atoms_list + 1] = atom end
end
if #atoms_list == 0 then
lines[#lines + 1] = "_(no atom projections)_"
lines[#lines + 1] = ""
else
-- Per-source forward-slash path (same one `emit_atom_stanza` / `emit_provenance_stanza` would derive;
-- computed once per `## <source>` heading and reused by each atom's `WORD N CALL ...` field).
local rel_path = src.path:gsub("\\\\", "/")
for _, atom in ipairs(atoms_list) do
lines[#lines + 1] = string.format(
"### atom: %s (line %d, %d words)",
atom.name, atom.line or 0, #(atom.paths.items or {}))
lines[#lines + 1] = ""
lines[#lines + 1] = "**Sourcemap** — per-word call site:"
lines[#lines + 1] = "```"
-- Per-atom invariant: call the per-atom renderers, NOT the per-source ones.
-- The per-source renderers enumerate every atom in `src`;
-- calling them in a per-atom loop would repeat the whole source under every `### atom:` heading.
lines[#lines + 1] = atoms_source_map.render_atom_source_map(atom):gsub("\n+$", "")
lines[#lines + 1] = "```"
lines[#lines + 1] = ""
lines[#lines + 1] = "**Provenance** — per-word definition + body:"
lines[#lines + 1] = "```"
lines[#lines + 1] = atoms_source_map.render_atom_provenance(atom, wc, rel_path):gsub("\n+$", "")
lines[#lines + 1] = "```"
lines[#lines + 1] = ""
end
end
end end
return total_atoms, total_annots, total_binds, total_macros, total_errors, total_warnings return table.concat(lines, "\n") .. "\n"
end end
-- (internal) Section renderer: per-source atom declarations. --- Render the consolidated per-module markdown (`build/<module>.atom_meta_report.md`).
local function render_module_atoms_section(add, results) --- Aggregates annotation + static-analysis content across all sources in `dir`.
add(SECTION_HEADER_ATOMS) --- Annotations come from re-running `annotation.validate()` per source (the existing pattern);
for _, r in ipairs(results) do --- static-analysis comes from `corpus.static_analysis_results[dir_basename]` (populated by `static_analysis.lua` — no second corpus_pipe_ctx build).
--- @param dir string
--- @param dir_sources SourceFile[]
--- @param annot_results AnnotationResult[]
--- @param sa_results table -- corpus.static_analysis_results[dir_basename]
--- @return string
local function render_module_meta_report(dir, dir_sources, annot_results, sa_results)
local dir_basename = source_basename(dir)
local lines = {
"# " .. dir_basename .. " — atom meta report",
"> Auto-generated by ps1_meta.lua (passes/report.lua). Do not edit.",
"",
}
local function add(s) lines[#lines + 1] = s end
-- Module summary table.
local n_atoms = 0
local n_annot = 0
local n_binds = 0
local n_macros = 0
local n_bare, n_proc = 0, 0
for _, r in ipairs(annot_results) do
n_atoms = n_atoms + #r.atoms
n_annot = n_annot + #r.annots
n_binds = n_binds + #r.binds
n_macros = n_macros + #r.macros
end
for _, a in ipairs(sa_results.atoms or {}) do
if a.kind == "comp_bare" then n_bare = n_bare + 1
elseif a.kind == "comp_proc" then n_proc = n_proc + 1
end
end
add("## Module summary"); add("")
add("| metric | value |"); add("|--------|-------|")
add(string.format("| sources | %d |", #dir_sources))
add(string.format("| atoms | %d (atoms: %d, comp_bare: %d, comp_proc: %d) |",
#(sa_results.atoms or {}),
#(sa_results.atoms or {}) - n_bare - n_proc, n_bare, n_proc))
add(string.format("| annotations | %d |", n_annot))
add(string.format("| binds structs | %d |", n_binds))
add(string.format("| macro decls | %d |", n_macros))
add(string.format("| findings | %d (errors: %d, warnings: %d, info: %d) |",
#(sa_results.findings or {}),
#(sa_results.errors or {}),
#(sa_results.warnings or {}),
#(sa_results.info or {})))
add("")
-- Sources
add("## Sources"); add("")
for _, s in ipairs(dir_sources) do add("- `" .. s.path .. "`") end
add("")
-- Atoms (annotation)
add("## Atoms"); add("")
add("| kind | name | source | line |"); add("|------|------|--------|------|")
for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source) local src_name = source_basename(r.source)
for _, a in ipairs(r.atoms) do for _, a in ipairs(r.atoms) do
add(string.format(" MipsAtom_(%s) line %d [%s]", a.name, a.line, src_name)) add(string.format("| atom | %s | %s | %d |", a.name, src_name, a.line))
end end
end end
add("") add("")
end
-- (internal) Section renderer: per-source annotation entries. -- Annotations
local function render_module_annots_section(add, results) add("## Annotations"); add("")
add(SECTION_HEADER_ANNOTS) if #annot_results == 0 then
for _, r in ipairs(results) do add("_(none)_")
local src_name = source_basename(r.source)
for _, a in ipairs(r.annots) do
add(format_annot_line(a, src_name))
end
end
add("")
end
-- (internal) Section renderer: per-source Binds_* struct declarations.
local function render_module_binds_section(add, results)
add(SECTION_HEADER_BINDS)
for _, r in ipairs(results) do
local src_name = source_basename(r.source)
for _, b in ipairs(r.binds) do
add(string.format(" %s line %d %d bytes [%s]", b.name, b.line, b.bytes, src_name))
for _, f in ipairs(b.fields) do
add(string.format(" +%2d: %s", f.offset, f.name))
end
end
end
add("")
end
-- (internal) Section renderer: per-source macro word-count declarations.
local function render_module_macros_section(add, results)
add(SECTION_HEADER_MACROS)
for _, r in ipairs(results) do
local src_name = source_basename(r.source)
for _, m in ipairs(r.macros) do
add(string.format(" %s line %d words=%d [%s]", m.name, m.line, m.words, src_name))
end
end
add("")
end
-- (internal) Section renderer: per-source errors (one-line + "(none)" if empty).
local function render_module_errors_section(add, results, total_errors)
add(SECTION_HEADER_ERRORS)
if total_errors == 0 then
add(" (none)")
else else
for _, r in ipairs(results) do add("| source | line | name | binds | reads | writes |")
add("|--------|------|------|-------|-------|--------|")
for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source) local src_name = source_basename(r.source)
for _, e in ipairs(r.errors) do for _, a in ipairs(r.annots) do
add(string.format(" ✗ line %d %s [%s]", e.line, e.msg, src_name)) local binds = a.binds or ""
local reads = (#a.reads > 0 and table.concat(a.reads, ",")) or ""
local writes = (#a.writes > 0 and table.concat(a.writes, ",")) or ""
add(string.format("| %s | %d | %s | %s | %s | %s |",
src_name, a.line, a.name, binds, reads, writes))
end end
end end
end end
add("") add("")
end
-- (internal) Section renderer: per-source warnings (one-line + "(none)" if empty). -- Binds_* structs
local function render_module_warnings_section(add, results, total_warnings) add("## Binds_* structs"); add("")
add(SECTION_HEADER_WARNINGS) if #annot_results == 0 then
if total_warnings == 0 then add("_(none)_")
add(" (none)")
else else
for _, r in ipairs(results) do for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source) local src_name = source_basename(r.source)
for _, w in ipairs(r.warnings) do for _, b in ipairs(r.binds) do
add(string.format(" ⚠ line %d %s [%s]", w.line, w.msg, src_name)) add(string.format("### %s (%s:%d, %d bytes)",
b.name, src_name, b.line, b.bytes))
for _, f in ipairs(b.fields) do
add(string.format("- `+%d %s`", f.offset, f.name))
end
add("")
end
end
end
-- Macro decls
add("## Macro word-count declarations"); add("")
if #annot_results == 0 then
add("_(none)_")
else
add("| source | line | macro declaration |")
add("|--------|------|-------------------|")
for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source)
for _, m in ipairs(r.macros) do
add(string.format("| %s | %d | %s |",
src_name, m.line, m.name))
end end
end end
end end
add("") add("")
-- Findings by atom (static-analysis)
add("## Static analysis — findings by atom"); add("")
local by_atom = {}
for _, f in ipairs(sa_results.findings or {}) do
by_atom[f.atom] = by_atom[f.atom] or {}
by_atom[f.atom][#by_atom[f.atom] + 1] = f
end
if next(by_atom) == nil then
add("_(no findings)_")
else
for _, a in ipairs(sa_results.atoms or {}) do
local fs = by_atom[a.name]
if fs then
add(string.format("### %s", a.name))
for _, f in ipairs(fs) do
add(string.format("- `[%s] %s`", f.check, f.msg))
end
add("")
end
end
end
-- Errors / Warnings / Info
local function add_findings(label, entries)
add(string.format("## %s", label))
if #entries == 0 then
add("_(none)_")
else
for _, e in ipairs(entries) do
add(string.format("- line %d %s", e.line, e.msg))
end
end
add("")
end
add_findings("Errors", sa_results.errors or {})
add_findings("Warnings", sa_results.warnings or {})
add_findings("Info", sa_results.info or {})
-- Per-atom cycle counts (path-aware)
add("## Per-atom cycle counts (path-aware, best case, no stalls)"); add("")
add("| atom | source | min | max | branches | paths | notes |")
add("|------|--------|-----|-----|----------|-------|-------|")
local sorted = {}
for _, a in ipairs(sa_results.atoms or {}) do sorted[#sorted + 1] = a end
table.sort(sorted, function(x, y)
return ((x.paths or {}).cycles_max or 0) > ((y.paths or {}).cycles_max or 0)
end)
for _, a in ipairs(sorted) do
local p = a.paths or {}
local src_name = a.source_path and source_basename(a.source_path) or ""
local notes = ""
if p.has_loops then notes = notes .. " [loop!]" end
if p.unknown_macros and #p.unknown_macros > 0 then
notes = notes .. " [unknown: " .. table.concat(p.unknown_macros, ", ") .. "]"
end
add(string.format("| %s | %s | %d | %d | %d | %d | %s |",
a.name, src_name,
p.cycles_min or 0, p.cycles_max or 0,
p.branches or 0, p.paths or 0, notes))
end
add("")
-- Per-source scan summary
add("## Per-source scan summary"); add("")
for _, src in ipairs(dir_sources) do
local src_atoms = {}
for _, a in ipairs(sa_results.atoms or {}) do
if a.source_path == src.path then src_atoms[#src_atoms + 1] = a end
end
if #src_atoms > 0 then
local mn, mx = math.huge, -1
for _, a in ipairs(src_atoms) do
local p = a.paths or {}
if (p.cycles_min or 0) < mn then mn = p.cycles_min or 0 end
if (p.cycles_max or 0) > mx then mx = p.cycles_max or 0 end
end
local path_str
if mx > 0 then
path_str = string.format(" cycles=%d..%d", mn, mx)
else
path_str = string.format(" %d cycles", mn)
end
add(string.format("- `%s` — %d atom%s%s",
src.basename, #src_atoms,
#src_atoms == 1 and "" or "s", path_str))
end
end
add("")
return table.concat(lines, "\n") .. "\n"
end end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- SECTION_RENDERERS — data-driven section dispatch (the plex pattern) -- REPORT_RENDERERS — data-driven report dispatch (one row per file kind)
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- -- `once = true` means render once at the project level (not per-module).
-- Each entry maps a section to its (header, render_fn). The render_fn signature: -- `basename(dir_basename)` yields the file's basename for that kind.
-- render_fn(add, results, totals) -- `gather(ctx, dir, dir_sources [, all_modules])` returns the rendered string.
-- add -- the `add(line)` closure from the surrounding report renderer local REPORT_RENDERERS = {
-- results -- AnnotationResult[] (per-source results) {
-- totals -- {atoms, annots, binds, macros, errors, warnings} counts name = "atom_meta_report",
-- ext = "md",
-- Sections that need to render "(none)" vs iterate use totals.errors / totals.warnings; basename = function(dir_basename) return dir_basename .. ".atom_meta_report" end,
-- other sections ignore the totals arg. once = false,
-- Adding a new section = 1 row here + 1 render_<thing>_section function. gather = function(ctx, dir, dir_sources)
local SECTION_RENDERERS = { -- Annotations: re-run `annotation.validate()` per source (the existing pattern).
{ header = SECTION_HEADER_ATOMS, render = render_module_atoms_section }, local annot_results = {}
{ header = SECTION_HEADER_ANNOTS, render = render_module_annots_section }, for _, src in ipairs(dir_sources) do
{ header = SECTION_HEADER_BINDS, render = render_module_binds_section }, if src.scan then
{ header = SECTION_HEADER_MACROS, render = render_module_macros_section }, local r = annotation.validate(ctx, src, nil)
{ header = SECTION_HEADER_ERRORS, render = function(add, results, totals) return render_module_errors_section(add, results, totals.errors) end }, r.source = src.path
{ header = SECTION_HEADER_WARNINGS, render = function(add, results, totals) return render_module_warnings_section(add, results, totals.warnings) end }, annot_results[#annot_results + 1] = r
end
end
-- Static-analysis: read stashed projection (no re-validate).
local dir_basename = dir:match("([^/\\]+)$") or dir
local sa_results = (ctx.shared.corpus.static_analysis_results or {})[dir_basename] or {}
return render_module_meta_report(dir, dir_sources, annot_results, sa_results)
end,
},
{
name = "atoms",
ext = "md",
basename = function(dir_basename) return dir_basename .. ".atoms" end,
once = false,
gather = function(ctx, dir, dir_sources)
return render_module_atoms_md(dir, dir_sources,
ctx.shared.corpus.word_counts or {})
end,
},
{
name = "summary",
ext = "md",
basename = function(_dir_basename) return "atom_meta_report.summary" end,
once = true,
gather = function(_ctx, _dir, _dir_sources, all_modules)
return render_project_summary(all_modules)
end,
},
} }
--- Render the per-MODULE annotation report (one `<dir_basename>.annotations.txt`).
--- @param dir string -- module directory path
--- @param sources SourceFile[] -- sources in this module
--- @param results AnnotationResult[] -- per-source validate() results
--- @return string -- the rendered report text
local function render_module_report(dir, sources, results)
local lines = {}
local function add(s) lines[#lines + 1] = s end
add(RULE_THICK)
add("ANNOTATION PASS — module " .. source_basename(dir))
add(RULE_THICK)
add(string.format("Sources: %d", #sources))
for _, s in ipairs(sources) do add(" " .. s.path) end
add("")
local total_atoms, total_annots, total_binds, total_macros, total_errors, total_warnings = tally_module_totals(results)
add(string.format("Atoms: %d Annotations: %d Binds structs: %d Macro decls: %d",
total_atoms, total_annots, total_binds, total_macros))
add("")
-- Bundle the totals so the section renderers don't need separate parameter lists.
-- Errors/warnings sections need their total count to decide "(none)" vs iterate.
-- Sections without totals (atoms/annots/binds/macros) ignore this arg.
local totals = {
atoms = total_atoms, annots = total_annots, binds = total_binds,
macros = total_macros, errors = total_errors, warnings = total_warnings,
}
-- THE per-section dispatch. ONE loop over SECTION_RENDERERS.
-- Each renderer writes its header + content via the `add` closure (pre-bound above).
-- Adding a new section = 1 row here + 1 render_<thing>_section function.
for _, section in ipairs(SECTION_RENDERERS) do
add(section.header)
section.render(add, results, totals)
add("")
end
return table.concat(lines, "\n") .. "\n"
end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Per-project summary -- M — public pass surface
-- ════════════════════════════════════════════════════════════════════════════
--- Render the per-project summary (`build/gen/annotation_validation.txt`).
--- Aggregates totals across all sources; lists per-source error counts if any source has errors.
--- @param all_results AnnotationResult[]
--- @return string
local function render_project_report(all_results)
local lines = {}
local function add(s) lines[#lines + 1] = s end
local total_atoms, total_annots, total_macros, total_binds = 0, 0, 0, 0
local total_errors, total_warnings = 0, 0
for _, r in ipairs(all_results) do
total_atoms = total_atoms + #r.atoms
total_annots = total_annots + #r.annots
total_macros = total_macros + #r.macros
total_binds = total_binds + #r.binds
total_errors = total_errors + #r.errors
total_warnings = total_warnings + #r.warnings
end
add(RULE_THICK)
add("ANNOTATION VALIDATION — project summary")
add(RULE_THICK)
add("")
add(string.format("Atoms: %d", total_atoms))
add(string.format("Annotations: %d", total_annots))
add(string.format("Macros: %d", total_macros))
add(string.format("Binds: %d", total_binds))
add("")
add(string.format("Errors: %d", total_errors))
add(string.format("Warnings: %d", total_warnings))
add("")
if total_errors > 0 then
add("Per-source error counts:")
for _, r in ipairs(all_results) do
if #r.errors > 0 then
local src_name = source_basename(r.source)
add(string.format(" %s : %d error(s)", src_name, #r.errors))
end
end
add("")
end
return table.concat(lines, "\n") .. "\n"
end
-- ════════════════════════════════════════════════════════════════════════════
-- Orchestration helpers
-- ════════════════════════════════════════════════════════════════════════════
-- (internal) Pull per-source validate() results from the annotation pass's stash.
-- The annotation pass runs first in the dep chain and caches results in `ctx.flags._annot_source_results`;
-- we read from there instead of re-validating each source.
-- Returns the list of module results + the flat list of all results (for the project-wide summary).
-- @param ctx PassCtx
-- @param dir_sources SourceFile[]
-- @return AnnotationResult[], AnnotationResult[]
local function lookup_module_results(ctx, dir_sources)
local src_cache = (ctx.flags and ctx.flags._annot_source_results) or {}
local module_results = {}
local all_results = {}
for _, src in ipairs(dir_sources) do
local result = src_cache[src.path]
if result then
result.source = src.path -- defensive (annotation tags it too; this guards against cache misses from earlier iterations)
module_results[#module_results + 1] = result
all_results[#all_results + 1] = result
end
end
return module_results, all_results
end
-- (internal) Does this module's results contain anything worth emitting?
-- @param module_results AnnotationResult[]
-- @return boolean
local function module_has_content(module_results)
for _, r in ipairs(module_results) do
if #r.atoms > 0 or #r.annots > 0 or #r.binds > 0
or #r.macros > 0 or #r.errors > 0 or #r.warnings > 0 then
return true
end
end
return false
end
-- (internal) Log a debug message if `_G[DEBUG_FLAG]` is truthy.
-- @param fmt string
local function debug_log(fmt, ...)
if _G[DEBUG_FLAG] then
io.stderr:write(string.format("[%s] " .. fmt, PASS_NAME, ...))
end
end
-- ════════════════════════════════════════════════════════════════════════════
-- M — module exports
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
local M = {} local M = {}
--- Run the report pass. --- Run the report pass. Emits 1 `atom_meta_report.summary.md` per build + 2 `atom_meta_report.md` + 2 `atoms.md` files per module (duffle + gte_hello).
--- Renders one `<dir_basename>.annotations.txt` per source-directory that has content, plus the project-wide `annotation_validation.txt` summary. --- Reads `corpus.static_analysis_results` (added in Phase 1) to populate per-module findings without re-running validate().
--- @param ctx PassCtx --- @param ctx PassCtx
--- @return PassResult --- @return PassResult
function M.run(ctx) function M.run(ctx)
local outputs = {} local outputs = {}
local errors = {} local corpus = ctx.shared and ctx.shared.corpus
local warnings = {} local by_dir = (corpus and corpus.sources_by_dir) or {}
local module_entries = (ctx.flags and ctx.flags._annot_results) or {} -- `out_path_root`: when the conventional `out_root` is `build/gen` (any spelling — relative, absolute, separator variants).
local by_dir = ctx.by_dir or duffle.group_sources_by_dir(ctx.sources) -- Write the md files to `build/` (parent of `gen/`) instead of nested under `gen/`.
-- Mirrors the `gdb_tape_atoms_runtime.gdb` relocation.
local function ends_with_gen(p)
return type(p) == "string" and (p:match("[/\\]gen[/\\]?$") ~= nil
or p == "build/gen" or p == "build\\gen")
end
local out_root_effective = ends_with_gen(ctx.out_root)
and ctx.out_root:gsub("[/\\]gen[/\\]?$", "")
or ctx.out_root
if not ctx.dry_run then duffle.ensure_dir(ctx.out_root) end duffle.ensure_dir(out_root_effective)
local all_results_for_summary = {} -- Aggregator for the project-wide `once = true` summary renderer.
for _, entry in ipairs(module_entries) do local all_modules = {}
debug_log("entry: dir=%s basename=%s atoms_count=%d dir_sources=%d\n", entry.dir, entry.dir_basename, entry.atoms_count, #(by_dir[entry.dir] or {}))
if entry.atoms_count > 0 or #(by_dir[entry.dir] or {}) > 0 then for dir, dir_sources in pairs(by_dir) do
local dir_sources = by_dir[entry.dir] or {} local dir_basename = dir:match("([^/\\]+)$") or dir
local module_results, all_results = lookup_module_results(ctx, dir_sources)
for _, r in ipairs(all_results) do -- Per-renderer dispatch for the per-module renderers (once = false).
all_results_for_summary[#all_results_for_summary + 1] = r for _, renderer in ipairs(REPORT_RENDERERS) do
if not renderer.once then
local body = renderer.gather(ctx, dir, dir_sources)
local out_path = out_root_effective .. "/" .. renderer.basename(dir_basename) .. "." .. renderer.ext
duffle.write_file(out_path, body)
outputs[#outputs + 1] = { kind = renderer.name, path = out_path }
end end
end
if module_has_content(module_results) then -- For the summary, compute per-module totals once (re-validating annotations per source — same pattern as the meta_report renderer).
local out_path = ctx.out_root .. "/" .. entry.dir_basename .. ".annotations.txt" local annot_results = {}
if not ctx.dry_run then for _, src in ipairs(dir_sources) do
duffle.write_file(out_path, render_module_report(entry.dir, dir_sources, module_results)) if src.scan then
end local r = annotation.validate(ctx, src, nil)
outputs[#outputs + 1] = { annotations_txt = out_path } r.source = src.path
else annot_results[#annot_results + 1] = r
debug_log(" -> no content; skipping\n")
end end
end end
local n_annot, n_binds, n_macros = 0, 0, 0
for _, r in ipairs(annot_results) do
n_annot = n_annot + #r.annots
n_binds = n_binds + #r.binds
n_macros = n_macros + #r.macros
end
local sa_results = (corpus.static_analysis_results or {})[dir_basename] or {}
all_modules[#all_modules + 1] = {
module = dir_basename,
atoms = #(sa_results.atoms or {}),
annots = n_annot,
binds = n_binds,
macros = n_macros,
findings = #(sa_results.findings or {}),
errors = #(sa_results.errors or {}),
warnings = #(sa_results.warnings or {}),
info = #(sa_results.info or {}),
}
end
-- Project-wide renderer (once = true): write the summary file.
for _, renderer in ipairs(REPORT_RENDERERS) do
if renderer.once then
local body = renderer.gather(ctx, nil, nil, all_modules)
local out_path = out_root_effective .. "/" .. renderer.basename("") .. "." .. renderer.ext
duffle.write_file(out_path, body)
outputs[#outputs + 1] = { kind = renderer.name, path = out_path }
end
end end
if not ctx.dry_run and #all_results_for_summary > 0 then return { outputs = outputs, errors = {}, warnings = {} }
local summary_path = ctx.out_root .. "/annotation_validation.txt"
duffle.write_file(summary_path, render_project_report(all_results_for_summary))
outputs[#outputs + 1] = { summary_txt = summary_path }
end
return { outputs = outputs, errors = errors, warnings = warnings }
end end
return M return M
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+44 -91
View File
@@ -1,11 +1,17 @@
--- word_count_eval.lua — Word-counting logic for the tape-atom metaprogram pipeline. --- word_count_eval.lua — Word-counting logic for the tape-atom metaprogram pipeline.
--- ---
--- Three responsibilities: --- Two responsibilities:
--- 1. **Public utilities** (used by `passes/components.lua`, `passes/offsets.lua`, `passes/annotation.lua`): --- 1. **Public utility** `M.count_token_words(token, wc)`: Used by `passes/offsets.lua`, `passes/annotation.lua`, and other passes.
--- - `M.count_token_words(token, wc)` — words emitted by one token --- 2. **Pass entry** `M.run(ctx)`: Loads the authored `word_count.metadata.h` into `ctx.shared.corpus.word_counts` for downstream passes.
--- - `M.scan_dir(dir, suffix)` — glob walk for *.macs.h --- The generated `.macs.h` files are OUTPUT artifacts and are NOT inputs to this pass;
--- 2. **Pass entry** `M.run(ctx)` — loads metadata.h + *.macs.h into `ctx.shared.word_counts` for downstream passes. --- Current component counts are owned by `passes/components.lua` (which populates `corpus.word_counts` and `corpus.component_body_index`
--- 3. **Internal helpers** for the body scanner. --- AFTER computing each current count from the just-built body + `corpus.word_counts`).
---
--- **Canonical contract**:
--- * `ctx.shared.corpus.word_counts` is the count table.
--- * `corpus.word_counts` is the sole count table. Consumers read `corpus.word_counts` directly.
--- * `ctx.shared.components` and `ctx.shared.component_body_index` are NOT created by this pass (projections only).
--- * No `.macs.h` recursive discovery (no `scan_dir`, no scan cache, no `_invalidate_scan_cache`).
--- ---
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex, --- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
--- Lua 5.3 compatible. --- Lua 5.3 compatible.
@@ -14,23 +20,11 @@
-- Module-scope requires + package.path setup -- Module-scope requires + package.path setup
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Resolve `arg[0]` to an absolute-ish script directory so that `require("duffle")` resolves against `scripts/` regardless of CWD.
-- Bootstrap: see `ps1_meta.lua` for the rationale.
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath). -- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
-- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator. -- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator.
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module. -- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
-- ════════════════════════════════════════════════════════════════════════════
-- Constants
-- ════════════════════════════════════════════════════════════════════════════
-- Required native extension: lfs (LuaFileSystem). Built by `update_deps.ps1` to
-- `toolchain/lfs/lfs.dll` and wired into package.cpath by `scripts/duffle_paths.lua`.
-- If lfs is missing, `require` throws — fail loud per the build-tool convention.
local lfs = require("lfs")
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Type declarations -- Type declarations
@@ -49,12 +43,12 @@ local lfs = require("lfs")
--- @field sources SourceFile[] -- all source files in the build --- @field sources SourceFile[] -- all source files in the build
--- @field metadata_path string -- path to word_count.metadata.h --- @field metadata_path string -- path to word_count.metadata.h
--- @field shared table -- cross-pass shared state --- @field shared table -- cross-pass shared state
--- @field shared.word_counts WordCounts -- populated by this pass --- @field shared.corpus table -- canonical corpus (required)
--- @field shared.corpus.word_counts WordCounts -- canonical count table (populated by this pass)
--- @field out_root string -- output root (e.g. "build/gen") --- @field out_root string -- output root (e.g. "build/gen")
--- @field project_root string -- project root (e.g. "code/") --- @field project_root string -- project root (e.g. "code/")
--- @field upstream table<string, table> -- per-pass upstream outputs --- @field upstream table<string, table> -- per-pass upstream outputs
--- @field flags table -- CLI flags --- @field flags table -- CLI flags
--- @field dry_run boolean -- if true, compute but don't write
--- @field verbose boolean -- if true, log diagnostic info --- @field verbose boolean -- if true, log diagnostic info
--- @class PassResult --- @class PassResult
@@ -76,8 +70,8 @@ local M = {}
--- For most tokens (regular MIPS instructions) this returns 1. --- For most tokens (regular MIPS instructions) this returns 1.
--- For `mac_X(...)` calls, this returns the resolved word count from `wc` (recursively if needed). For `nop2` etc., returns wc[name]. --- For `mac_X(...)` calls, this returns the resolved word count from `wc` (recursively if needed). For `nop2` etc., returns wc[name].
--- For unknown macros, returns 1 and (optionally) warns. --- For unknown macros, returns 1 and (optionally) warns.
--- @param token string -- a single token from split_top_level_commas --- @param token string -- a single token from split_top_level_commas
--- @param wc WordCounts -- the shared word-count table --- @param wc WordCounts -- the shared word-count table
--- @return integer --- @return integer
function M.count_token_words(token, wc) function M.count_token_words(token, wc)
local s = duffle.trim(token) local s = duffle.trim(token)
@@ -92,83 +86,42 @@ function M.count_token_words(token, wc)
return 1 return 1
end end
-- ┌────────────────────────────────────────────────────────────────────┐
-- │ Shared utility: scan_dir │
-- └────────────────────────────────────────────────────────────────────┘
-- Cache the scan_dir result per (dir, suffix) in package.loaded.
-- The cache persists for the lifetime of the Lua process (cleared when ps1_meta.lua exits).
-- If a build removes/creates .macs.h files mid-process, the caller can invalidate by calling `M._invalidate_scan_cache()`.
local SCAN_CACHE_KEY = "__word_count_eval_scan_cache__"
--- Scan `code/` for files matching `suffix` (e.g. `*.macs.h`).
--- Native directory enumeration via lfs (~2ms). Zero subprocess spawns.
--- @param dir string -- project root directory
--- @param suffix string -- file pattern, e.g. "*.macs.h"
--- @return string[]
function M.scan_dir(dir, suffix)
local key = dir .. "\0" .. suffix
local cache = package.loaded[SCAN_CACHE_KEY]
if cache and cache[key] then return cache[key] end
local results = {}
local code_dir = dir .. "/code"
if lfs.attributes(code_dir, "mode") == "directory" then
for mod_name in lfs.dir(code_dir) do
if mod_name ~= "." and mod_name ~= ".." then
local gen_path = code_dir .. "/" .. mod_name .. "/gen"
if lfs.attributes(gen_path, "mode") == "directory" then
for fname in lfs.dir(gen_path) do
if fname:match("%.macs%.h$") then
results[#results + 1] = gen_path .. "/" .. fname
end
end
end
end
end
end
-- Cache the result (including empty results).
cache = cache or {}
cache[key] = results
package.loaded[SCAN_CACHE_KEY] = cache
return results
end
--- Invalidate the scan cache (call after creating new .macs.h files in the same Lua process — usually not needed).
function M._invalidate_scan_cache() package.loaded[SCAN_CACHE_KEY] = nil end
-- ┌────────────────────────────────────────────────────────────────────┐ -- ┌────────────────────────────────────────────────────────────────────┐
-- │ Pass entry: M.run(ctx) — "word-counts" pass │ -- │ Pass entry: M.run(ctx) — "word-counts" pass │
-- └────────────────────────────────────────────────────────────────────┘ -- └────────────────────────────────────────────────────────────────────┘
--- Load metadata.h + scan for existing *.macs.h files into ctx.shared.word_counts. --- Load the authored `word_count.metadata.h` into `ctx.shared.corpus.word_counts`.
--- Loading the .macs.h files is idempotent: entries from later (current-build) .macs.h files override metadata.h entries of the same name. --- Generated `.macs.h` files are OUTPUT artifacts and are NOT scanned as inputs.
--- Current component counts are computed and inserted by `passes/components.lua`
--- after the components pass iterates `corpus.source_order` and writes each source's `<dir_basename>.macs.h` file.
---
--- Contract:
--- * `ctx.shared.corpus` MUST exist (canonical corpus ownership).
--- * `ctx.metadata_path` MUST be a readable file path to the authored `word_count.metadata.h`.
--- * The pass assigns exactly one table to `corpus.word_counts`.
--- Consumers read the corpus-owned table directly.
--- Consumers must read `corpus.word_counts` directly.
--- @param ctx PassCtx --- @param ctx PassCtx
--- @return PassResult --- @return PassResult
function M.run(ctx) function M.run(ctx)
local wc = {} -- 1. Canonical-corpus ownership gate.
local corpus = ctx.shared and ctx.shared.corpus
-- 1. Load metadata.h (the encoding-macro source of truth). if type(corpus) ~= "table" then
local meta_counts = duffle.load_word_counts(ctx.metadata_path) error("word_count_eval.run requires ctx.shared.corpus (canonical corpus). The fixture must install the corpus before running this pass.", 0)
for name, count in pairs(meta_counts) do wc[name] = count end
-- 2. Scan project_root recursively for *.macs.h files (component-macro source).
local macs_files = M.scan_dir(ctx.project_root, "*.macs.h")
for _, macs_path in ipairs(macs_files) do
local ok, mc = pcall(duffle.load_word_counts, macs_path)
if not ok then
io.stderr:write(string.format("[word_count_eval] parse error in '%s': %s\n", macs_path, tostring(mc)))
elseif type(mc) ~= "table" then
io.stderr:write(string.format("[word_count_eval] '%s' did not return a table (got %s)\n", macs_path, type(mc)))
else
for name, count in pairs(mc) do wc[name] = count end
end
end end
ctx.shared.word_counts = wc -- 2. metadata_path gate.
if type(ctx.metadata_path) ~= "string" or ctx.metadata_path == "" then
error("word_count_eval.run requires ctx.metadata_path (path to the authored word_count.metadata.h).", 0)
end
-- 3. Load authored metadata. Generated .macs.h files are NOT scanned
-- (the pass computes their counts from the just-built bodies after disk emission; see passes/components.lua).
local wc = duffle.load_word_counts(ctx.metadata_path)
-- 4. Assign the count table. ONE assignment, no copy. The assignment creates no secondary alias.
corpus.word_counts = wc
return { outputs = {}, errors = {}, warnings = {} } return { outputs = {}, errors = {}, warnings = {} }
end end
+290 -483
View File
File diff suppressed because it is too large Load Diff
+1 -16
View File
@@ -39,13 +39,7 @@ pop-location
# ════════════════════════════════════════════════════════════════════════════ # ════════════════════════════════════════════════════════════════════════════
# PCSX-Redux — built via MSBuild (VS2022) # PCSX-Redux — built via MSBuild (VS2022)
#
# Requires: Visual Studio 2022 with the C++ desktop workload. # Requires: Visual Studio 2022 with the C++ desktop workload.
# The .vcxproj files target platform toolset v145, but VS2022 ships v143;
# we pass /p:PlatformToolset=v143 to retarget at build time (no file edits).
# NuGet packages (glfw, luajit.native, libFFmpeg-lite, x64sentry) are
# restored automatically by MSBuild on first build.
#
# Output: toolchain\pcsx-redux\vsprojects\x64\Debug\pcsx-redux.exe # Output: toolchain\pcsx-redux\vsprojects\x64\Debug\pcsx-redux.exe
# ════════════════════════════════════════════════════════════════════════════ # ════════════════════════════════════════════════════════════════════════════
@@ -65,8 +59,7 @@ $path_pcsx_sln = join-path $path_pcsx_redux 'vsprojects\pcsx-redux.sln'
& $msbuild_exe $path_pcsx_sln /p:Configuration=Release /p:Platform=x64 /p:PlatformToolset=v143 /m /v:minimal & $msbuild_exe $path_pcsx_sln /p:Configuration=Release /p:Platform=x64 /p:PlatformToolset=v143 /m /v:minimal
# Locate luajit via scoop. `luajit.exe` is on PATH via scoop's shim; # Locate luajit via scoop. `luajit.exe` is on PATH via scoop's shim;
# we use `scoop prefix` to find the install root for the include dir # we use `scoop prefix` to find the install root for the include dir (needed to compile lpeg against luajit's headers).
# (needed to compile lpeg against luajit's headers).
# If scoop or luajit is missing, fail fast with an actionable message. # If scoop or luajit is missing, fail fast with an actionable message.
$luajit_prefix = & scoop prefix luajit 2>$null $luajit_prefix = & scoop prefix luajit 2>$null
if (-not $luajit_prefix -or -not (Test-Path (Join-Path $luajit_prefix 'bin/luajit.exe'))) { if (-not $luajit_prefix -or -not (Test-Path (Join-Path $luajit_prefix 'bin/luajit.exe'))) {
@@ -87,7 +80,6 @@ if (-not $lua_inc_dir) {
# Generate lpeg.dll by compiling the 6 source files directly. # Generate lpeg.dll by compiling the 6 source files directly.
# `gcc` is on PATH (scoop's shim puts it there). # `gcc` is on PATH (scoop's shim puts it there).
# The source files: lpcap.c lpcode.c lpcset.c lpprint.c lptree.c lpvm.c # The source files: lpcap.c lpcode.c lpcset.c lpprint.c lptree.c lpvm.c
# (per the lpeg makefile — no `make.lua` template generator in this version).
# Link against luajit's import library (`libluajit-5.1.a`) for the Lua C API symbols (lua_*, luaL_*). # Link against luajit's import library (`libluajit-5.1.a`) for the Lua C API symbols (lua_*, luaL_*).
$luajit_lib_dir = Join-Path $luajit_prefix 'lib' $luajit_lib_dir = Join-Path $luajit_prefix 'lib'
$lpeg_sources = @('lpcap.c', 'lpcode.c', 'lpcset.c', 'lpprint.c', 'lptree.c', 'lpvm.c') $lpeg_sources = @('lpcap.c', 'lpcode.c', 'lpcset.c', 'lpprint.c', 'lptree.c', 'lpvm.c')
@@ -103,8 +95,6 @@ pop-location
# ════════════════════════════════════════════════════════════════════════════ # ════════════════════════════════════════════════════════════════════════════
# lfs (LuaFileSystem) — compiled from pcsx-redux's vendored luafilesystem source. # lfs (LuaFileSystem) — compiled from pcsx-redux's vendored luafilesystem source.
# Used by word_count_eval.lua :: scan_dir for native directory enumeration (~2ms)
# instead of spawning `dir /b /s` as a subprocess (~56ms).
# Source: toolchain/pcsx-redux/third_party/luafilesystem/src/lfs.c # Source: toolchain/pcsx-redux/third_party/luafilesystem/src/lfs.c
# Output: toolchain/lfs/lfs.dll # Output: toolchain/lfs/lfs.dll
# ════════════════════════════════════════════════════════════════════════════ # ════════════════════════════════════════════════════════════════════════════
@@ -118,11 +108,6 @@ $lfs_dll_import = join-path $luajit_lib_dir 'libluajit-5.1.dll.a'
# ════════════════════════════════════════════════════════════════════════════ # ════════════════════════════════════════════════════════════════════════════
# OpenBIOS — built from the PCSX-Redux source tree via make + mipsel-none-elf # OpenBIOS — built from the PCSX-Redux source tree via make + mipsel-none-elf
#
# OpenBIOS is an open-source PS1 BIOS implementation (no retail BIOS dump needed).
# It builds with the MIPS cross-toolchain (`mipsel-none-elf-gcc`, on PATH via the `mips` toolchain installer)
# + `make` (on PATH via scoop).
#
# Output: toolchain\pcsx-redux\src\mips\openbios\openbios.bin # Output: toolchain\pcsx-redux\src\mips\openbios\openbios.bin
# ════════════════════════════════════════════════════════════════════════════ # ════════════════════════════════════════════════════════════════════════════