Author SHA1 Message Date
ed 27a9038e0d req c11, 2026-07-26 17:36:46 -04:00
ed 8c8d2e54aa remove cruft 2026-07-26 14:40:57 -04:00
ed 80a35aa23a WIP: Better step debug on atom components, better db_skip annotation, lots of curation passes on lua.
Still don't have this thing in its final state for  the curse but its close.
2026-07-26 13:55:47 -04:00
ed f247d56c32 Debug vis ergonomics 2026-07-25 13:19:35 -04:00
ed 590ff1e2ec Curation pass: reduce nested conditional branching in some defnitions. 2026-07-25 13:00:36 -04:00
ed 653e18ee28 remove code related to dry run and dep graph rendering (ps1 meta) 2026-07-25 11:59:41 -04:00
ed ebb876fe89 report.lua: Remove redudnant section formatting/header 2026-07-25 11:25:12 -04:00
ed 1b40b16c0e Review pass. 2026-07-25 11:20:53 -04:00
28 changed files with 1560 additions and 2046 deletions
+10 -5
View File
@@ -11,7 +11,7 @@
* Pure macro anntation. * Pure macro anntation.
* --------------- * ---------------
* Don't want to constraint the macro usage to some attribute placment constraint, etc, don't want ot dela with the compiler. * Don't want to constraint the macro usage to some attribute placment constraint, etc, don't want ot dela with the compiler.
* atom_info, atom_bind, atom_reads, atom_writes, atom_label, atom_dbg_skip_over each expand to a C comment or to nothing * atom_info, atom_bind, atom_reads, atom_writes, atom_label, atom_dbg_skip each expand to a C comment or to nothing
* (C preprocessor strips them to whitespace). * (C preprocessor strips them to whitespace).
* *
* ============================================================================ * ============================================================================
@@ -90,13 +90,18 @@
#define atom_info(...) /* atom_info(__VA_ARGS__) */ #define atom_info(...) /* atom_info(__VA_ARGS__) */
/* ---------------------------------------------------------------------------- /* ----------------------------------------------------------------------------
* DEBUG SOURCE-STEP MARKERS * DEBUG SOURCE-STEP MARKER
* *
* Place atom_dbg_skip_over() before a MipsAtom_, MipsAtomComp_, or MipsAtomComp_Proc_. * Place `atom_dbg_skip` (BARE) before a MipsAtom_, MipsAtomComp_, or MipsAtomComp_Proc_.
* The following declaration kind determines whether the marker selects a whole atom or a component inline view. * The following declaration kind determines whether the marker selects a whole atom or a component inline view.
* The source scanner associates the marker with that declaration; placement diagnostics are handled by the annotation pass. * The source scanner associates the marker with that declaration; placement diagnostics are handled by the annotation pass.
*
* Example:
* atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(rret_addr), nop };
* atom_dbg_skip MipsAtomComp_(ac_yield) { ... };
* atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, { ... });
* ----------------------------------------------------------------------------*/ * ----------------------------------------------------------------------------*/
#define atom_dbg_skip_over() /* atom_dbg_skip_over: skip the following atom or component source view */ #define atom_dbg_skip /* atom_dbg_skip: skip the following atom or component source view */
/* ---------------------------------------------------------------------------- /* ----------------------------------------------------------------------------
* Typed-view annotations (Registry for DWARF RR_<R_X> chain resolution) * Typed-view annotations (Registry for DWARF RR_<R_X> chain resolution)
@@ -117,7 +122,7 @@
* The preferred correlation mechanism; atom_ctx is the escape hatch for non-natural cases. * The preferred correlation mechanism; atom_ctx is the escape hatch for non-natural cases.
* *
* All three expand to C comments * All three expand to C comments
* (the bare-token convention matching `atom_reg` and `atom_dbg_skip_over`). * (the bare-token convention matching `atom_reg` and `atom_dbg_skip`).
* The Lua scanner reads the bare tokens in source-as-written; the C preprocessor strips them. * The Lua scanner reads the bare tokens in source-as-written; the C preprocessor strips them.
* ----------------------------------------------------------------------------*/ * ----------------------------------------------------------------------------*/
#define atom_type(T) /* atom_type: associate <T> with the preceding enum entry (enum site) or this register (atom-info site) */ #define atom_type(T) /* atom_type: associate <T> with the preceding enum entry (enum site) or this register (atom-info site) */
+3
View File
@@ -218,3 +218,6 @@ IA_ void assert(U8 cond) { if(cond){return;} else{debug_trap(); ms_exit_process(
#endif #endif
#pragma endregion Debug #pragma endregion Debug
#endif #endif
#define GCC_OPTIMIZATION_DISABLE _Pragma("GCC push_options") _Pragma("GCC optimize(\"O0\")")
#define GCC_OPTIMIZATION_ENABLE _Pragma("GCC pop_options")
+14
View File
@@ -9,6 +9,12 @@
#define WORD_COUNT(name, count) enum { words_##name = (count) }; #define WORD_COUNT(name, count) enum { words_##name = (count) };
#endif #endif
/* atom_dbg_skip */
/* ---------------------------------------------------------------------------
* MACRO ATOM Components (Reusable Assembly Components)
* These do NOT yield. They are expanded inline inside Tape Atoms.
* ---------------------------------------------------------------------------*/
// The 'Yield' sequence for Tape Atoms (mac_yield).
#define mac_yield(...) \ #define mac_yield(...) \
load_word(R_AtomJmp, R_TapePtr, 0) \ load_word(R_AtomJmp, R_TapePtr, 0) \
, add_ui_self( R_TapePtr, S_(MipsCode)) \ , add_ui_self( R_TapePtr, S_(MipsCode)) \
@@ -16,6 +22,7 @@
, nop , nop
WORD_COUNT(mac_yield, 4) WORD_COUNT(mac_yield, 4)
/* atom_dbg_skip */
/* Words: 3; Loads 3 S2 indices from the face array */ /* Words: 3; Loads 3 S2 indices from the face array */
#define mac_load_tri_indices(...) \ #define mac_load_tri_indices(...) \
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)) \ load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)) \
@@ -23,6 +30,8 @@ WORD_COUNT(mac_yield, 4)
, load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)) , load_half_u(R_T2, R_FaceCursor, 2 * S_(S2))
WORD_COUNT(mac_load_tri_indices, 3) WORD_COUNT(mac_load_tri_indices, 3)
/* atom_dbg_skip */
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
#define mac_gte_load_tri_verts(...) \ #define mac_gte_load_tri_verts(...) \
shift_lleft(R_AT, R_T0, v3s2_byteoff) \ shift_lleft(R_AT, R_T0, v3s2_byteoff) \
, add_u_self(R_AT, R_VertBase) \ , add_u_self(R_AT, R_VertBase) \
@@ -74,16 +83,19 @@ WORD_COUNT(mac_insert_ot_tag_f3, 11)
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */ , store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
WORD_COUNT(mac_insert_ot_tag_g4, 11) WORD_COUNT(mac_insert_ot_tag_g4, 11)
/* atom_dbg_skip */
#define mac_pack_color_word(off, cmd, r, g, b) \ #define mac_pack_color_word(off, cmd, r, g, b) \
load_upper_i(R_AT, (cmd) << 8 | (b)) \ load_upper_i(R_AT, (cmd) << 8 | (b)) \
, or_i_self( R_AT, ((g) << 8) | (r)) \ , or_i_self( R_AT, ((g) << 8) | (r)) \
, store_word( R_AT, R_PrimCursor, (off)) , store_word( R_AT, R_PrimCursor, (off))
WORD_COUNT(mac_pack_color_word, 3) WORD_COUNT(mac_pack_color_word, 3)
/* atom_dbg_skip */
#define mac_format_f3_color(r, g, b) \ #define mac_format_f3_color(r, g, b) \
mac_pack_color_word(O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) mac_pack_color_word(O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b)
WORD_COUNT(mac_format_f3_color, 3) WORD_COUNT(mac_format_f3_color, 3)
/* atom_dbg_skip */
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3. /* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */ * PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
#define mac_gte_store_f3_post_rtpt(...) \ #define mac_gte_store_f3_post_rtpt(...) \
@@ -99,6 +111,7 @@ WORD_COUNT(mac_gte_store_f3_post_rtpt, 3)
, mac_pack_color_word(O_(Poly_G4,c3), 0, r3,g3,b3) , mac_pack_color_word(O_(Poly_G4,c3), 0, r3,g3,b3)
WORD_COUNT(mac_format_g4_color, 12) WORD_COUNT(mac_format_g4_color, 12)
/* atom_dbg_skip */
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the /* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
* G4 triangle portion to p0/p1/p2. * G4 triangle portion to p0/p1/p2.
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). * PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
@@ -113,6 +126,7 @@ WORD_COUNT(mac_format_g4_color, 12)
, gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2)) , gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2))
WORD_COUNT(mac_gte_store_g4_p012_post_rtpt_pre_rtps, 3) WORD_COUNT(mac_gte_store_g4_p012_post_rtpt_pre_rtps, 3)
/* atom_dbg_skip */
/* Words: 1; Stores the V3 screen coord to the G4's p3 slot. /* Words: 1; Stores the V3 screen coord to the G4's p3 slot.
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its * PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its
* single-vertex result to SXY2; SXY0 still holds v0.screen from the * single-vertex result to SXY2; SXY0 still holds v0.screen from the
+2 -2
View File
@@ -2,7 +2,7 @@
* duffle DSL — GPU Vendor Mnemonics (opt-in) * duffle DSL — GPU Vendor Mnemonics (opt-in)
* ============================================================================ * ============================================================================
* *
* Provides the PSYQ-style CamelCase aliases for the canonical duffle GPU primitive setters and OT operations. * Provides the PSYQ-style CamelCase aliases for the duffle GPU primitive setters and OT operations.
* The duffle snake_case names are primary; this header is for users who prefer the PSYQ SDK function names from the legacy C API. * The duffle snake_case names are primary; this header is for users who prefer the PSYQ SDK function names from the legacy C API.
* *
* USAGE: #include "duffle/gp_vendor_sym.h" // after gp.h * USAGE: #include "duffle/gp_vendor_sym.h" // after gp.h
@@ -24,7 +24,7 @@
* The gp0_cmd_* / gp1_cmd_* byte constants are already short and descriptive; no vendor alias is provided for them. * The gp0_cmd_* / gp1_cmd_* byte constants are already short and descriptive; no vendor alias is provided for them.
* *
* The vendor mnemonics are NOT registered with the duffle word-count metadata (word_counts.metadata.h). * The vendor mnemonics are NOT registered with the duffle word-count metadata (word_counts.metadata.h).
* They expand to the duffle canonical macros which DO have word-count entries * They expand to the duffle macros which DO have word-count entries
* (the ones emitted by mac_format_f3_color / mac_gte_store_f3 / etc.). Verification: V13 (objdump byte-identical) holds. * (the ones emitted by mac_format_f3_color / mac_gte_store_f3 / etc.). Verification: V13 (objdump byte-identical) holds.
* ============================================================================ */ * ============================================================================ */
+30 -43
View File
@@ -353,41 +353,34 @@ enum { _C2_TX_SUBS_ = 0
/* GTE command words for the common cases. /* GTE command words for the common cases.
* *
* These are pure compile-time integer constants — the C compiler * These are pure compile-time integer constants — the C compiler constant-folds them into `.word` directives in .rodata.
* constant-folds them into `.word` directives in .rodata. Use them * Use them inside `asm_inline(...)` blocks (see `gte_rtpt` below for the idiom).
* inside `asm_inline(...)` blocks (see `gte_rtpt` below for the
* canonical idiom).
* *
* Decomposition (per the `enc_gte_<field>` definitions above): * Decomposition (per the `enc_gte_<field>` definitions above):
* gte_cmdw_<name> = gte_cmd_base | enc_gte_cmd(<cmd>) * gte_cmdw_<name> = gte_cmd_base | enc_gte_cmd(<cmd>)
* The SF/MX/V/CV/LM fields are all zero in the common cases (standard * The SF/MX/V/CV/LM fields are all zero in the common cases
* rotation-matrix, no scaling factor, V0 vector, translation vector, * (standard rotation-matrix, no scaling factor, V0 vector, translation vector, no clamp),
* no clamp), so the only varying bits are the `cmd` field. * so the only varying bits are the `cmd` field.
* *
* Naming follows the file's convention: `gte_cmd_*` is the raw * Naming follows the file's convention: `gte_cmd_*` is the raw 6-bit `cmd` field id, `gte_cmdw_*`
* 6-bit `cmd` field id, `gte_cmdw_*` is the fully-encoded 32-bit * is the fully-encoded 32-bit instruction word ready to drop into a `.word` directive.
* instruction word ready to drop into a `.word` directive.
* *
* -------------------------------------------------------------------------- * --------------------------------------------------------------------------
* PsyQ-compatibility note (RTPS/RTPT): * PsyQ-compatibility note (RTPS/RTPT):
* The original Sony PsyQ `inline_n.h` ships RTPT as `cop2 0x0280030` and * The original Sony PsyQ `inline_n.h` ships RTPT as `cop2 0x0280030` and RTPS as `cop2 0x0180001`.
* RTPS as `cop2 0x0180001`. Both have `0x20` set in the upper-reserved * Both have `0x20` set in the upper-reserved region (bit 21) AND `sf=1` (bit 19) — i.e. the "no division" flag.
* region (bit 21) AND `sf=1` (bit 19) — i.e. the "no division" flag. * Per psx-spec these bits are reserved/must-be-zero,
* Per psx-spec these bits are reserved/must-be-zero, but the real GTE * but the real GTE hardware and PCSX-Redux's GTE model both IGNORE them on these two commands
* hardware and PCSX-Redux's GTE model both IGNORE them on these two * (the perspective divide happens regardless of `sf`).
* commands (the perspective divide happens regardless of `sf`).
* *
* If we emit a strictly-spec-compliant word (`sf=0`, reserved bits * If we emit a strictly-spec-compliant word (`sf=0`, reserved bits clear),
* clear), PCSX-Redux's GTE checks those bits more strictly than the * PCSX-Redux's GTE checks those bits more strictly than the silicon does and RTPT silently no-ops —
* silicon does and RTPT silently no-ops — the floor's screen * the floor's screen coordinates come out as raw projection-of-rotation (Z never divided),
* coordinates come out as raw projection-of-rotation (Z never * `nclip` ends up wrong, and the triangle is culled.
* divided), `nclip` ends up wrong, and the triangle is culled.
* *
* So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to * So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern everyone has shipped for 25 years.
* match the working bit pattern everyone has shipped for 25 years. * NCLIP/OP/MVMVA stay spec-clean — their reserved bits really are zero in the original PsyQ source.
* NCLIP/OP/MVMVA stay spec-clean — their reserved bits really are
* zero in the original PsyQ source.
* -------------------------------------------------------------------------- * --------------------------------------------------------------------------
*/ */
#define gte_cmdw_psyq_compat (1u << 21 | enc_gte_sf(gte_sf_integer)) #define gte_cmdw_psyq_compat (1u << 21 | enc_gte_sf(gte_sf_integer))
@@ -425,8 +418,8 @@ enum { _C2_TX_SUBS_ = 0
* (XY at offset 0) and C2_VZ0 (Z at offset 4) using `lwc2`. * (XY at offset 0) and C2_VZ0 (Z at offset 4) using `lwc2`.
* *
* Uses string-style GCC inline asm with `%0` substitution because the * Uses string-style GCC inline asm with `%0` substitution because the
* base register `r0` is a runtime GPR chosen by the compiler — it cannot * base register `r0` is a runtime GPR chosen by the compiler.
* be encoded into a static `.word` constant. * It cannot be encoded into a static `.word` constant.
* *
* Usage: * Usage:
* asm_gte_load_v0(svector_ptr); * asm_gte_load_v0(svector_ptr);
@@ -458,26 +451,21 @@ enum {
/* gte_load_vN(r_ptr, base) — placeholder-punned lwc2 loaders /* gte_load_vN(r_ptr, base) — placeholder-punned lwc2 loaders
* *
* Emits `.word` constants encoding `lwc2 $N, off(<base>)` for the chosen * Emits `.word` constants encoding `lwc2 $N, off(<base>)` for the chosen GTE vector register, where `<base>` is the GPR number you pass in
* GTE vector register, where `<base>` is the GPR number you pass in
* (typically one of R_T4..R_T9 for the standard "3-pointer" pattern). * (typically one of R_T4..R_T9 for the standard "3-pointer" pattern).
* *
* The caller MUST bind `r_ptr` to that same GPR via a register variable: * The caller MUST bind `r_ptr` to that same GPR via a register variable:
* register V3_S2* p_in_12 __asm__("$12") = my_ptr; * register V3_S2* p_in_12 __asm__("$12") = my_ptr;
* gte_load_v0(p_in_12, R_T4); // R_T4 = 12, base is $12 * gte_load_v0(p_in_12, R_T4); // R_T4 = 12, base is $12
* *
* Then `"r"(r_ptr)` inside the asm binds to $12 (the only register * Then `"r"(r_ptr)` inside the asm binds to $12 (the only register `p_in_12` can live in),
* `p_in_12` can live in), which is exactly the register the .word * which is exactly the register the .word constants expect. A `"$12"` clobber would conflict with the register-variable binding
* constants expect. A `"$12"` clobber would conflict with the * ("asm specifier for variable conflicts with asm clobber list"), so we omit it.
* register-variable binding ("asm specifier for variable conflicts * The other ABI-clobbers ($2/$8/$9/$31) stay because the GTE instructions don't touch caller-saved GPRs but the kernel does treat them as volatile.
* with asm clobber list"), so we omit it. The other ABI-clobbers
* ($2/$8/$9/$31) stay because the GTE instructions don't touch
* caller-saved GPRs but the kernel does treat them as volatile.
* *
* WHICH REGISTER TO PICK * WHICH REGISTER TO PICK
* ---------------------- * ----------------------
* Any caller-saved GPR is safe. Recommended default for an RTPT-style * Any caller-saved GPR is safe. Recommended default for an RTPT-style 3-pointer pipeline:
* 3-pointer pipeline:
* gte_load_v0(p0, R_T4); // $12 * gte_load_v0(p0, R_T4); // $12
* gte_load_v1(p1, R_T5); // $13 * gte_load_v1(p1, R_T5); // $13
* gte_load_v2(p2, R_T6); // $14 * gte_load_v2(p2, R_T6); // $14
@@ -490,8 +478,7 @@ enum {
* clobbers section : "$2", "$8", ..., "memory" (from asm_clobber) * clobbers section : "$2", "$8", ..., "memory" (from asm_clobber)
* 3 colons total, GCC-legal. No string-syntax mnemonics in the .word body. * 3 colons total, GCC-legal. No string-syntax mnemonics in the .word body.
* *
* The `asm_clobber(...)` helper from gcc_asm.h prepends the colon that * The `asm_clobber(...)` helper from gcc_asm.h prepends the colon that starts the clobbers section. */
* starts the clobbers section. */
#define gte_load_v0(r_ptr, base) asm volatile( \ #define gte_load_v0(r_ptr, base) asm volatile( \
asm_words( gte_lw_v0_xy(base), gte_lw_v0_z(base) ) \ asm_words( gte_lw_v0_xy(base), gte_lw_v0_z(base) ) \
asm_rpins, r_use(r_ptr) \ asm_rpins, r_use(r_ptr) \
@@ -510,11 +497,11 @@ enum {
asm_clobber: rlit(R_V0), rlit(R_T0), rlit(R_T1), rlit(R_RA), clb_mem_drain \ asm_clobber: rlit(R_V0), rlit(R_T0), rlit(R_T1), rlit(R_RA), clb_mem_drain \
) )
/* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — the canonical prelude to gte_cmd_rtpt. /* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — prelude to gte_cmd_rtpt.
* *
* Loads all three GTE input vectors (6 words) from three separate pointers, * Loads all three GTE input vectors (6 words) from three separate pointers,
* one per GTE vector register, each loaded from its own base GPR. Caller * one per GTE vector register, each loaded from its own base GPR.
* must bind each `pN` to `bN` via a register variable. * Caller must bind each `pN` to `bN` via a register variable.
* *
* register V3_S2* p0 rgcc(R_T4) = verts[0].ptr; // → __asm__("$12") * register V3_S2* p0 rgcc(R_T4) = verts[0].ptr; // → __asm__("$12")
* register V3_S2* p1 rgcc(R_T5) = verts[1].ptr; // → __asm__("$13") * register V3_S2* p1 rgcc(R_T5) = verts[1].ptr; // → __asm__("$13")
+1 -1
View File
@@ -2,7 +2,7 @@
* duffle DSL — GTE Vendor Mnemonics (opt-in) * duffle DSL — GTE Vendor Mnemonics (opt-in)
* ============================================================================ * ============================================================================
* *
* Provides the textbook MIPS assembly mnemonics for the GTE/COP2 instructions as thin aliases to the canonical duffle macros in gte.h. * Provides the textbook MIPS assembly mnemonics for the GTE/COP2 instructions as thin aliases to the duffle macros in gte.h.
* The duffle names are primary; this header is for users who prefer the textbook mnemonics. * The duffle names are primary; this header is for users who prefer the textbook mnemonics.
* *
* USAGE: #include "duffle/gte_vendor_sym.h" // after gte.h * USAGE: #include "duffle/gte_vendor_sym.h" // after gte.h
+10 -14
View File
@@ -57,10 +57,10 @@ enum {
* ---------------------------------------------------------------------------*/ * ---------------------------------------------------------------------------*/
/* The 'Exit' Atom */ /* The 'Exit' Atom */
MipsAtom_(tape_exit) { jump_reg(rret_addr), nop }; atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(rret_addr), nop };
/* Generalized Tape Engine Runner */ /* Generalized Tape Engine Runner */
FI_ void tape_run(Slice_MipsCode tape) { register U4* tp rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile( NI_ void tape_run(Slice_MipsCode tape) { register U4* tp rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile(
asm_words( asm_words(
add_ui( R_SP, R_SP, -MipsStackAlignment) /* Allocate stack space */ add_ui( R_SP, R_SP, -MipsStackAlignment) /* Allocate stack space */
, store_word( R_RA, R_SP, 0) /* Safely backup $ra to the stack */ , store_word( R_RA, R_SP, 0) /* Safely backup $ra to the stack */
@@ -104,23 +104,21 @@ FI_ Slice_MipsCode tb_slice(TapeBuilder tb) { return (Sl
* ---------------------------------------------------------------------------*/ * ---------------------------------------------------------------------------*/
// The 'Yield' sequence for Tape Atoms (mac_yield). // The 'Yield' sequence for Tape Atoms (mac_yield).
atom_dbg_skip_over() atom_dbg_skip MipsAtomComp_(ac_yield) {
MipsAtomComp_(ac_yield) {
load_word(R_AtomJmp, R_TapePtr, 0), load_word(R_AtomJmp, R_TapePtr, 0),
add_ui_self( R_TapePtr, S_(MipsCode)), add_ui_self( R_TapePtr, S_(MipsCode)),
jump_reg( R_AtomJmp), nop, jump_reg( R_AtomJmp), nop,
}; };
/* Words: 3; Loads 3 S2 indices from the face array */ /* Words: 3; Loads 3 S2 indices from the face array */
MipsAtomComp_(ac_load_tri_indices) { atom_dbg_skip MipsAtomComp_(ac_load_tri_indices) {
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)), load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)), load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)), load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
}; };
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */ /* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
atom_dbg_skip_over() atom_dbg_skip MipsAtomComp_(ac_gte_load_tri_verts) {
MipsAtomComp_(ac_gte_load_tri_verts) {
shift_lleft(R_AT, R_T0, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0), shift_lleft(R_AT, R_T0, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
shift_lleft(R_AT, R_T1, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1), shift_lleft(R_AT, R_T1, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
shift_lleft(R_AT, R_T2, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2), shift_lleft(R_AT, R_T2, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
@@ -159,7 +157,7 @@ MipsAtomComp_(ac_insert_ot_tag_g4) {
/* Words: 3; Emits one (cmd|color) word to R_PrimCursor at the given /* Words: 3; Emits one (cmd|color) word to R_PrimCursor at the given
* byte offset. Internal helper used by the *_format_*_color macros. */ * byte offset. Internal helper used by the *_format_*_color macros. */
FI_ MipsAtom ac_pack_color_word(U4 off, U4 cmd, U1 r, U1 g, U1 b) FI_ MipsAtom ac_pack_color_word(U4 off, U4 cmd, U1 r, U1 g, U1 b)
MipsAtomComp_Proc_(ac_pack_color_word, { atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, {
load_upper_i(R_AT, (cmd) << 8 | (b)), load_upper_i(R_AT, (cmd) << 8 | (b)),
or_i_self( R_AT, ((g) << 8) | (r)), or_i_self( R_AT, ((g) << 8) | (r)),
store_word( R_AT, R_PrimCursor, (off)), store_word( R_AT, R_PrimCursor, (off)),
@@ -168,11 +166,11 @@ MipsAtomComp_Proc_(ac_pack_color_word, {
/* Words: 3; Emits the F3 command+color word (cmd byte | BLUE | GREEN | RED) /* Words: 3; Emits the F3 command+color word (cmd byte | BLUE | GREEN | RED)
* Args: _r, _g, _b are 8-bit RGB byte values (not raw 16-bit fields). */ * Args: _r, _g, _b are 8-bit RGB byte values (not raw 16-bit fields). */
FI_ MipsAtom ac_format_f3_color(U1 r, U1 g, U1 b) FI_ MipsAtom ac_format_f3_color(U1 r, U1 g, U1 b)
MipsAtomComp_Proc_(ac_format_f3_color, { mac_pack_color_word(O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) }) atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, { mac_pack_color_word(O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3. /* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */ * PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
MipsAtomComp_(ac_gte_store_f3_post_rtpt) { atom_dbg_skip MipsAtomComp_(ac_gte_store_f3_post_rtpt) {
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_F3,p0)), gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_F3,p0)),
gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_F3,p1)), gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_F3,p1)),
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_F3,p2)), gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_F3,p2)),
@@ -200,7 +198,7 @@ MipsAtomComp_Proc_(ac_format_g4_color, {
* three registers aligned with v0/v1/v2 you must store before RTPS). * three registers aligned with v0/v1/v2 you must store before RTPS).
* The macro name declares the pipeline position; check #6 (GTE state- * The macro name declares the pipeline position; check #6 (GTE state-
* machine validation) verifies the call site matches the declaration. */ * machine validation) verifies the call site matches the declaration. */
MipsAtomComp_(ac_gte_store_g4_p012_post_rtpt_pre_rtps) { atom_dbg_skip MipsAtomComp_(ac_gte_store_g4_p012_post_rtpt_pre_rtps) {
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_G4,p0)), gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_G4,p0)),
gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_G4,p1)), gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_G4,p1)),
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2)), gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2)),
@@ -212,7 +210,7 @@ MipsAtomComp_(ac_gte_store_g4_p012_post_rtpt_pre_rtps) {
* earlier RTPT — DO NOT read SXY0 here, that's the bug this name * earlier RTPT — DO NOT read SXY0 here, that's the bug this name
* prevents). * prevents).
*/ */
MipsAtomComp_(ac_gte_store_g4_p3_post_rtps) { gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p3)) }; atom_dbg_skip MipsAtomComp_(ac_gte_store_g4_p3_post_rtps) { gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p3)) };
#pragma endregion Macro Atom Components #pragma endregion Macro Atom Components
@@ -296,7 +294,6 @@ internal MipsAtom_(set_gte_world) atom_info(
/* DIAGNOSTIC 1: Pure tape loop test */ /* DIAGNOSTIC 1: Pure tape loop test */
internal MipsAtom_(diag_yield) { mac_yield() }; internal MipsAtom_(diag_yield) { mac_yield() };
// TODO(Ed): Reduce magic numbers/offsets
/* DIAGNOSTIC 2: Pure memory test (No GTE). Draws a fixed cyan triangle. */ /* DIAGNOSTIC 2: Pure memory test (No GTE). Draws a fixed cyan triangle. */
internal MipsAtom_(diag_color) { internal MipsAtom_(diag_color) {
store_word( R_0, R_T7, 0), store_word( R_0, R_T7, 0),
@@ -325,7 +322,6 @@ internal MipsAtom_(diag_color) {
mac_yield() mac_yield()
}; };
// TODO(Ed): Reduce magic numbers/offsets
/* DIAGNOSTIC 3: Pure GTE test (No Memory Writes) */ /* DIAGNOSTIC 3: Pure GTE test (No Memory Writes) */
internal MipsAtom_(diag_gte) { internal MipsAtom_(diag_gte) {
/* Load 3 indices */ /* Load 3 indices */
+1 -1
View File
@@ -436,7 +436,7 @@ enum { _BitOffsets = 0
/* --- Shift-amount alias (matches the gas convention `\p3 = shamt`) --- */ /* --- Shift-amount alias (matches the gas convention `\p3 = shamt`) --- */
#define shift_amount(rd, rt, n) shift_lleft(rd, rt, n) #define shift_amount(rd, rt, n) shift_lleft(rd, rt, n)
/* nop — canonical sll $0, $0, 0 */ /* nop — sll $0, $0, 0 */
#define nop shift_lleft(rdiscard, rdiscard, 0) #define nop shift_lleft(rdiscard, rdiscard, 0)
#define nop2 nop, nop #define nop2 nop, nop
+1 -1
View File
@@ -2,7 +2,7 @@
* duffle DSL — MIPS Vendor Mnemonics (opt-in) * duffle DSL — MIPS Vendor Mnemonics (opt-in)
* ============================================================================ * ============================================================================
* *
* Provides the textbook MIPS assembly mnemonics as thin aliases to the canonical duffle macros in mips.h. * Provides the textbook MIPS assembly mnemonics as thin aliases to the duffle macros in mips.h.
* The duffle names are primary; this header is for users who prefer the textbook mnemonics. * The duffle names are primary; this header is for users who prefer the textbook mnemonics.
* *
* USAGE: #include "duffle/mips_vendor_sym.h" // after mips.h * USAGE: #include "duffle/mips_vendor_sym.h" // after mips.h
+2 -6
View File
@@ -187,6 +187,7 @@ void gp_display_frame(DoubleBuffer* screen_buf, S4* active_buf_id, U4* ordering_
void render(void) { void render(void) {
} }
GCC_OPTIMIZATION_DISABLE
void update(PrimitiveArena* pa, U4* ordering_buf) void update(PrimitiveArena* pa, U4* ordering_buf)
{ {
orderingtbl_clear_reverse(ordering_buf, OrderingTbl_Len); orderingtbl_clear_reverse(ordering_buf, OrderingTbl_Len);
@@ -384,12 +385,6 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
TapeBuilder tb = tb_make_old(& tape_arena); tb_scope(& tb) { TapeBuilder tb = tb_make_old(& tape_arena); tb_scope(& tb) {
// Skip set_gte_world atom for diagnostics to isolate the triangle loop // Skip set_gte_world atom for diagnostics to isolate the triangle loop
for (U4 i = 0; i < Floor_num_faces; i++) { for (U4 i = 0; i < Floor_num_faces; i++) {
// =======================================================
// SWAP EMIT TO TEST DIFFERENT PARTS OF THE PIPELINE:
// =======================================================
// 1. code_diag_yield -> Tests Tape Engine jump logic
// 2. code_diag_color -> Tests OT and Prim Arena memory
// 3. code_diag_gte -> Tests Vertex arrays and GTE Math
// tb_emit(& tb, code_diag_yield); // tb_emit(& tb, code_diag_yield);
// tb_emit(& tb, code_diag_color); // tb_emit(& tb, code_diag_color);
// tb_emit(& tb, code_diag_gte); // tb_emit(& tb, code_diag_gte);
@@ -400,6 +395,7 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
pa->used = (U4)prim_cursor - (U4)r_(pa->buf)[smem.active_buf_id]; pa->used = (U4)prim_cursor - (U4)r_(pa->buf)[smem.active_buf_id];
} }
} }
GCC_OPTIMIZATION_ENABLE
int main(void) int main(void)
{ {
+1 -1
View File
@@ -103,7 +103,7 @@ MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(f
mac_yield() mac_yield()
}; };
// atom_dbg_skip_over() atom_dbg_skip
internal internal
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3) MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase) , atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
+1 -1
View File
@@ -140,7 +140,7 @@ function compile-unit { param(
$f_arch_no_shared, $f_arch_no_shared,
$f_arch_no_stack_prot $f_arch_no_stack_prot
) )
# $compile_args += $f_std_c23 $compile_args += $f_std_c11
$compile_args += ($f_include + $path_psyq_imyu_inc) $compile_args += ($f_include + $path_psyq_imyu_inc)
$compile_args += ($f_include + $path_nugget) $compile_args += ($f_include + $path_nugget)
+336 -433
View File
File diff suppressed because it is too large Load Diff
+7 -6
View File
@@ -11,11 +11,10 @@
--- ``` --- ```
--- ---
--- That small bootstrap: (a) locates this helper via `arg[0]` / `debug.getinfo`, --- That small bootstrap: (a) locates this helper via `arg[0]` / `debug.getinfo`,
--- (b) loads it (which sets `package.path` + `package.cpath` via cached `git rev-parse`), --- (b) loads it (which sets `package.path` + `package.cpath`),
--- (c) at the bottom calls `require("duffle")` (now resolvable since `package.path` was just set) and returns the duffle M. --- (c) at the bottom calls `require("duffle")` (now resolvable since `package.path` was just set) and returns the duffle M.
--- Net effect: the caller gets the duffle module in one statement; no separate `dofile(...)` + `require("duffle")` dance. --- Net effect: the caller gets the duffle module in one statement; no separate `dofile(...)` + `require("duffle")` dance.
--- ---
--- Replaces the prior 2-line (entry) or 4-line (pass) pattern that had the call site do its own path resolution + duplicated setup.
local M = {} local M = {}
@@ -27,9 +26,6 @@ local CACHE_KEY = "__duffle_repo_root__"
--- parent of the directory containing this script. We derive it directly from `debug.getinfo(1, "S").source` --- parent of the directory containing this script. We derive it directly from `debug.getinfo(1, "S").source`
--- (returns `@<path>` for the currently-running chunk). --- (returns `@<path>` for the currently-running chunk).
--- ---
--- Replaces the prior `io.popen("git rev-parse --show-toplevel")` approach, which cost ~100-180ms per
--- LuaJIT process on Windows due to git's CLI startup. The path-derive approach costs <1ms.
---
--- If `debug.getinfo` can't parse this script's path (shouldn't happen — dofile always populates source), --- If `debug.getinfo` can't parse this script's path (shouldn't happen — dofile always populates source),
--- return nil and let `M.setup()` fail loud. --- return nil and let `M.setup()` fail loud.
--- @return string|nil --- @return string|nil
@@ -61,7 +57,12 @@ end
function M.setup() function M.setup()
local repo_root = find_repo_root() local repo_root = find_repo_root()
if not repo_root then if not repo_root then
io.stderr:write("[duffle_paths] git rev-parse failed -- not in a git repo?\n") -- Unreachable in practice: find_repo_root() derives the repo root from this script's
-- own source path via debug.getinfo(1, "S").source (no subprocess, no git CLI, <1ms).
-- A nil return means the source path did not match the expected
-- <repo>/scripts/duffle_paths.lua layout — a packaging bug, not a "missing git repo"
-- condition. os.exit(2) is retained so a real failure surfaces loud rather than
-- silently producing an unconfigured module table.
os.exit(2) os.exit(2)
end end
+2 -130
View File
@@ -616,7 +616,7 @@ end
--- - ELF32 symtab entry = 16 bytes (`st_name:4 + st_value:4 + st_size:4 + st_info:1 + st_other:1 + st_shndx:2`); offsets within each entry are zero-based wire offsets. --- - ELF32 symtab entry = 16 bytes (`st_name:4 + st_value:4 + st_size:4 + st_info:1 + st_other:1 + st_shndx:2`); offsets within each entry are zero-based wire offsets.
--- - Direct Lua `string.byte`/`string.sub`/`string.find` boundaries receive `+ 1`. --- - Direct Lua `string.byte`/`string.sub`/`string.find` boundaries receive `+ 1`.
--- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded. --- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded.
--- - We strip the `code_` prefix to match the previous `read_nm` output. --- - The `code_` prefix is stripped (MipsAtom_ macros emit bare atom names, no `code_` prefix).
--- - `st_size > 0` filter excludes undefined/imported symbols. --- - `st_size > 0` filter excludes undefined/imported symbols.
--- @param elf_path Path --- @param elf_path Path
--- @return table<string, {integer, integer}> --- @return table<string, {integer, integer}>
@@ -654,7 +654,7 @@ function M.read_nm(elf_path)
-- Extract the name from .strtab (null-terminated C string). -- Extract the name from .strtab (null-terminated C string).
local name_end = strtab:find("\0", st_name_off + 1, true) or (st_name_off + 1) local name_end = strtab:find("\0", st_name_off + 1, true) or (st_name_off + 1)
local name = strtab:sub(st_name_off + 1, name_end - 1) local name = strtab:sub(st_name_off + 1, name_end - 1)
-- Filter: keep all symbol-table symbols (atoms emit their name as the bare `<name>` since the `code_` prefix was removed from the MipsAtom_ macro). -- Filter: keep all symbol-table symbols (atoms emit their name as the bare `<name>` — MipsAtom_ macros strip the `code_` prefix).
-- The atoms_source_map pass already filters out non-atom symbols via the source-map.txt cross-ref. -- The atoms_source_map pass already filters out non-atom symbols via the source-map.txt cross-ref.
if name and #name > 0 then if name and #name > 0 then
local st_value = M.read_u32_le(symtab, entry_off + SYM_ST_VALUE) local st_value = M.read_u32_le(symtab, entry_off + SYM_ST_VALUE)
@@ -791,132 +791,4 @@ end
-- I/O helpers: atoms source-map + native directory glob -- I/O helpers: atoms source-map + native directory glob
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- Parse a FORMAT_VERSION <expected_version> atoms-meta file (sourcemap or provenance).
--- Shared by M.parse_source_map_file + M.parse_provenance_file.
--- The two callers differ only in how they parse WORD lines; that's `extract_word(line)`.
--- Returns the standard `{name -> {total, words}}` shape.
--- Returns `{}` on format-version mismatch (and logs to stderr).
--- @param path string
--- @param expected_version integer
--- @param extract_word fun(line: string): table|nil -- caller-supplied per-line parser
--- @return table<string, table>
function M.parse_atom_records(path, expected_version, extract_word)
local out = {}
local cur_name, cur_words = nil, {}
for raw in io.lines(path) do
local line = raw
if line:match("^#") then
local ver = line:match("^# FORMAT_VERSION%s+(%d+)")
if ver and tonumber(ver) ~= expected_version then
io.stderr:write(string.format(
"[elf_dwarf.parse_atom_records] version mismatch (got %s, expected %d) in %s\n",
ver, expected_version, path))
return {}
end
-- skip other comments
elseif line:sub(1, 4) == "ATOM" then
-- ATOM <name> "<abs-source-path>" <total>
local _, _, name = line:find("ATOM%s+(%S+)%s+\"[^\"]*\"%s+(%d+)")
if name then
cur_name = name
cur_words = {}
out[name] = { total = 0, words = cur_words }
end
elseif line == "ENDATOM" then
-- Update the recorded total from the entries count
-- (matches the `lines[1] = lines[1]:gsub(" 0$", " " .. total)` patch in atoms_source_map.lua:170).
if cur_name and out[cur_name] then
out[cur_name].total = #cur_words
end
cur_name, cur_words = nil, {}
elseif line:sub(1, 4) == "WORD" and cur_name then
local field = extract_word(line)
if field then
cur_words[#cur_words + 1] = field
end
end
end
return out
end
--- Parse a FORMAT_VERSION <expected_version> `*.atoms.sourcemap.txt` file.
--- Returns `{name -> {total = N, words = {{pos, line}, ...}}}`.
--- Returns `{}` on format-version mismatch (and logs to stderr).
---
--- **Wire format** (emitted by `passes/atoms_source_map.lua`):
--- ```
--- # FORMAT_VERSION <n>
--- ATOM <name> "<abs-source-path>" <total>
--- WORD <n> LINE <line> TEXT <text...>
--- ...
--- ENDATOM
--- ```
---
--- **Conventions:** the in-memory shape uses `{pos, line, text}`
--- (`atoms_source_map.lua:142`); the `.txt` file uses `WORD <n>` so the parser maps `n` → `pos` field name.
--- @param sm_path Path
--- @param expected_version integer -- expected FORMAT_VERSION line
--- @return table<string, table>
function M.parse_source_map_file(sm_path, expected_version)
return M.parse_atom_records(sm_path, expected_version, function(line)
local _, n, _, src_line = line:find("WORD%s+(%d+)%s+LINE%s+(%d+)")
if n and src_line then
return { pos = tonumber(n), line = tonumber(src_line) }
end
end)
end
--- Parse a FORMAT_VERSION <expected_version> `*.atoms.provenance.txt` file.
--- Returns `{name -> {total = N, words = {{pos, call_file, call_line, comp_name, comp_file, comp_line}, ...}}}`.
--- Returns `{}` on format-version mismatch (and logs to stderr).
---
--- **Wire format** (emitted by `passes/atoms_source_map.lua`):
--- ```
--- # FORMAT_VERSION <n>
--- ATOM <name> "<abs-source-path>" <total>
--- WORD <n> CALL <src-file>:<src-line> RAW
--- WORD <n> CALL <src-file>:<src-line> MACRO <comp_name> "<comp-file>:<comp-line>"
--- ...
--- ENDATOM
--- ```
---
--- **Used by** `passes/dwarf_injection.lua` to:
--- - group consecutive MACRO rows into component invocations (one `DW_TAG_inlined_subroutine` each)
--- - emit abstract `DW_TAG_subprogram` per unique component name
--- - extend `.debug_line` so stepping into a `mac_X(...)` lands on the component's source line.
--- @param prov_path string -- path to *.atoms.provenance.txt
--- @param expected_version integer -- expected FORMAT_VERSION line
--- @return table<string, table>
function M.parse_provenance_file(prov_path, expected_version)
return M.parse_atom_records(prov_path, expected_version, function(line)
-- Two accepted shapes:
-- WORD <n> CALL <call-file>:<call-line> RAW
-- WORD <n> CALL <call-file>:<call-line> MACRO <comp_name> "<comp-file>:<comp-line>"
local pos, call_file, call_line, comp_name, comp_file, comp_line =
line:match('WORD%s+(%d+)%s+CALL%s+(.-):(%d+)%s+MACRO%s+(%S+)%s+"([^"]*):(%d+)"')
if pos then
return {
pos = tonumber(pos),
call_file = call_file,
call_line = tonumber(call_line),
comp_name = comp_name,
comp_file = comp_file,
comp_line = tonumber(comp_line),
}
end
-- RAW row.
local raw_pos, raw_file, raw_line = line:match('WORD%s+(%d+)%s+CALL%s+(.-):(%d+)%s+RAW')
if raw_pos then
return {
pos = tonumber(raw_pos),
call_file = raw_file,
call_line = tonumber(raw_line),
comp_name = nil,
comp_file = nil,
comp_line = nil,
}
end
end)
end
return M return M
-5
View File
@@ -28,11 +28,6 @@ param(
$ErrorActionPreference = 'Stop' $ErrorActionPreference = 'Stop'
$gdbInitPath = [System.IO.Path]::GetFullPath((Join-Path $PSScriptRoot '..\build\gen\hello_gte.gdbinit'))
if (-not (Test-Path -LiteralPath $gdbInitPath -PathType Leaf)) {
Write-Warning "Generated GDB skip sidecar missing (non-fatal): $gdbInitPath. Run the GTE build to regenerate it; debugger launch will continue without generated skip-over commands."
}
# ── Pre-checks ── # ── Pre-checks ──
foreach ($p in @($PcsxPath, $ExePath, $HelperZip)) { foreach ($p in @($PcsxPath, $ExePath, $HelperZip)) {
if (-not (Test-Path $p)) { if (-not (Test-Path $p)) {
+55 -104
View File
@@ -1,21 +1,20 @@
--- passes/annotation.lua — Atom-annotation DSL validator. --- passes/annotation.lua — Atom-annotation DSL validator.
--- ---
--- Validates `MipsAtom_(name) atom_info(atom_bind(Binds_X), atom_reads(...), atom_writes(...)) { ... }` declarations in source files. --- Validates `MipsAtom_(name) atom_info(atom_bind(Binds_X), atom_reads(...), atom_writes(...)) { ... }` declarations in source files.
--- Also reads: `Binds_*` struct declarations (`typedef Struct_(Binds_X) { ... };`) --- Also reads `Binds_*` struct declarations (`typedef Struct_(Binds_X) { ... };`).
--- ---
--- Source scanning: done ONCE upstream by `duffle.scan_source()` (ps1_meta.lua pre-scans each source and stashes the result in `src.scan`). --- `duffle.scan_source()` scans each source once upstream; `ps1_meta.lua` stores that result in `src.scan`.
--- ---
--- Writes: --- Ownership: the canonical `ctx.shared.corpus` supplies cross-source registries, while each `src.scan` supplies its source's declarations and bodies.
--- - `<ctx.out_root>/<dir_basename>.errors.h` — one per module, with `#error` directives on findings (the C compile will surface the error) --- A context without `ctx.shared.corpus` is rejected with an explicit canonical-corpus message.
--- - The annotations.txt report is rendered by `passes/report.lua` from the per-module results stashed in `ctx.flags._annot_results`
--- ---
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex, Lua 5.3 compatible --- Writes `<ctx.out_root>/<dir_basename>.errors.h` once per module, with `#error` directives for findings that the C compile surfaces.
--- `passes/report.lua` renders annotations.txt from `corpus.sources_by_dir`, re-validating each source through `M.validate()`.
---
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex, Lua 5.3 compatible.
-- Bootstrap: same as entry scripts. See `ps1_meta.lua` for the rationale. -- Bootstrap follows the entry scripts; `scripts/duffle_paths.lua` sets package.path and package.cpath. See `ps1_meta.lua` for the rationale.
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath). -- `debug.getinfo(1, "S").source` locates this file for standalone and orchestrated runs, then `duffle_paths.lua` returns the loaded `duffle` module.
-- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator.
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
local write_file = duffle.write_file local write_file = duffle.write_file
@@ -45,8 +44,6 @@ local ensure_dir = duffle.ensure_dir
--- @field project_root string --- @field project_root string
--- @field upstream table<string, table> --- @field upstream table<string, table>
--- @field flags table --- @field flags table
--- @field flags._annot_results table[] -- stashed by annotation pass; consumed by report.lua
--- @field dry_run boolean
--- @field verbose boolean --- @field verbose boolean
--- @class PassResult --- @class PassResult
@@ -64,15 +61,15 @@ local ensure_dir = duffle.ensure_dir
--- @field writes string[] -- R_* names (write targets) --- @field writes string[] -- R_* names (write targets)
--- @field errors string[]|nil -- parse-time errors from scan_source (atom_info body malformed) --- @field errors string[]|nil -- parse-time errors from scan_source (atom_info body malformed)
--- @class SkipOverMarker -- sub-shape of scan_source.lua's @class SkipOverMarker --- @class DebugSkipMarker -- sub-shape of scan_source.lua's @class DebugSkipMarker
--- @field marker_kind string -- exact marker ident (always "atom_dbg_skip_over") --- @field marker_kind string -- exact marker ident read from source. Only "atom_dbg_skip" (bare) is positive.
--- @field marker_line integer --- @field marker_line integer
--- @field args string|nil -- trimmed text inside the parens (nil when has_parens is false) --- @field args string|nil -- trimmed text inside the parens (nil when has_parens is false)
--- @field has_parens boolean --- @field has_parens boolean
--- @field is_bare boolean -- true iff marker_kind == "atom_dbg_skip" AND has_parens == false (the only positive form)
--- @field pending boolean -- true while awaiting the following declaration --- @field pending boolean -- true while awaiting the following declaration
--- @field superseded_by_marker_line integer|nil -- set on a marker that was bumped out of the pending slot --- @field superseded_by_marker_line integer|nil -- set on a marker that was bumped out of the pending slot
--- @field target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed --- @field target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed
--- @field declaration_line integer|nil
--- @class Finding --- @class Finding
--- @field line integer -- source line (or 0 for pass-level) --- @field line integer -- source line (or 0 for pass-level)
@@ -106,10 +103,8 @@ local ensure_dir = duffle.ensure_dir
-- Per-check functions (the CHECK_RULES table's payload) -- Per-check functions (the CHECK_RULES table's payload)
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- --
-- Each check has a uniform `append_to_findings` shape (errors[] / warnings[] / info[]). --- The dispatcher in `validate()` routes each result by convention: existence checks write errors[] and shape checks write warnings[].
-- The dispatcher in `validate()` decides which findings list each check writes to — by convention, --- `macro_word_drift` writes errors[] for missing or mismatched metadata and info[] for a match.
-- "existence" checks (declaration must exist, struct must exist) write errors[]; "shape" checks (writes/reads must be wave-context) write warnings[].
-- The `macro_word_drift` check writes both errors[] (missing/mismatch) and info[] (match).
--- Check: every annotated atom must have a matching MipsAtom_(name) declaration. --- Check: every annotated atom must have a matching MipsAtom_(name) declaration.
--- @param a AtomAnnotation --- @param a AtomAnnotation
@@ -140,9 +135,7 @@ local function check_unique_annotation(pipe_ctx, findings)
end end
--- Check: BIND atoms must reference a real Binds_* struct. --- Check: BIND atoms must reference a real Binds_* struct.
--- Emitting a warning here keeps the annotation pass from being stop-on-error for the common test-fixture case, --- I keep this as a warning so the annotation pass can report the common test-fixture case; `check_abi_handoff` in static analysis supplies the build-stopping error.
--- while still surfacing the issue in the report.
--- The static-analysis report remains the source of truth for build-stopping errors.
--- @param a AtomAnnotation --- @param a AtomAnnotation
--- @param pipe_ctx PipeCtx --- @param pipe_ctx PipeCtx
--- @param findings Findings --- @param findings Findings
@@ -184,9 +177,8 @@ local function check_macro_word_drift(m, wc, findings)
} }
end end
--- Check: atom_dbg_reg_default(R_X, <type>) must target a register declared as a debug-visible alias in `pipe_ctx.register_alias_registry`, --- Check: atom_dbg_reg_default(R_X, <type>) targets an alias in `pipe_ctx.register_alias_registry` and a type in `pipe_ctx.type_name_registry`.
--- with a type name found in `pipe_ctx.type_name_registry`. --- Pointer depth remains bounded to 0 or 1, and duplicate defaults remain errors.
--- Pointer depth is still bounded to 0 or 1. Duplicate defaults are still detected.
--- @param _src SourceFile -- unused (kept for the per_source shape) --- @param _src SourceFile -- unused (kept for the per_source shape)
--- @param pipe_ctx PipeCtx --- @param pipe_ctx PipeCtx
--- @param findings Findings --- @param findings Findings
@@ -235,10 +227,8 @@ local function check_semantic_reg_defaults(_src, pipe_ctx, findings)
end end
end end
--- Check: atom_reg_types(R_X, <type>) entries must point to a register declared in `pipe_ctx.register_alias_registry`, with a type name found in `pipe_ctx.type_name_registry`. --- Check: atom_reg_types(R_X, <type>) entries target an alias in `pipe_ctx.register_alias_registry` and a type in `pipe_ctx.type_name_registry`.
--- The alias ident `R_<n>` now encodes the GPR identity only for entries that are explicitly opted in via the bare `atom_reg` marker. --- A bare `atom_reg` marker opts the `R_<n>` alias into GPR identity; references to R_T0..R_T3 require the same explicit marker.
--- R_T0..R_T3 are intentionally NOT auto-included (per the prototype principle: no auto-include of wave-context; explicit opt-in only).
--- The check fires for any R_T0..R_T3 reference that hasn't been opted in via `#define atom_reg`.
--- @param _src SourceFile --- @param _src SourceFile
--- @param pipe_ctx PipeCtx --- @param pipe_ctx PipeCtx
--- @param findings Findings --- @param findings Findings
@@ -269,7 +259,7 @@ local function check_atom_reg_types(_src, pipe_ctx, findings)
end end
end end
--- Check: atom_view(Binds_X) entries must reference a real Binds_* struct and that struct must declare at least one field. --- Check: atom_view(Binds_X) entries reference a Binds_* struct with at least one field.
--- @param _src SourceFile --- @param _src SourceFile
--- @param pipe_ctx PipeCtx --- @param pipe_ctx PipeCtx
--- @param findings Findings --- @param findings Findings
@@ -298,8 +288,7 @@ local function check_atom_view_layout(_src, pipe_ctx, findings)
end end
end end
--- Check: Binds_* structs may not have duplicate field names --- Check: Binds_* structs require unique field names because atom_view uses those names for typed-field lookup in gdb.
--- (they would defeat the typed-field name lookup that atom_view exposes in gdb).
--- @param _src SourceFile --- @param _src SourceFile
--- @param pipe_ctx PipeCtx --- @param pipe_ctx PipeCtx
--- @param findings Findings --- @param findings Findings
@@ -322,26 +311,30 @@ local function check_binds_no_duplicate_fields(_src, pipe_ctx, findings)
end end
end end
-- Check: skip-over markers must satisfy shape + placement constraints. -- Check: debug-skip markers must satisfy shape + placement constraints.
--- Walks the priority list once; at most one error is appended per marker so that a single source-level defect does not cascade into multiple findings. --- Walks the priority list once; each marker produces at most one error, so one source defect yields one finding.
--- Priority order (first defect wins): --- Priority order (first defect wins):
--- 1. has_parens == false -> requires parentheses: marker() --- 1. marker_kind ~= "atom_dbg_skip" -> legacy/renamed spelling (use `atom_dbg_skip`)
--- 2. args ~= "" -> takes no arguments --- 2. marker_kind == "atom_dbg_skip" AND has_parens -> parenthesized form (the marker is bare-only)
--- 3. superseded_by_marker_line -> duplicate marker (cite superseding line) --- 3. args ~= "" -> takes no arguments
--- 4. pending + no target_kind -> dangling (no following declaration) --- 4. superseded_by_marker_line -> duplicate marker (cite superseding line)
--- 5. unsupported target_kind -> marker precedes an unrelated declaration --- 5. pending + no target_kind -> dangling (no following declaration)
--- Valid markers before whole-atom / bare-component / proc-component declarations emit no error and remain in src.scan.skip_over.atoms / .components. --- 6. unsupported target_kind -> marker precedes an unrelated declaration
--- @param marker SkipOverMarker --- Valid markers stamp `debug_skip` on whole-atom, bare-component, and proc-component declaration records in scan_source.lua.
--- @param marker DebugSkipMarker
--- @param _pipe_ctx PipeCtx -- unused today; kept for plex-shape consistency with per_annot --- @param _pipe_ctx PipeCtx -- unused today; kept for plex-shape consistency with per_annot
--- @param findings Findings --- @param findings Findings
local function check_skip_marker(marker, _pipe_ctx, findings) local function check_skip_marker(marker, _pipe_ctx, findings)
local kind = marker.marker_kind local kind = marker.marker_kind
local line = marker.marker_line local line = marker.marker_line
if not marker.has_parens then -- Left `scan.debug_skip_markers` with production records for `atom_dbg_skip` only; other identifiers take the walker's unrelated branch.
if marker.has_parens then
findings.errors[#findings.errors + 1] = { findings.errors[#findings.errors + 1] = {
line = line, line = line,
msg = string.format("%s marker at line %d requires parentheses: marker()", kind, line), msg = string.format("%s marker at line %d must be bare; the parenthesized form is no longer accepted (use `atom_dbg_skip MipsAtom_(name) { ... }`)",
kind, line),
} }
return return
end end
@@ -386,11 +379,8 @@ end
--- Warn when a source references an unregistered alias. --- Warn when a source references an unregistered alias.
--- ---
--- R_TapePtr / R_AtomJmp / R_PrimCursor / R_FaceCursor / R_VertBase / R_OtBase are the context aliases opted in via `#define atom_reg` in lottes_tape.h. --- R_TapePtr, R_AtomJmp, R_PrimCursor, R_FaceCursor, R_VertBase, and R_OtBase opt in through `#define atom_reg` in lottes_tape.h.
--- A source referencing an unregistered R_X emits one pass-level info entry --- When a source uses an unregistered R_X, this check emits one pass-level info entry for that source and directs C-ABI register names to explicit alias registration.
--- (emitted only when at least one such rejection lands in this source) tells users where to look.
---
--- This check directs raw C-ABI register names to explicit alias registration.
--- @param _src SourceFile --- @param _src SourceFile
--- @param pipe_ctx PipeCtx --- @param pipe_ctx PipeCtx
--- @param findings Findings --- @param findings Findings
@@ -423,7 +413,7 @@ end
-- per_annot(annot, pipe_ctx, findings) -- runs once per AtomAnnotation -- per_annot(annot, pipe_ctx, findings) -- runs once per AtomAnnotation
-- post(pipe_ctx, findings) -- runs once after all per_annot calls complete (full-corpus aggregation) -- post(pipe_ctx, findings) -- runs once after all per_annot calls complete (full-corpus aggregation)
-- per_macro(macro, wc, findings) -- runs once per TAPE_WORDS / _Pragma macro declaration -- per_macro(macro, wc, findings) -- runs once per TAPE_WORDS / _Pragma macro declaration
-- per_skip_marker(marker, pipe_ctx, findings) -- runs once per src.scan.skip_over.markers entry -- per_skip_marker(marker, pipe_ctx, findings) -- runs once per src.scan.debug_skip_markers entry
-- --
-- Adding a new check = 1 row here + 1 function above. The `validate()` dispatch loop never needs editing. -- Adding a new check = 1 row here + 1 function above. The `validate()` dispatch loop never needs editing.
@@ -446,14 +436,8 @@ local CHECK_RULES = {
-- --
-- Pure check: read from src.scan, run validations, emit findings. The scan was done once upstream. -- Pure check: read from src.scan, run validations, emit findings. The scan was done once upstream.
--- Build the corpus-wide pipe_ctx ONCE per pass run. --- Builds one pass-wide pipe_ctx from the merged `corpus.*` registries and source-ordered `corpus.atom_infos`; per-source declarations and bodies remain in `src.scan`.
--- Reads the merged `corpus.*` registries (canonical cross-source lookups), --- The module ownership contract above requires callers to construct `ctx.shared.corpus` through `build_ctx`; the error message below enforces that gate.
--- and the corpus-wide `atom_infos` list (preserving source order + duplicates).
--- The corpus is the source of truth; per-source scans retain body / declaration
--- ownership via `src.scan` and the per-source `atoms` / `atom_infos` projections.
---
--- Canonical ownership: a context without `ctx.shared.corpus` is rejected with an explicit canonical-corpus message.
--- No per-source fallback synthesis is performed; callers MUST construct a canonical ctx through `build_ctx`.
--- @param ctx PassCtx --- @param ctx PassCtx
--- @return PipeCtx --- @return PipeCtx
local function build_corpus_pipe_ctx(ctx) local function build_corpus_pipe_ctx(ctx)
@@ -464,9 +448,7 @@ local function build_corpus_pipe_ctx(ctx)
.. "no per-source fallback is supported)", 0) .. "no per-source fallback is supported)", 0)
end end
-- Corpus atom_infos preserves source-order + duplicates; -- `corpus.atom_infos` preserves source order and duplicates; I precompute counts here for `check_unique_annotation` and the per-source checks.
-- the per-check `check_unique_annotation` post-rule still flags duplicate annotation
-- names within this list. We pre-compute the annot_counts map here so the per_source checks can iterate it without re-walking.
local annot_counts = {} local annot_counts = {}
for _, info in ipairs(corpus.atom_infos or {}) do for _, info in ipairs(corpus.atom_infos or {}) do
if info and info.atom_name then if info and info.atom_name then
@@ -474,10 +456,9 @@ local function build_corpus_pipe_ctx(ctx)
end end
end end
-- The pipe_ctx views REFERENCE the corpus tables directly (no copies).
-- Every consumer of these fields observes mutations via the canonical corpus without independently mutable registry construction. -- Every consumer of these fields observes mutations via the canonical corpus without independently mutable registry construction.
return { return {
-- Cross-source lookup tables (canonical corpus projections). -- Cross-source lookup tables from corpus.
register_alias_registry = corpus.register_alias_registry or {}, register_alias_registry = corpus.register_alias_registry or {},
type_name_registry = corpus.type_name_registry or {}, type_name_registry = corpus.type_name_registry or {},
atom_views = corpus.atom_views or {}, atom_views = corpus.atom_views or {},
@@ -491,8 +472,7 @@ local function build_corpus_pipe_ctx(ctx)
annot_counts = annot_counts, annot_counts = annot_counts,
-- Corpus-wide collisions (recorded by scan_source.merge_corpus_registries). -- Corpus-wide collisions (recorded by scan_source.merge_corpus_registries).
collisions = corpus.collisions or {}, collisions = corpus.collisions or {},
-- wc still consumed by check_macro_word_drift; reads from the canonical -- `check_macro_word_drift` reads `corpus.word_counts`, populated by word_count_eval.run.
-- `corpus.word_counts` table (built by word_count_eval.run).
word_counts = corpus.word_counts or {}, word_counts = corpus.word_counts or {},
} }
end end
@@ -500,9 +480,10 @@ end
--- Validate one source against its pre-scanned SourceScan payload + the corpus-wide pipe_ctx. --- Validate one source against its pre-scanned SourceScan payload + the corpus-wide pipe_ctx.
--- @param ctx PassCtx --- @param ctx PassCtx
--- @param src SourceFile --- @param src SourceFile
--- @param corpus_pipe_ctx PipeCtx -- built once per pass from corpus registries --- @param corpus_pipe_ctx PipeCtx|nil -- built once per pass from corpus registries; nil builds the same projection here.
--- @return AnnotatedResult --- @return AnnotatedResult
local function validate(ctx, src, corpus_pipe_ctx) local function validate(ctx, src, corpus_pipe_ctx)
corpus_pipe_ctx = corpus_pipe_ctx or build_corpus_pipe_ctx(ctx)
local scan = src.scan local scan = src.scan
-- Project the pre-scanned atoms to the AtomEntry shape this pass needs. -- Project the pre-scanned atoms to the AtomEntry shape this pass needs.
@@ -528,11 +509,7 @@ local function validate(ctx, src, corpus_pipe_ctx)
} }
end end
-- Build the per-source pipe_ctx (Fleury: expose structure). -- Build a per-source pipe_ctx: shared lookups come from `corpus_pipe_ctx`, while declarations, bodies, types, views, defaults, and occurrences come from `src.scan`.
-- Cross-source visibility comes from `corpus_pipe_ctx`;
-- per-source declaration / body ownership comes from `src.scan`.
-- pipe_ctx.types / pipe_ctx.atom_views / pipe_ctx.seen_defaults / pipe_ctx.type_occurrences
-- are projected from the per-source scan so the per_source check rules can iterate the source-local occurrences.
local seen_defaults = {} local seen_defaults = {}
for reg, _ in pairs(scan.types or {}) do for reg, _ in pairs(scan.types or {}) do
seen_defaults[reg] = (seen_defaults[reg] or 0) + 1 seen_defaults[reg] = (seen_defaults[reg] or 0) + 1
@@ -552,8 +529,7 @@ local function validate(ctx, src, corpus_pipe_ctx)
seen_defaults = seen_defaults, seen_defaults = seen_defaults,
atom_infos_list = atom_infos_list, atom_infos_list = atom_infos_list,
binds_list = scan.binds or {}, binds_list = scan.binds or {},
-- Source-derived registries: still populated from the scan payload as a convenience for callers that want source-local visibility. -- See the module ownership contract; these shared lookup tables come from corpus_pipe_ctx.
-- The canonical cross-source lookup tables live in corpus_pipe_ctx.
register_alias_registry = corpus_pipe_ctx.register_alias_registry, register_alias_registry = corpus_pipe_ctx.register_alias_registry,
type_name_registry = corpus_pipe_ctx.type_name_registry, type_name_registry = corpus_pipe_ctx.type_name_registry,
} }
@@ -564,9 +540,7 @@ local function validate(ctx, src, corpus_pipe_ctx)
-- Each check writes to the list appropriate for its severity. -- Each check writes to the list appropriate for its severity.
local findings = { errors = {}, warnings = {}, info = {} } local findings = { errors = {}, warnings = {}, info = {} }
-- Propagate parse-time errors from scan_source's atom_info parsing. -- Lift parse-time errors already recorded in scan_source's atom_info payload into this pass's findings list.
-- These are errors found in the atom_info(...) body itself (e.g., malformed args).
-- They are pre-existing in the scan payload — we just lift them into our findings list.
for _, a in ipairs(annots) do for _, a in ipairs(annots) do
if a.errors then if a.errors then
for _, msg in ipairs(a.errors) do for _, msg in ipairs(a.errors) do
@@ -590,11 +564,9 @@ local function validate(ctx, src, corpus_pipe_ctx)
if rule.post then rule.post(pipe_ctx, findings) end if rule.post then rule.post(pipe_ctx, findings) end
end end
-- Per-skip-marker rules. -- scan_source records each marker in scan.debug_skip_markers; this loop validates each record independently and emits at most one error per marker.
-- Each raw marker recorded by scan_source (in scan.skip_over.markers) is validated independently; -- Valid markers stamp `debug_skip = true` on the following atom or component declaration, which downstream consumers read directly.
-- the check emits at most one error per marker. local skip_markers = scan.debug_skip_markers or {}
-- Valid markers stay attached to scan.skip_over.atoms /.components for dwarf_injection.lua consumer.
local skip_markers = scan.skip_over and scan.skip_over.markers or {}
for _, marker in ipairs(skip_markers) do for _, marker in ipairs(skip_markers) do
for _, rule in ipairs(CHECK_RULES) do for _, rule in ipairs(CHECK_RULES) do
if rule.per_skip_marker then rule.per_skip_marker(marker, pipe_ctx, findings) end if rule.per_skip_marker then rule.per_skip_marker(marker, pipe_ctx, findings) end
@@ -640,7 +612,6 @@ end
--- Render `<dir_basename>.errors.h` with `#error` directives for every error found across all sources in the directory. --- Render `<dir_basename>.errors.h` with `#error` directives for every error found across all sources in the directory.
--- Empty directories (no errors, no atoms) produce no file. --- Empty directories (no errors, no atoms) produce no file.
local function emit_module_errors_h(ctx, dir_basename, atoms_count, errors, sources) local function emit_module_errors_h(ctx, dir_basename, atoms_count, errors, sources)
if ctx.dry_run then return nil end
if atoms_count == 0 and #errors == 0 then if atoms_count == 0 and #errors == 0 then
return nil return nil
end end
@@ -668,17 +639,6 @@ local function emit_module_errors_h(ctx, dir_basename, atoms_count, errors, sour
return out_path return out_path
end end
--- Stash aggregated per-module results for the report pass to consume.
local function emit_module_annotations_stub(ctx, dir, dir_basename, atoms_count)
ctx.flags = ctx.flags or {}
ctx.flags._annot_results = ctx.flags._annot_results or {}
ctx.flags._annot_results[#ctx.flags._annot_results + 1] = {
dir = dir,
dir_basename = dir_basename,
atoms_count = atoms_count,
}
end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- M.run — orchestrator entry -- M.run — orchestrator entry
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -697,15 +657,12 @@ function M.run(ctx)
local errors = {} local errors = {}
local warnings = {} local warnings = {}
-- Build the corpus-wide pipe_ctx ONCE per pass run. -- Build the shared pipe_ctx once for this run; every validate() call sees the same cross-source registries.
-- The corpus owns the canonical cross-source registries; per-source scans retain body / declaration ownership. -- The corpus owns the canonical cross-source registries; per-source scans retain body / declaration ownership.
-- The pipe_ctx is shared across every validate() invocation in this M.run so cross-source visibility is constant.
local corpus_pipe_ctx = build_corpus_pipe_ctx(ctx) local corpus_pipe_ctx = build_corpus_pipe_ctx(ctx)
local corpus = ctx.shared.corpus local corpus = ctx.shared.corpus
-- Per-DIRECTORY (per-module) aggregation. -- Group `corpus.sources_by_dir` by module, validate every source in each bucket, and emit one errors.h per directory.
-- Group sources by `src.dir`, validate every source in the dir, then emit ONE errors.h per dir.
-- The corpus owns `sources_by_dir`; this pass reads the corpus bucket directly.
local by_dir = (corpus and corpus.sources_by_dir) or {} local by_dir = (corpus and corpus.sources_by_dir) or {}
for dir, dir_sources in pairs(by_dir) do for dir, dir_sources in pairs(by_dir) do
@@ -713,13 +670,9 @@ function M.run(ctx)
local dir_atoms = 0 local dir_atoms = 0
local dir_errors = {} local dir_errors = {}
local dir_warnings = {} local dir_warnings = {}
-- Per-source validate() results, cached for the report pass (it reads from this instead of re-validating each source).
ctx.flags = ctx.flags or {}
ctx.flags._annot_source_results = ctx.flags._annot_source_results or {}
for _, src in ipairs(dir_sources) do for _, src in ipairs(dir_sources) do
local result = validate(ctx, src, corpus_pipe_ctx) local result = validate(ctx, src, corpus_pipe_ctx)
result.source = src.path -- tag for downstream rendering result.source = src.path -- tag for downstream rendering
ctx.flags._annot_source_results[src.path] = result -- stash so report.lua reads from cache instead of re-running validate()
dir_atoms = dir_atoms + #result.atoms dir_atoms = dir_atoms + #result.atoms
for _, e in ipairs(result.errors) do for _, e in ipairs(result.errors) do
dir_errors[#dir_errors + 1] = { line = e.line, msg = e.msg, source = src.path } dir_errors[#dir_errors + 1] = { line = e.line, msg = e.msg, source = src.path }
@@ -735,8 +688,6 @@ function M.run(ctx)
if err_path then if err_path then
table.insert(outputs, { errors_h = err_path }) table.insert(outputs, { errors_h = err_path })
end end
emit_module_annotations_stub(ctx, dir, dir_basename, dir_atoms)
end end
return { outputs = outputs, errors = errors, warnings = warnings } return { outputs = outputs, errors = errors, warnings = warnings }
+48 -86
View File
@@ -1,23 +1,23 @@
--- passes/atoms_source_map.lua — Per-.word source-line map emitter for tape atoms. --- passes/atoms_source_map.lua — Per-.word source-line map emitter for tape atoms.
--- ---
--- Reads the canonical `atom.paths` projection produced by the upstream `emission_model` pass. --- Writer: this pass, given `atom.paths` (the per-atom mutable surface owned by `emission_model`). Readers:
--- The ordered `items` stream, dense `word_events`, and `invocations` views are the only semantic inputs to this pass; --- `passes/dwarf_injection.lua` (synthesizes DW_TAG_inlined_subroutine + per-word line program rows) and the gdb-runtime
--- it emits one `WORD N LINE L TEXT T` line per emitted `.word`. --- wrapper at `scripts/gdb/gdb_tape_atoms.gdb` (loads the source map via `source <path>`).
--- ---
--- **Two output forms** (per the workspace's per-emission-form pattern from --- Inputs from `atom.paths`: the ordered `items` stream, dense `word_events`, `invocations` views. Outputs: one
--- `guide_metaprogram_ssdl.md`): --- `WORD N LINE L TEXT T` line per emitted `.word`, plus the per-word provenance form that DWARF synthesis consumes.
--- 1. **Canonical text form** — `<out_root>/<basename>.atoms.sourcemap.txt`. ---
--- Format-version-tagged for forward-compat. --- **Two output forms** (per the workspace's per-emission-form pattern from `guide_metaprogram_ssdl.md`):
--- Lives in `<out_root>/` (build/gen). --- 1. **Sourcemap.txt form** — `<out_root>/<basename>.atoms.sourcemap.txt`. Format-version-tagged for forward-compat.
--- Matches the convention used by `annotation.lua` (`<out_root>/<basename>.errors.h`) + `static_analysis.lua` (`<out_root>/<basename>.static_analysis.txt`). --- Lives in `<out_root>/` (build/gen). Mirrors the convention used by `annotation.lua`
--- (`<out_root>/<basename>.errors.h`) and `static_analysis.lua` (`<out_root>/<basename>.static_analysis.txt`).
--- Compile artifacts (`*.macs.h`, `*.offsets.h`) stay in `<source_dir>/gen/`. --- Compile artifacts (`*.macs.h`, `*.offsets.h`) stay in `<source_dir>/gen/`.
--- 2. **gdb-runtime form** — `<ctx.out_root>/gdb_tape_atoms_runtime.gdb` --- 2. **gdb-runtime form** — `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`. A pure gdb command script — addresses come
--- (pure gdb command script; addresses pre-computed via `nm`; the 9 user commands defined as `define ... end` blocks). --- from `nm`, the 9 user commands are static `define ... end` blocks. Emitted when `ctx.flags.gdb_runtime` is true
--- Emitted ONLY when `ctx.flags.gdb_runtime` is true AND `ctx.flags.elf_path` points to an existing ELF. --- AND `ctx.flags.elf_path` points to an existing ELF. Useful for `gdb-multiarch --without-python` users
--- The gdb runtime form lets `gdb-multiarch --without-python` users (the common case on Windows MinGW builds) --- (the common case on Windows MinGW builds) — `source <path>` loads it with no Python / Tcl / Guile required.
--- load the source-map data via `source <path>` — no Python/Tcl/Guile required.
--- ---
--- **Output format** (canonical text form): --- **Output format** (sourcemap.txt form):
--- ``` --- ```
--- # FORMAT_VERSION 1 --- # FORMAT_VERSION 1
--- # auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT --- # auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT
@@ -31,11 +31,9 @@
--- ENDATOM --- ENDATOM
--- ``` --- ```
--- ---
--- Marker records are zero-width in `atom.paths.items`; they do not appear in --- Marker records are zero-width in `atom.paths.items`, so they emit no WORD rows in the dense word view.
--- the dense word view and therefore emit no WORD rows.
--- ---
--- **Conventions:** tabs (1/level), EmmyLua annotations, no regex, --- **Conventions:** tabs (1/level), EmmyLua annotations, no regex, Lua 5.3 compatible.
--- Lua 5.3 compatible.
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Module-scope requires + package.path setup -- Module-scope requires + package.path setup
@@ -62,18 +60,16 @@ local FORMAT_VERSION = 1
--- @class AtomSourceMapCtx --- @class AtomSourceMapCtx
--- @field shared table -- `ctx.shared` --- @field shared table -- `ctx.shared`
--- @field shared.corpus table -- canonical source-order corpus --- @field shared.corpus table -- source-order registry; single writer is build_ctx
--- @field shared.word_counts table -- identity alias of `corpus.word_counts` --- @field shared.word_counts table
--- @field out_root string -- output root (e.g. "build/gen") --- @field out_root string -- output root (e.g. "build/gen")
--- @field dry_run boolean -- if true, compute but don't write
--- @field flags table -- `ctx.flags`; reads `flags.gdb_runtime` + `flags.elf_path` --- @field flags table -- `ctx.flags`; reads `flags.gdb_runtime` + `flags.elf_path`
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Canonical atom-path renderers -- Atom-path renderers
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- Join canonical words to canonical word items. `items` supplies the ordered --- Join word boundaries (from `items`) to per-word call text + source lines (from `word_events`).
--- word boundaries, while `word_events` supplies call text and source lines.
--- @param atom table --- @param atom table
--- @return table[], integer --- @return table[], integer
local function canonical_word_entries(atom) local function canonical_word_entries(atom)
@@ -100,11 +96,11 @@ local function canonical_word_entries(atom)
return entries, #events return entries, #events
end end
--- Render one atom's provenance stanza. Format 1 remains: --- Render one atom's provenance stanza. Format 1 line shapes:
--- `WORD N CALL <src-path>:<src-line> MACRO <name> "<def-path>:<def-line>" BODY <line>` --- `WORD N CALL <src-path>:<src-line> MACRO <name> "<def-path>:<def-line>" BODY <line>` (component invocation)
--- `WORD N CALL <src-path>:<src-line> RAW` --- `WORD N CALL <src-path>:<src-line> RAW` (raw `.word` outside any mac_* component)
--- Component identity comes from the canonical outermost invocation record; --- Component identity comes from the outermost invocation record; the count-table lookup confirms the component was
--- the count-table lookup is the canonical component declaration witness. --- declared in `corpus.word_counts` (populated by word_count_eval + components passes).
--- @param src table --- @param src table
--- @param atom table --- @param atom table
--- @param wc table -- identity alias of corpus.word_counts --- @param wc table -- identity alias of corpus.word_counts
@@ -163,7 +159,7 @@ local function render_provenance(src, wc)
return table.concat(lines, "\n") .. "\n" return table.concat(lines, "\n") .. "\n"
end end
--- Render one atom's stanza for the canonical text form (ATOM header line, N WORD lines, ENDATOM marker). --- Render one atom's stanza for the sourcemap.txt form (ATOM header line, N WORD lines, ENDATOM marker).
--- Returns (lines, total_words). --- Returns (lines, total_words).
--- @param src table --- @param src table
--- @param atom table --- @param atom table
@@ -184,8 +180,8 @@ local function emit_atom_stanza(src, atom)
return lines, total return lines, total
end end
--- Render the full source map file content for one source (one .atoms.sourcemap.txt per source). --- Render the full source map file content for one source (one .atoms.sourcemap.txt per source). Mirrors offsets.lua's
--- Mirrors offsets.lua's `project_atoms` shape: scan.atoms + scan.raw_atoms, no kind filter. --- `project_atoms` shape: scan.atoms + scan.raw_atoms, no kind filter.
--- @param src table --- @param src table
--- @param wc table --- @param wc table
--- @return string --- @return string
@@ -220,8 +216,7 @@ local function gdb_escape(s)
return (s:gsub("\\", "\\\\"):gsub('"', '\\"')) return (s:gsub("\\", "\\\\"):gsub('"', '\\"'))
end end
--- Build the list of atoms with addresses + word entries. --- Build the list of atoms with addresses + word entries. Shared helper for the gdb-runtime file emission.
--- Shared helper for the gdb-runtime file emission.
--- @param ctx PassCtx --- @param ctx PassCtx
--- @return table[] -- list of {idx, name, src_path, file_base, addr, size_bytes, words, entries} --- @return table[] -- list of {idx, name, src_path, file_base, addr, size_bytes, words, entries}
local function build_atom_table(ctx) local function build_atom_table(ctx)
@@ -257,12 +252,12 @@ local function build_atom_table(ctx)
return matched return matched
end end
--- Append the 9 gdb command definitions to `lines`. Pure gdb scripting no Python, no Tcl, no Guile required. --- Append the 9 gdb command definitions to `lines`. Pure gdb scripting — addresses come from `nm`, the convenience
--- **Fully hardcoded per-atom** because gdb doesn't do nested `$` substitution in var names --- vars set in `emit_gdb_runtime` provide printf args, and each command is a static sequence of `printf` / `tbreak` /
--- `$__atom_name_$__i` inside a `while` loop is treated as one literal identifier, not a concat. --- `if ... end` blocks. The Lua pass emits N atoms' worth of lines; runtime iteration is gdb's job.
--- ---
--- Each command is a static sequence of `printf` / `tbreak` / `if ... end` blocks. --- Why hardcoded per-atom: gdb's `$` substitution doesn't concat inside var names — `$__atom_name_$__i` in a `while`
--- The Lua pass emits N atoms' worth of lines — no runtime iteration. --- loop resolves to one literal identifier, not `name_i`. Compile-time emission is the only path.
--- @param lines table -- output line buffer (mutated in place) --- @param lines table -- output line buffer (mutated in place)
--- @param matched table -- list of atom records from `build_atom_table` --- @param matched table -- list of atom records from `build_atom_table`
local function append_gdb_commands(lines, matched) local function append_gdb_commands(lines, matched)
@@ -391,31 +386,6 @@ local function append_gdb_commands(lines, matched)
lines[#lines + 1] = "end" lines[#lines + 1] = "end"
lines[#lines + 1] = "" lines[#lines + 1] = ""
-- ── show_c2 ──
-- GTE data regs (COP2). pcsx-redux's gdb stub doesn't expose COP2 (only 72 regs: 32 GPR + COP0 + FPR).
-- curl http://localhost:8080/api/v1/lua/gte
-- We keep the command definition as a stub that points the user at the plugin.
lines[#lines + 1] = "define show_c2"
lines[#lines + 1] = ' echo "[gdb_tape_atoms] show_c2: gdb stub does not expose COP2 in this build."'
lines[#lines + 1] = ' echo "[gdb_tape_atoms] Use scripts/pcsx_debug_helper.zip + curl http://localhost:8080/api/v1/lua/gte"'
lines[#lines + 1] = ' echo "[gdb_tape_atoms] (or pcsx-redux Debug > Registers window for a native view)"'
lines[#lines + 1] = "end"
lines[#lines + 1] = "document show_c2"
lines[#lines + 1] = " Stub. The gdb stub in this pcsx-redux build does not expose COP2 regs."
lines[#lines + 1] = " For GTE data + control state, use the pcsx_debug_helper Lua plugin or the"
lines[#lines + 1] = " pcsx-redux Debug > Registers window."
lines[#lines + 1] = "end"
lines[#lines + 1] = ""
-- ── show_c2ctl ──
lines[#lines + 1] = "define show_c2ctl"
lines[#lines + 1] = ' echo "[gdb_tape_atoms] show_c2ctl: see show_c2 for the same workaround."'
lines[#lines + 1] = "end"
lines[#lines + 1] = "document show_c2ctl"
lines[#lines + 1] = " Stub. Same workaround as show_c2."
lines[#lines + 1] = "end"
lines[#lines + 1] = ""
-- ── wave_ctx ── -- ── wave_ctx ──
lines[#lines + 1] = "define wave_ctx" lines[#lines + 1] = "define wave_ctx"
lines[#lines + 1] = ' printf "$t4 = R_FaceCursor 0x%08x\\n", $t4' lines[#lines + 1] = ' printf "$t4 = R_FaceCursor 0x%08x\\n", $t4'
@@ -428,9 +398,9 @@ local function append_gdb_commands(lines, matched)
lines[#lines + 1] = "end" lines[#lines + 1] = "end"
end end
--- Emit the gdb-runtime file (post-link). Pure gdb scripting — no Python. --- Emit the gdb-runtime file (post-link). Pure gdb scripting — addresses come from `mipsel-none-elf-nm -S`, get embedded
--- Reads ELF addresses via `mipsel-none-elf-nm -S`, embeds them in `<ctx.out_root>/gdb_tape_atoms_runtime.gdb` --- in `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`, and load via `set $var = ...` + `define ... end` blocks at gdb
--- so gdb loads the data via `set $var = ...` + `define ... end` blocks at source-time. --- source-time.
--- @param ctx PassCtx --- @param ctx PassCtx
local function emit_gdb_runtime(ctx) local function emit_gdb_runtime(ctx)
if not (ctx.flags and ctx.flags.gdb_runtime) then return end if not (ctx.flags and ctx.flags.gdb_runtime) then return end
@@ -489,12 +459,9 @@ local function emit_gdb_runtime(ctx)
lines[#lines + 1] = 'printf "[gdb_tape_atoms] runtime loaded %d atoms from %s\\n", $__atom_count, $__elf_path' lines[#lines + 1] = 'printf "[gdb_tape_atoms] runtime loaded %d atoms from %s\\n", $__atom_count, $__elf_path'
local out_path = ctx.out_root .. "/gdb_tape_atoms_runtime.gdb" local out_path = ctx.out_root .. "/gdb_tape_atoms_runtime.gdb"
if not ctx.dry_run then
duffle.ensure_dir(duffle.dirname(out_path)) duffle.ensure_dir(duffle.dirname(out_path))
duffle.write_file_lf(out_path, table.concat(lines, "\n") .. "\n") duffle.write_file_lf(out_path, table.concat(lines, "\n") .. "\n")
end -- io.stderr:write(string.format("[atoms_source_map] wrote %s (%d atoms)\n", out_path, #matched))
io.stderr:write(string.format(
"[atoms_source_map] wrote %s (%d atoms)\n", out_path, #matched))
end end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -503,10 +470,10 @@ end
local M = {} local M = {}
--- Pass entry: emit one `<out_root>/<basename>.atoms.sourcemap.txt` per source file that contains at least one `MipsAtom_(name)` / `MipsCode code_<name>` declaration. --- Pass entry. For each source that declares at least one `MipsAtom_(name)` / `MipsCode code_<name>`, emit two files
--- Also emits `<out_root>/<basename>.atoms.provenance.txt`: --- in `<out_root>/`: `<basename>.atoms.sourcemap.txt` (per-word call-site map) and `<basename>.atoms.provenance.txt`
--- per-.word provenance with `mac_X(...)` component resolution back to the component's definition file:line + the per-word body line. --- (per-word definition + body line, resolved via the outermost `mac_X(...)` invocation). When `ctx.flags.gdb_runtime`
--- Optionally also emit `<ctx.out_root>/gdb_tape_atoms_runtime.gdb` when `ctx.flags.gdb_runtime` is true. --- is true and `ctx.flags.elf_path` exists, also emit the post-link gdb script `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`.
--- @param ctx PassCtx --- @param ctx PassCtx
--- @return PassResult --- @return PassResult
function M.run(ctx) function M.run(ctx)
@@ -519,8 +486,7 @@ function M.run(ctx)
error("atoms_source_map.run requires ctx.shared.corpus.source_order (canonical corpus).", 0) error("atoms_source_map.run requires ctx.shared.corpus.source_order (canonical corpus).", 0)
end end
-- Word counts are owned by `corpus.word_counts`. -- Word counts come from `corpus.word_counts` (populated by word_count_eval + components passes).
-- The canonical owner is `corpus.word_counts` (populated by `passes/word_count_eval.lua` + `passes/components.lua`).
local wc = corpus.word_counts or {} local wc = corpus.word_counts or {}
if not next(wc) then if not next(wc) then
warnings[#warnings + 1] = { warnings[#warnings + 1] = {
@@ -529,11 +495,13 @@ function M.run(ctx)
} }
end end
-- Always emit the canonical text form (per-source). -- Always emit the text form (per-source).
for _, src in ipairs(corpus.source_order) do for _, src in ipairs(corpus.source_order) do
local has_projection = false local has_projection = false
for _, atom in ipairs((src.scan or {}).atoms or {}) do for _, atom in ipairs((src.scan or {}).atoms or {}) do
if atom.paths then has_projection = true; break end if (atom.kind == "atom" or atom.kind == "raw_atom") and atom.paths then
has_projection = true; break
end
end end
if not has_projection then if not has_projection then
for _, atom in ipairs((src.scan or {}).raw_atoms or {}) do for _, atom in ipairs((src.scan or {}).raw_atoms or {}) do
@@ -542,21 +510,15 @@ function M.run(ctx)
end end
if has_projection then if has_projection then
local basename = duffle.basename_no_ext(src.path) local basename = duffle.basename_no_ext(src.path)
-- (1) atoms.sourcemap.txt — format-1 per-word call-site map. -- (1) atoms.sourcemap.txt — format-1 per-word call-site map.
local sourcemap_path = ctx.out_root .. "/" .. basename .. ".atoms.sourcemap.txt" local sourcemap_path = ctx.out_root .. "/" .. basename .. ".atoms.sourcemap.txt"
local sourcemap_body = render_source_map(src) local sourcemap_body = render_source_map(src)
-- (2) atoms.provenance.txt — format-1 per-word definition/body map. -- (2) atoms.provenance.txt — format-1 per-word definition/body map.
local prov_path = ctx.out_root .. "/" .. basename .. ".atoms.provenance.txt" local prov_path = ctx.out_root .. "/" .. basename .. ".atoms.provenance.txt"
local prov_body = render_provenance(src, wc) local prov_body = render_provenance(src, wc)
if not ctx.dry_run then
duffle.ensure_dir(duffle.dirname(sourcemap_path)) duffle.ensure_dir(duffle.dirname(sourcemap_path))
duffle.write_file_lf(sourcemap_path, sourcemap_body) duffle.write_file_lf(sourcemap_path, sourcemap_body)
duffle.write_file_lf(prov_path, prov_body) duffle.write_file_lf(prov_path, prov_body)
end
outputs[#outputs + 1] = { kind = "report", path = sourcemap_path } outputs[#outputs + 1] = { kind = "report", path = sourcemap_path }
outputs[#outputs + 1] = { kind = "report", path = prov_path } outputs[#outputs + 1] = { kind = "report", path = prov_path }
end end
+54 -119
View File
@@ -1,25 +1,16 @@
--- passes/components.lua — Component-macro header generator. --- passes/components.lua — Component-macro header generator.
--- ---
--- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`) --- Ownership: `corpus.word_counts`, `corpus.components`, and `corpus.component_body_index`.
--- for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations, then does per-source backward lookups --- Scanner owns `declaration_comment` and `debug_skip` on each declaration record; this pass projects both forward.
--- for the function-args string (from the preceding `FI_ MipsAtom ac_X(...)` function declaration)
--- and the preceding comment block (for LSP/IntelliSense signature docs).
--- ---
--- Emits a per-directory `<dir_basename>.macs.h` containing one `#define mac_X(sig) \` macro per component + `WORD_COUNT(mac_X, N)` --- Reads the pre-scanned SourceScan payload from `duffle.scan_source` for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations,
--- entries for downstream offset computation. --- then resolves the function-args string from the preceding `FI_ MipsAtom ac_X(...)` declaration via a backward walk.
---
--- Emits one `<dir_basename>.macs.h` per source with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
--- ---
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex, --- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
--- Lua 5.3 compatible. --- Lua 5.3 compatible.
--- @class Component
--- @field name string
--- @field body string
--- @field args string|nil
--- @field line integer
--- @field comment string|nil
--- @class M
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Module-scope requires + package.path setup -- Module-scope requires + package.path setup
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -31,7 +22,6 @@
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module. -- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
local word_count_eval = require("word_count_eval")
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Constants -- Constants
@@ -69,12 +59,10 @@ local GEN_SUBDIR = "gen"
--- @field sources SourceFile[] -- all source files in the build --- @field sources SourceFile[] -- all source files in the build
--- @field metadata_path string -- path to word_count.metadata.h --- @field metadata_path string -- path to word_count.metadata.h
--- @field shared table -- cross-pass shared state --- @field shared table -- cross-pass shared state
--- @field shared.word_counts table<string, integer> -- populated by word-counts + components
--- @field out_root string -- output root (e.g. "build/gen") --- @field out_root string -- output root (e.g. "build/gen")
--- @field project_root string -- project root (e.g. "code/") --- @field project_root string -- project root (e.g. "code/")
--- @field upstream table<string, table> -- per-pass upstream outputs --- @field upstream table<string, table> -- per-pass upstream outputs
--- @field flags table -- CLI flags --- @field flags table -- CLI flags
--- @field dry_run boolean -- if true, compute but don't write
--- @field verbose boolean -- log diagnostic info --- @field verbose boolean -- log diagnostic info
--- @class PassResult --- @class PassResult
@@ -87,7 +75,9 @@ local GEN_SUBDIR = "gen"
--- @field body string -- brace-delimited body (without the braces) --- @field body string -- brace-delimited body (without the braces)
--- @field args string|nil -- function-args string (function form only) --- @field args string|nil -- function-args string (function form only)
--- @field line integer -- source line of the declaration --- @field line integer -- source line of the declaration
--- @field comment string|nil -- preceding `/* */` or `//` comment block (signature doc) --- @field comment string|nil -- scanner-owned `declaration_comment`; the components pass reads it from the scanner record
--- @field kind string -- "comp_bare" | "comp_proc"
--- @field debug_skip boolean -- mirror of `a.debug_skip` (scanner-owned); true iff a bare `atom_dbg_skip` marker immediately preceded the declaration
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Local helpers (file I/O + path normalization) -- Local helpers (file I/O + path normalization)
@@ -96,7 +86,11 @@ local GEN_SUBDIR = "gen"
local M = {} local M = {}
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Back-walk helpers (composed into the 2 entry points below: find_function_args_for + preceding_comment_block) -- Back-walk helpers (composed into the entry point below: find_function_args_for)
--
-- Only the function-args lookup for proc components occurs here.
-- The preceding-comment walk occur in `scan_source.lua` — `a.declaration_comment` carries the resolved comment,
-- so this file reads it forward rather than re-walking the source.
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- Find the args of the function declaration that immediately precedes a `MipsAtomComp_Proc_` invocation of the given name. --- Find the args of the function declaration that immediately precedes a `MipsAtomComp_Proc_` invocation of the given name.
@@ -143,79 +137,6 @@ local function find_function_args_for(source, name, before_pos)
return inner return inner
end end
--- Find the contiguous comment block immediately preceding `pos` in `source`.
--- Returns the comment text (with the `/* */` or `//` markers preserved) or an empty string if no comment is adjacent.
---
--- Used to copy signature comments from the source declaration (`MipsAtomComp_` / `MipsAtomComp_Proc_` / function decl)
--- over to the generated `mac_X` macro, so LSP/IntelliSense displays the args doc.
--- @param source string
--- @param pos integer
--- @return string
local function preceding_comment_block(source, pos)
local scan_pos = pos
local pieces = {}
while true do
-- Skip whitespace (space/tab/newline/CR) backward from `scan_pos`,
-- returning the position of the first non-whitespace char.
local non_ws = scan_pos - 1
while non_ws > 0 do
local ch = source:sub(non_ws, non_ws)
if ch == " " or ch == "\t" or ch == "\n" or ch == "\r" then
non_ws = non_ws - 1
else
break
end
end
if non_ws == 0 then break end
local is_block_close = non_ws >= 2 and source:sub(non_ws - 1, non_ws) == "*/"
local is_line_end = source:sub(non_ws, non_ws) == "\n" or source:sub(non_ws, non_ws) == "\r"
if is_block_close then
-- Find the opening `/*` for a block comment whose `*/` ends at `non_ws`.
-- Walk back from `non_ws` over `/*` candidates.
local prefix = source:sub(1, non_ws - 1)
local open_at = nil
for scan = #prefix - 1, 1, -1 do
if prefix:sub(scan, scan + 1) == "/*" then
open_at = scan
break
end
end
if not open_at then break end
-- Walk back from `open_at` over leading spaces + tabs to include the indentation before the `/*`.
local block_start = open_at
while block_start > 1 do
local ch = source:sub(block_start - 1, block_start - 1)
if ch == " " or ch == "\t" then
block_start = block_start - 1
else
break
end
end
table.insert(pieces, 1, source:sub(block_start, non_ws))
scan_pos = block_start
elseif is_line_end then
-- Walk back from `non_ws` to the start of the source line (the most recent `\n` or position 1).
local line_start = non_ws
while line_start > 1 and source:sub(line_start - 1, line_start - 1) ~= "\n" do
line_start = line_start - 1
end
local line = source:sub(line_start, non_ws)
if line:sub(1, 2) == "//" then
table.insert(pieces, 1, line)
scan_pos = line_start - 1
else
break
end
else
break
end
end
if #pieces == 0 then return "" end
return table.concat(pieces, "\n")
end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Argument-name extraction -- Argument-name extraction
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -247,7 +168,7 @@ local function extract_arg_names(args_str)
local ident_start = ident_end local ident_start = ident_end
while ident_start > 0 do while ident_start > 0 do
local ch = trimmed:sub(ident_start, ident_start) local ch = trimmed:sub(ident_start, ident_start)
if duffle.is_alnum(ch) or ch == "_" then if duffle.is_alnum_byte(string.byte(ch)) or ch == "_" then
ident_start = ident_start - 1 ident_start = ident_start - 1
else else
break break
@@ -267,8 +188,12 @@ end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- Project pre-scanned MipsAtomComp_ / MipsAtomComp_Proc_ entries into Component shape. --- Project pre-scanned MipsAtomComp_ / MipsAtomComp_Proc_ entries into Component shape.
--- Does per-source backward lookups for args (preceding function decl) and comment (preceding comment block). --- Reads the scanner-owned `declaration_comment` (resolved by scan_source.lua, skipping backward across an associated bare `atom_dbg_skip` marker when present).
--- Per-source backward lookups remain in place only for the function `args` of proc components.
--- That lookup is unique to components.lua and stays separate from the declaration-comment walk.
--- Carries `body_tokens` forward from scan-source so word_count_rec reads from the precomputed table instead of calling duffle.tokenize_body again. --- Carries `body_tokens` forward from scan-source so word_count_rec reads from the precomputed table instead of calling duffle.tokenize_body again.
--- Carries the scanner-owned `debug_skip` flag forward so the generated projection can emit `/* atom_dbg_skip */`
--- before the authored comment and so `update_canonical_components` can mirror the same field onto `corpus.components[name]`.
--- @param source string -- the full source text (needed for backward lookups) --- @param source string -- the full source text (needed for backward lookups)
--- @param scan table -- SourceScan from duffle.scan_source --- @param scan table -- SourceScan from duffle.scan_source
--- @return Component[] --- @return Component[]
@@ -277,7 +202,9 @@ local function project_components(source, scan)
for _, a in ipairs(scan.atoms) do for _, a in ipairs(scan.atoms) do
if a.kind == "comp_bare" or a.kind == "comp_proc" then if a.kind == "comp_bare" or a.kind == "comp_proc" then
local args = find_function_args_for(source, a.raw_name, a.ident_pos) local args = find_function_args_for(source, a.raw_name, a.ident_pos)
local comment = preceding_comment_block(source, a.ident_pos) -- Comment ownership: scan_source.lua stamps `declaration_comment` on the record by walking backward past any associated bare marker.
-- The pass reads `declaration_comment` directly.
local comment = a.declaration_comment or ""
out[#out + 1] = { out[#out + 1] = {
line = a.line, line = a.line,
name = a.name, name = a.name,
@@ -287,6 +214,7 @@ local function project_components(source, scan)
args = args, args = args,
comment = comment, comment = comment,
kind = a.kind, -- "comp_bare" | "comp_proc"; provenance emitter reads this. kind = a.kind, -- "comp_bare" | "comp_proc"; provenance emitter reads this.
debug_skip = a.debug_skip == true,
} }
end end
end end
@@ -471,6 +399,9 @@ end
--- Build the list of lines for one component --- Build the list of lines for one component
--- (signature comment, `#define mac_X(...)` line with backslash-continued tokens, then `WORD_COUNT(mac_X, N)` entry). --- (signature comment, `#define mac_X(...)` line with backslash-continued tokens, then `WORD_COUNT(mac_X, N)` entry).
--- For skipped components, a `/* atom_dbg_skip */` marker comment is emitted immediately before the authored comment block.
--- The marker is a single line, the comment comes next, and the `#define` line follows. The `debug_skip` stamp is scanner-owned
--- (`a.debug_skip == true` on the declaration record); the components pass projects it directly.
--- @param c Component --- @param c Component
--- @param components Component[] --- @param components Component[]
--- @param wc table<string, integer> --- @param wc table<string, integer>
@@ -478,6 +409,13 @@ end
local function build_component_lines(c, counts) local function build_component_lines(c, counts)
local lines = {} local lines = {}
-- Marker comment: emitted once for every skipped component.
-- The marker is scanner-owned (declared by `atom_dbg_skip` immediately before the declaration in the source);
-- the components pass projects `c.debug_skip` and emits the marker as a generated comment.
if c.debug_skip then
lines[#lines + 1] = "/* atom_dbg_skip */"
end
if c.comment and c.comment ~= "" then if c.comment and c.comment ~= "" then
for _, line in ipairs(split_comment_lines(c.comment)) do for _, line in ipairs(split_comment_lines(c.comment)) do
lines[#lines + 1] = line lines[#lines + 1] = line
@@ -545,7 +483,6 @@ end
--- Emit a per-source `.macs.h` header with the `mac_X` macros + `WORD_COUNT` entries. --- Emit a per-source `.macs.h` header with the `mac_X` macros + `WORD_COUNT` entries.
--- Writes in BINARY mode so LF line endings are preserved (the git blob is LF; Windows text-mode would emit CRLF and break the byte-identical diff). --- Writes in BINARY mode so LF line endings are preserved (the git blob is LF; Windows text-mode would emit CRLF and break the byte-identical diff).
--- Honors `ctx.dry_run`: prints the intended path but does not write the file.
--- @param ctx PassCtx --- @param ctx PassCtx
--- @param src SourceFile --- @param src SourceFile
--- @param components Component[] --- @param components Component[]
@@ -563,11 +500,6 @@ local function emit_component_macros_h(ctx, src, components, counts)
end end
local content = table.concat(lines, "\n") .. "\n" local content = table.concat(lines, "\n") .. "\n"
if ctx.dry_run then
print(string.format(" -> %s (dry-run)", out_path))
return out_path
end
duffle.ensure_dir(out_dir) duffle.ensure_dir(out_dir)
duffle.write_file_lf(out_path, content) duffle.write_file_lf(out_path, content)
print(string.format(" -> %s", out_path)) print(string.format(" -> %s", out_path))
@@ -578,9 +510,9 @@ end
-- Pass entry -- Pass entry
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- (internal) Extend the canonical `corpus.word_counts` with this source's component macros so offsets sees them without re-reading the file. --- (internal) Extend `corpus.word_counts` with this source's component macros so offsets sees them without re-reading the file.
--- First declaration wins: a later caller's count is dropped (the existing entry from the first source is preserved). --- First declaration wins: a later caller's count is dropped (the existing entry from the first source is preserved).
--- @param corpus table -- the canonical corpus --- @param corpus table -- the corpus
--- @param components Component[] --- @param components Component[]
--- @param counts table<string, integer> -- precomputed word counts (from count_all_components) --- @param counts table<string, integer> -- precomputed word counts (from count_all_components)
local function update_canonical_word_counts(corpus, components, counts) local function update_canonical_word_counts(corpus, components, counts)
@@ -598,18 +530,21 @@ end
--- @field line integer -- definition source line (line of `MipsAtomComp_(ac_X)` / `MipsAtomComp_Proc_(ac_X, ...)`) --- @field line integer -- definition source line (line of `MipsAtomComp_(ac_X)` / `MipsAtomComp_Proc_(ac_X, ...)`)
--- @field path string -- absolute source path of the definition --- @field path string -- absolute source path of the definition
--- @field kind string -- "comp_bare" | "comp_proc" --- @field kind string -- "comp_bare" | "comp_proc"
--- @field debug_skip boolean -- mirror of the scanner-owned `a.debug_skip`; consumers read this directly
--- (internal) Populate the canonical `corpus.components` projection with this source's components-by-name map. --- (internal) Populate `corpus.components` with this source's components-by-name map.
--- First declaration wins; later declarations of the same bare name are dropped and recorded as a collision via `corpus.collisions` (kind = "component"). --- First declaration wins; later declarations of the same bare name are dropped and recorded as a collision via `corpus.collisions` (kind = "component").
--- The pass does NOT write to `ctx.shared.components` (ownership follows the canonical contract). --- The pass does NOT write to `ctx.shared.components` (ownership follows the canonical contract).
--- @param corpus table -- the canonical corpus --- The `debug_skip` field mirrors the scanner-owned declaration record (`c.debug_skip`).
--- No parallel skip map is built here; consumers that need the per-component skip state read `corpus.components[name].debug_skip` directly.
--- @param corpus table -- the corpus
--- @param src SourceFile --- @param src SourceFile
--- @param components Component[] --- @param components Component[]
local function update_canonical_components(corpus, src, components) local function update_canonical_components(corpus, src, components)
local rel_path = src.path:gsub("\\", "/") local rel_path = src.path:gsub("\\", "/")
for _, c in ipairs(components) do for _, c in ipairs(components) do
-- Keyed by bare name (e.g. `yield`, `load_tri_indices`). -- Keyed by bare name (e.g. `yield`, `load_tri_indices`).
-- The atoms_source_map pass looks up components by bare name from the canonical corpus; -- The atoms_source_map pass looks up components by bare name from the corpus;
-- `mac_` prefix lives at the call-site identifier and is stripped before lookup. -- `mac_` prefix lives at the call-site identifier and is stripped before lookup.
if corpus.components[c.name] == nil then if corpus.components[c.name] == nil then
corpus.components[c.name] = { corpus.components[c.name] = {
@@ -617,10 +552,11 @@ local function update_canonical_components(corpus, src, components)
line = c.line, line = c.line,
path = rel_path, path = rel_path,
kind = c.kind or "comp_bare", kind = c.kind or "comp_bare",
debug_skip = c.debug_skip == true,
} }
else else
-- A second declaration of the same bare name: record a typed collision so static-analysis + the report can surface it. -- A second declaration of the same bare name: record a typed collision so static-analysis + the report can surface it.
-- Identical-shape declarations (same path + line) do NOT record a collision (the first-wins entry already covers the case). -- Identical-shape declarations (same path + line) reuse the first-wins entry without a collision record.
local existing = corpus.components[c.name] local existing = corpus.components[c.name]
if existing.path ~= rel_path or existing.line ~= c.line then if existing.path ~= rel_path or existing.line ~= c.line then
local kind = c.kind or "comp_bare" local kind = c.kind or "comp_bare"
@@ -638,10 +574,10 @@ local function update_canonical_components(corpus, src, components)
end end
end end
--- (internal) Populate the canonical `corpus.component_body_index` projection with this source's body index entries. --- (internal) Populate `corpus.component_body_index` with this source's body index entries.
--- First declaration wins; later declarations are dropped (no separate collision record: the components collision is already surfaced by `update_canonical_components`). --- First declaration wins; later declarations are dropped (no separate collision record: the components collision is already surfaced by `update_canonical_components`).
--- The pass does NOT write to `ctx.shared.component_body_index` (the legacy corpus owns this projection). --- The pass writes to `corpus.component_body_index` only (the corpus owns this projection).
--- @param corpus table -- the canonical corpus --- @param corpus table -- the corpus
--- @param src SourceFile --- @param src SourceFile
--- @param components Component[] --- @param components Component[]
--- @param scan table -- the SourceScan payload (for line_of) --- @param scan table -- the SourceScan payload (for line_of)
@@ -668,13 +604,13 @@ function M.run(ctx)
local errors = {} local errors = {}
local warnings = {} local warnings = {}
-- Canonical-corpus ownership gate. -- Corpus ownership gate.
local corpus = ctx.shared and ctx.shared.corpus local corpus = ctx.shared and ctx.shared.corpus
if type(corpus) ~= "table" then if type(corpus) ~= "table" then
error("components.run requires ctx.shared.corpus (canonical corpus).", 0) error("components.run requires ctx.shared.corpus.", 0)
end end
if type(corpus.source_order) ~= "table" then if type(corpus.source_order) ~= "table" then
error("components.run requires ctx.shared.corpus.source_order (canonical corpus).", 0) error("components.run requires ctx.shared.corpus.source_order.", 0)
end end
if type(corpus.word_counts) ~= "table" then if type(corpus.word_counts) ~= "table" then
error("components.run requires ctx.shared.corpus.word_counts; " error("components.run requires ctx.shared.corpus.word_counts; "
@@ -682,25 +618,24 @@ function M.run(ctx)
.. "(see PASSES deps).", 0) .. "(see PASSES deps).", 0)
end end
-- Canonical projection ownership: -- Projection ownership:
-- * `corpus.word_counts["mac_"..name]` — current component count -- * `corpus.word_counts["mac_"..name]` — current component count
-- * `corpus.components[name]` — bare-name component definition -- * `corpus.components[name]` — bare-name component definition
-- * `corpus.component_body_index[name]` — body / line_of / source index -- * `corpus.component_body_index[name]` — body / line_of / source index
-- The pass does NOT mutate `ctx.shared.components` or `ctx.shared.component_body_index` -- The pass writes to the corpus only; consumers read from the corpus directly.
-- (ownership follows the canonical corpus; consumers read from the corpus directly).
for _, src in ipairs(corpus.source_order) do for _, src in ipairs(corpus.source_order) do
-- project_components reads from src.scan + does backward lookups on src.text -- project_components reads from src.scan + does backward lookups on src.text
local components = project_components(src.text, src.scan) local components = project_components(src.text, src.scan)
if #components > 0 then if #components > 0 then
-- Compute all component word counts once per source. -- Compute all component word counts once per source.
-- Use `corpus.word_counts` (the canonical count table) so the recursive lookup sees both authored-metadata entries -- Use `corpus.word_counts` so the recursive lookup sees both authored-metadata entries
-- (loaded by word_count_eval.run) AND same-source component entries (populated earlier in this loop by `update_canonical_word_counts`). -- (loaded by word_count_eval.run) AND same-source component entries (populated earlier in this loop by `update_canonical_word_counts`).
local counts = count_all_components(components, corpus.word_counts) local counts = count_all_components(components, corpus.word_counts)
local macs_path = emit_component_macros_h(ctx, src, components, counts) local macs_path = emit_component_macros_h(ctx, src, components, counts)
if macs_path then if macs_path then
outputs[#outputs + 1] = { macs_h = macs_path } outputs[#outputs + 1] = { macs_h = macs_path }
-- Populate the canonical projections AFTER disk emission (so the byte-identical `.macs.h` contract is preserved before any current-count mutation). -- Populate the projections AFTER disk emission (so the byte-identical `.macs.h` contract is preserved before any current-count mutation).
update_canonical_word_counts(corpus, components, counts) update_canonical_word_counts(corpus, components, counts)
update_canonical_components(corpus, src, components) update_canonical_components(corpus, src, components)
update_canonical_component_body_index(corpus, src, components, src.scan) update_canonical_component_body_index(corpus, src, components, src.scan)
File diff suppressed because it is too large Load Diff
+111 -66
View File
@@ -1,30 +1,30 @@
--- passes/emission_model.lua: Per-atom emission projection. --- passes/emission_model.lua: Per-atom emission projection.
--- ---
--- The `emission-model` pass owns `atom.paths` (the canonical per-atom mutable surface) --- The `emission-model` pass owns `atom.paths`, the canonical per-atom mutable surface for atoms and raw atoms with bodies in `ctx.shared.corpus.source_order`.
--- for every atom-with-body and every raw atom-with-body declared in `ctx.shared.corpus.source_order`. --- For each atom, the pass invokes `duffle.project_emission(body_text, component_index, word_counts, components)`.
--- For each such atom, the pass invokes `duffle.project_emission(body_text, component_index, word_counts)` --- It stores the ordered `items` stream plus the dense `word_events` / `markers` / `invocations` views on `atom.paths`.
--- and stores the ordered `items` stream plus the dense `word_events` / `markers` / `invocations` views on `atom.paths`.
--- ---
--- Public boundary: --- Public boundary:
--- * `M.run(ctx)` is the only entry point. --- * `M.run(ctx)` is the only entry point.
--- * The pass returns `{outputs = {}, errors = ..., warnings = ...}`. --- * The pass returns `{outputs = {}, errors = ..., warnings = ...}`.
--- Pass kind = `validation` → `PASS_KIND_STOP_ON_ERROR.validation` keeps build-stopping semantics (no policy change in this task). --- Pass kind = `validation` → `PASS_KIND_STOP_ON_ERROR.validation` preserves the existing build-stopping policy.
--- ---
--- Source-order discipline: --- Source-order discipline:
--- * `corpus.source_order` is the canonical ordering of source records. --- * `corpus.source_order` sets the source-record order.
--- * For each source, the pass iterates `src.scan.atoms` and `src.scan.raw_atoms` IN SOURCE ORDER, preserving declaration order. --- * Within each source, the pass visits `src.scan.atoms` and `src.scan.raw_atoms` in declaration order.
--- ---
--- Per-atom projection fields on `atom.paths`: --- Per-atom projection fields on `atom.paths`:
--- `tokens`, `line_in_body`, `items`, `word_events`, `markers`, `invocations`, `errors`, `warnings`. --- `tokens`, `line_in_body`, `items`, `word_events`, `markers`, `invocations`, `errors`, `warnings`.
--- The dense views are built from `items` only; the pass never re-walks source text or tokens. --- The construction walk appends `items` and derives each dense view from that ordered stream.
--- ---
--- Component expansion and construction validation: --- Component expansion and construction validation:
--- * known `mac_X(...)` calls recursively expand component bodies; --- * known `mac_X(...)` calls recursively expand component bodies;
--- * invocation records retain monotonic IDs, parent IDs, immediate call text, and the immutable outermost root call text; --- * invocation records retain monotonic IDs, parent IDs, immediate call text, and the immutable outermost root call text;
--- * component cycles retain balanced invocation boundaries and emit a `cycle` construction error without recursing indefinitely; --- * invocation construction stamps `debug_skip` from `corpus.components[name].debug_skip` at the construction site (no second pass, no source parse, no parallel lookup);
--- * component cycles close balanced invocation boundaries and emit a `cycle` construction error at the recursive edge;
--- * declared-vs-measured component word counts emit `count_mismatch` construction errors; opaque uncounted macros emit warnings. --- * declared-vs-measured component word counts emit `count_mismatch` construction errors; opaque uncounted macros emit warnings.
--- ---
--- The pass does NOT consult `_code_macros` / `_code_macro_bodies`. Those private tables are owned by `passes.scan_source` and stripped before this pass runs. --- `passes.scan_source` strips its private `_code_macros` / `_code_macro_bodies` tables before this pass runs.
local M = {} local M = {}
@@ -39,10 +39,27 @@ local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
-- ───────────────────────────────────────────────────────────────────────── -- ─────────────────────────────────────────────────────────────────────────
-- Convert the recursive walk's body-relative line numbers into physical source lines once. -- Convert the recursive walk's body-relative line numbers into physical source lines once.
-- Consumers read these canonical fields rather than rebuilding line state or tokenizing source again. -- The walker builds `line_of` from `body_text` and stamps body-relative line numbers (1..N) into `item.line` and `invocation.call_line`.
-- This function converts those values to physical source lines at the close site with the forwarded source `line_of` closure.
--
-- `call_line` discipline:
-- * ROOT invocations (`inv.parent_id == 0`) receive body-relative `call_line` values directly from `M.LineIndex(body_text)` in the walker.
-- The source `line_of` closure supplies physical lines at the close site, so this function converts each root value exactly once.
-- * INNER invocations (`inv.parent_id ~= 0`) receive physical `call_line` values directly from the COMPONENT's `line_of` in the walker.
-- Recursive descent forwards that closure through `corpus.component_body_index[name].line_of`; those values arrive physical and remain unchanged.
--
-- After this function, every `inv.call_line` is physical. DWARF and provenance output read it directly.
-- The word-event loop forwards the already-physical `outer_inv.call_line` into `we.call_line` for words inside an invocation.
local function stamp_root_provenance(projection, atom_record, src, corpus) local function stamp_root_provenance(projection, atom_record, src, corpus)
local root_line_of = src.scan and src.scan.line_of local root_line_of = src.scan and src.scan.line_of
local root_body_line = root_line_of and root_line_of((atom_record.body_off or 1) - 1) assert(type(root_line_of) == "function"
, "emission_model: src.scan.line_of is required (canonical LineIndex closure over the source text) to stamp physical provenance")
assert(type(atom_record.body_off) == "number"
, "emission_model: atom_record.body_off (byte offset of the body's first byte in source) is required to derive `root_body_line`. The scanner must populate body_off for every atom record.")
-- `root_body_line` is the physical source line of the ATOM HEADER byte containing the opening `{`; that byte is one byte BEFORE `atom_record.body_off`.
-- The walker assigns line 2 to the body's first content line because line 1 is the trailing `\n` after `{`. Body-text line k therefore maps to `root_body_line + (k - 1)`.
-- `body_off - 1` points at the opening `{`, whose line index identifies the header line. `body_off` points after `{` and would shift every word row forward by one line.
local root_body_line = root_line_of(atom_record.body_off - 1)
or atom_record.line or 0 or atom_record.line or 0
local component_index = corpus.component_body_index or {} local component_index = corpus.component_body_index or {}
local word_items = {} local word_items = {}
@@ -51,27 +68,32 @@ local function stamp_root_provenance(projection, atom_record, src, corpus)
if item.kind == "word" then word_items[#word_items + 1] = item end if item.kind == "word" then word_items[#word_items + 1] = item end
end end
-- Resolve one word's physical body line, where the byte containing that word appears in source.
-- * Component expansions carry `invocation_ids`; the component's full-file `line_of` leaves `item.line` physical.
-- * Raw tokens in the root atom body carry an empty `invocation_ids` list and a body-relative `item.line`; convert them here.
local function body_line_for(event, item) local function body_line_for(event, item)
local body_line_of = root_line_of
local body_off = atom_record.body_off or 0
local ids = event.invocation_ids or {} local ids = event.invocation_ids or {}
-- The innermost open invocation identifies which line index the walker used.
-- A component `line_of` makes `item.line` physical; the atom's `body_text` line index makes it body-relative.
if ids and #ids > 0 then
local inner_id = ids[#ids] local inner_id = ids[#ids]
local inner_inv = inner_id and projection.invocations[inner_id] local inner_inv = inner_id and projection.invocations[inner_id]
if inner_inv then if inner_inv then
local component = component_index[inner_inv.component_name] local component = component_index[inner_inv.component_name]
if component and component.line_of then if component and component.line_of then
-- Component-body walkers already receive the declaration source's full line index, so their item.line is physical. -- Walker used `comp.line_of`, which is the source's physical LineIndex. item.line is already physical.
return item.line or 0 return item.line or 0
end end
end end
local first_line = body_line_of and body_line_of(math.max(1, body_off - 1)) or root_body_line end
return (first_line or 0) + (item.line or 1) - 1 -- RAW root-body word: item.line is body-text's 1-based line number (the first content line is line 2 because line 1 is the trailing `\n` after `{`).
-- Convert body-text-relative → physical using `root_body_line + (item.line - 1)`.
return (root_body_line or 0) + (item.line or 1) - 1
end end
-- Stamp root-source path onto invocation records whose `call_path` was left empty by the walker. -- Stamp the root source path onto invocation records whose `call_path` the walker left empty.
-- The walker passes `body_entry.source` to `emit_invoke_begin` as the call_path argument; for the root body_entry created by `M.project_emission` that source is "" -- The walker passes `body_entry.source` to `emit_invoke_begin`; `M.project_emission` creates the root `body_entry` with source `""`, leaving its `call_path` empty.
-- (the caller passes only the body text). -- This stamp gives every invocation a physical `call_path` matching `passes/atoms_source_map.lua`'s in-memory provenance projection.
-- After this stamp every invocation record has a physical call_path that matches what `passes/atoms_source_map.lua` matches the in-memory provenance projection.
local root_path = src.path or "" local root_path = src.path or ""
for _, inv in ipairs(projection.invocations) do for _, inv in ipairs(projection.invocations) do
if inv.call_path == nil or inv.call_path == "" then if inv.call_path == nil or inv.call_path == "" then
@@ -79,6 +101,35 @@ local function stamp_root_provenance(projection, atom_record, src, corpus)
end end
end end
-- Normalize `inv.call_line` to a physical source line.
-- * ROOT invocations (`parent_id == 0`) carry body-relative `call_line` values from `M.LineIndex(body_text)`; convert them once with `root_body_line`.
-- * INNER invocations (`parent_id ~= 0`) carry physical `call_line` values from the component's `line_of`; retain them unchanged.
for _, inv in ipairs(projection.invocations) do
if inv.parent_id == 0 then
inv.call_line = (root_body_line or 0) + (inv.call_line or 1) - 1
end
end
-- Build `body_lines` for each invocation.
-- `atoms_source_map` and `dwarf_injection` read `inv.body_lines[k]` directly from the invocation record created here.
-- Component words already carry physical `item.line` values from the walker's COMPONENT line index, so `body_line_for` returns them unchanged.
for _, inv in ipairs(projection.invocations) do
local sw = inv.start_word
local ew = inv.end_word
local bls = {}
for i = sw, ew do
local it = projection.items and projection.items[i]
if it and it.kind == "word" then
local fake_event = { invocation_ids = { inv.id } }
bls[#bls + 1] = body_line_for(fake_event, it) or 0
end
end
inv.body_lines = bls
end
-- Resolve each `word_event`'s physical `body_line` and `call_line`.
-- For words inside an invocation, `we.call_line` identifies the OUTER atom source line containing the `mac_X(...)` token that triggered expansion.
-- The root-invocation conversion above makes every `inv.call_line` physical; forward it directly and use each raw word's `body_line` as the fallback.
for index, we in ipairs(projection.word_events) do for index, we in ipairs(projection.word_events) do
local item = word_items[index] or {} local item = word_items[index] or {}
local body_line = body_line_for(we, item) local body_line = body_line_for(we, item)
@@ -88,7 +139,10 @@ local function stamp_root_provenance(projection, atom_record, src, corpus)
local call_line = body_line local call_line = body_line
local outer_id = we.outermost_invocation_id or 0 local outer_id = we.outermost_invocation_id or 0
local outer_inv = projection.invocations[outer_id] local outer_inv = projection.invocations[outer_id]
if outer_inv then call_line = (root_body_line or 0) + (outer_inv.call_line or 1) - 1 end if outer_inv then
-- `outer_inv.call_line` is physical after the conversion loop above, so use it directly.
call_line = outer_inv.call_line
end
we.call_line = call_line we.call_line = call_line
if we.def_path == nil or we.def_path == "" then we.def_path = src.path or "" end if we.def_path == nil or we.def_path == "" then we.def_path = src.path or "" end
@@ -103,7 +157,8 @@ local function project_atom(atom_record, src, corpus)
local body = atom_record.body or "" local body = atom_record.body or ""
local wc = corpus.word_counts or {} local wc = corpus.word_counts or {}
local cbi = corpus.component_body_index or {} local cbi = corpus.component_body_index or {}
local proj = duffle.project_emission(body, cbi, wc) -- That construction site stamps `invocation.debug_skip` while appending each record to `proj.invocations`.
local proj = duffle.project_emission(body, cbi, wc, corpus.components)
local paths = { local paths = {
tokens = atom_record.body_tokens or {}, tokens = atom_record.body_tokens or {},
line_in_body = duffle.build_body_line_index(body), line_in_body = duffle.build_body_line_index(body),
@@ -123,7 +178,7 @@ end
-- Run the emission-model pass. -- Run the emission-model pass.
-- ───────────────────────────────────────────────────────────────────────── -- ─────────────────────────────────────────────────────────────────────────
--- @param ctx PassCtx -- { shared = { corpus = ... }, out_root, dry_run, ... } --- @param ctx PassCtx -- { shared = { corpus = ... }, out_root, ... }
--- @return PassResult --- @return PassResult
function M.run(ctx) function M.run(ctx)
local outputs = {} local outputs = {}
@@ -134,53 +189,43 @@ function M.run(ctx)
if type(corpus) ~= "table" then error("emission_model: ctx.shared.corpus is required (canonical projection)", 0) end if type(corpus) ~= "table" then error("emission_model: ctx.shared.corpus is required (canonical projection)", 0) end
if type(corpus.source_order) ~= "table" then error("emission_model: ctx.shared.corpus.source_order is required", 0) end if type(corpus.source_order) ~= "table" then error("emission_model: ctx.shared.corpus.source_order is required", 0) end
-- Walk every source in canonical source order; for each source, iterate atoms. -- Project once, collect errors + warnings for one atom.
-- Atom declarations (`kind == "atom"` / `"raw_atom"`) receive the canonical `atom.paths` projection; component declarations -- Kind must be one of: atom | raw_atom | comp_bare | comp_proc.
-- (`comp_bare` / `comp_proc`) are recursively expanded by atom projections and do not get an independent projection themselves. local function process_atom(atom, src)
-- Test-only fixtures that need per-component word events may still consume `duffle.expand_word_events` if not (atom and atom.body) then return end
-- (which remains available; emission-model owns the canonical per-atom projection). local kind = atom.kind
if kind ~= "atom" and kind ~= "raw_atom" and kind ~= "comp_bare" and kind ~= "comp_proc" then
return
end
local proj = project_atom(atom, src, corpus)
for _, e in ipairs(proj.errors) do
-- Preserve `kind` (cycle / count_mismatch / unbalanced) so readers dispatch on the diagnostic class and leave the message string as display text.
errors[#errors + 1] = {
kind = e.kind,
line = e.line,
msg = e.msg,
source = e.source or src.path,
}
end
for _, w in ipairs(proj.warnings) do
warnings[#warnings + 1] = {
kind = w.kind,
line = w.line,
msg = w.msg,
}
end
end
-- Walk `corpus.source_order`; within each source, visit atoms followed by raw_atoms.
-- Recognized kinds (atom | raw_atom | comp_bare | comp_proc) each receive the atom.paths projection via duffle.project_emission.
-- Components are macros inlined into atom bodies; focused tests and isolated component analyses consume atom.paths directly.
for _, src in ipairs(corpus.source_order) do for _, src in ipairs(corpus.source_order) do
local scan = src.scan or {} local scan = src.scan or {}
for _, atom in ipairs(scan.atoms or {}) do for _, atom in ipairs(scan.atoms or {}) do
if atom and atom.body and (atom.kind == "atom" or atom.kind == "raw_atom") then process_atom(atom, src)
local proj = project_atom(atom, src, corpus)
for _, e in ipairs(proj.errors) do
-- Preserve `kind` (cycle / count_mismatch / unbalanced) so readers can dispatch on the diagnostic class without re-parsing the message string.
errors[#errors + 1] = {
kind = e.kind,
line = e.line,
msg = e.msg,
source = e.source or src.path,
}
end
for _, w in ipairs(proj.warnings) do
warnings[#warnings + 1] = {
kind = w.kind,
line = w.line,
msg = w.msg,
}
end
end
end end
for _, atom in ipairs(scan.raw_atoms or {}) do for _, atom in ipairs(scan.raw_atoms or {}) do
if atom and atom.body then process_atom(atom, src)
local proj = project_atom(atom, src, corpus)
for _, e in ipairs(proj.errors) do
errors[#errors + 1] = {
kind = e.kind,
line = e.line,
msg = e.msg,
source = e.source or src.path,
}
end
for _, w in ipairs(proj.warnings) do
warnings[#warnings + 1] = {
kind = w.kind,
line = w.line,
msg = w.msg,
}
end
end
end end
end end
+1 -4
View File
@@ -48,9 +48,8 @@ local OFFSET_MACRO_COL = 44
--- @class PassCtx --- @class PassCtx
--- @field shared table -- cross-pass shared state --- @field shared table -- cross-pass shared state
--- @field shared.corpus table -- canonical corpus projection --- @field shared.corpus table -- canonical corpus projection
--- @field shared.word_counts table -- compatibility alias to corpus.word_counts --- @field shared.word_counts table
--- @field out_root string -- output root (e.g. "build/gen") --- @field out_root string -- output root (e.g. "build/gen")
--- @field dry_run boolean -- if true, compute but don't write
--- @class PassResult --- @class PassResult
--- @field outputs table[] -- {kind=, path=} entries describing emit files --- @field outputs table[] -- {kind=, path=} entries describing emit files
@@ -220,10 +219,8 @@ local function process_source(ctx, src)
if #atoms_data == 0 then return nil end if #atoms_data == 0 then return nil end
local out_path = src.dir .. "/gen/" .. duffle.basename_no_ext(src.dir) .. ".offsets.h" local out_path = src.dir .. "/gen/" .. duffle.basename_no_ext(src.dir) .. ".offsets.h"
if not ctx.dry_run then
duffle.ensure_dir(duffle.dirname(out_path)) duffle.ensure_dir(duffle.dirname(out_path))
duffle.write_file(out_path, generate_header(src.path:gsub("/", "\\"), atoms_data)) duffle.write_file(out_path, generate_header(src.path:gsub("/", "\\"), atoms_data))
end
return out_path return out_path
end end
+23 -26
View File
@@ -5,8 +5,8 @@
--- - `build/gen/<dir_basename>.annotations.txt` — one per source-directory containing atoms; aggregates across all sources in the directory. --- - `build/gen/<dir_basename>.annotations.txt` — one per source-directory containing atoms; aggregates across all sources in the directory.
--- - `build/gen/annotation_validation.txt` — the project summary. --- - `build/gen/annotation_validation.txt` — the project summary.
--- ---
--- The annotation pass stashes per-MODULE summary entries in `ctx.flags._annot_results` (set by `passes/annotation.lua`). --- The annotation pass emits `errors.h` files per module and the canonical `corpus.sources_by_dir` projection groups sources by directory.
--- This pass re-validates each source via `annotation.validate()` to get the detailed per-source results needed for the report. --- This pass iterates the canonical dir projection directly and re-validates each source via `annotation.validate()` to get the detailed per-source results.
--- ---
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex, --- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
--- Lua 5.3 compatible. --- Lua 5.3 compatible.
@@ -25,6 +25,11 @@
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
-- Load the annotation pass so we can re-validate each source against the canonical corpus projection.
-- The annotation pass exposes `M.validate`, which returns the per-source AnnotationResult (atoms / annots / macros / binds / errors / warnings)
-- that the report pass renders into the per-module `<dir_basename>.annotations.txt` output.
local annotation = dofile(_bootstrap_dir .. "annotation.lua")
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Constants -- Constants
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -67,8 +72,6 @@ local PASS_NAME = "report"
--- @field project_root string -- project root (e.g. "code/") --- @field project_root string -- project root (e.g. "code/")
--- @field upstream table<string, table> -- per-pass upstream outputs --- @field upstream table<string, table> -- per-pass upstream outputs
--- @field flags table -- CLI flags + per-pass stash --- @field flags table -- CLI flags + per-pass stash
--- @field flags._annot_results ModuleEntry[] -- stashed by annotation pass
--- @field dry_run boolean -- if true, compute but don't write
--- @field verbose boolean -- if true, log diagnostic info --- @field verbose boolean -- if true, log diagnostic info
--- @class PassResult --- @class PassResult
@@ -315,9 +318,7 @@ local function render_module_report(dir, sources, results)
-- Each renderer writes its header + content via the `add` closure (pre-bound above). -- Each renderer writes its header + content via the `add` closure (pre-bound above).
-- Adding a new section = 1 row here + 1 render_<thing>_section function. -- Adding a new section = 1 row here + 1 render_<thing>_section function.
for _, section in ipairs(SECTION_RENDERERS) do for _, section in ipairs(SECTION_RENDERERS) do
add(section.header)
section.render(add, results, totals) section.render(add, results, totals)
add("")
end end
return table.concat(lines, "\n") .. "\n" return table.concat(lines, "\n") .. "\n"
@@ -377,21 +378,20 @@ end
-- Orchestration helpers -- Orchestration helpers
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- (internal) Pull per-source validate() results from the annotation pass's stash. --- (internal) Re-validate every source in a directory against the canonical corpus projection.
--- The annotation pass runs first in the dep chain and caches results in `ctx.flags._annot_source_results`; --- Calls `annotation.validate()` per source to produce the per-source AnnotationResult (atoms / annots / macros / binds / errors / warnings)
--- we read from there instead of re-validating each source. --- that the report renderer consumes. Eeach report pass run is reproducible from the corpus.
--- Returns the list of module results + the flat list of all results (for the project-wide summary). --- Returns the list of module results + the flat list of all results (for the project-wide summary).
--- @param ctx PassCtx --- @param ctx PassCtx
--- @param dir_sources SourceFile[] --- @param dir_sources SourceFile[]
--- @return AnnotationResult[], AnnotationResult[] --- @return AnnotationResult[], AnnotationResult[]
local function lookup_module_results(ctx, dir_sources) local function lookup_module_results(ctx, dir_sources)
local src_cache = (ctx.flags and ctx.flags._annot_source_results) or {}
local module_results = {} local module_results = {}
local all_results = {} local all_results = {}
for _, src in ipairs(dir_sources) do for _, src in ipairs(dir_sources) do
local result = src_cache[src.path] if src.scan then
if result then local result = annotation.validate(ctx, src, nil)
result.source = src.path -- defensive (annotation tags it too; this guards against cache misses from earlier iterations) result.source = src.path -- tag for downstream rendering
module_results[#module_results + 1] = result module_results[#module_results + 1] = result
all_results[#all_results + 1] = result all_results[#all_results + 1] = result
end end
@@ -435,30 +435,27 @@ function M.run(ctx)
local errors = {} local errors = {}
local warnings = {} local warnings = {}
local module_entries = (ctx.flags and ctx.flags._annot_results) or {} -- Module grouping comes from `corpus.sources_by_dir` (the canonical projection).
-- Read module grouping from `corpus.sources_by_dir` (the canonical projection). -- Iterate it directly; no private cache, no per-pass stash.
-- Module grouping comes from `corpus.sources_by_dir`.
local corpus = ctx.shared and ctx.shared.corpus local corpus = ctx.shared and ctx.shared.corpus
local by_dir = (corpus and corpus.sources_by_dir) or {} local by_dir = (corpus and corpus.sources_by_dir) or {}
if not ctx.dry_run then duffle.ensure_dir(ctx.out_root) end duffle.ensure_dir(ctx.out_root)
local all_results_for_summary = {} local all_results_for_summary = {}
for _, entry in ipairs(module_entries) do for dir, dir_sources in pairs(by_dir) do
debug_log("entry: dir=%s basename=%s atoms_count=%d dir_sources=%d\n", entry.dir, entry.dir_basename, entry.atoms_count, #(by_dir[entry.dir] or {})) local dir_basename = dir:match("([^/\\]+)$") or dir
debug_log("dir=%s basename=%s sources=%d\n", dir, dir_basename, #dir_sources)
if entry.atoms_count > 0 or #(by_dir[entry.dir] or {}) > 0 then if #dir_sources > 0 then
local dir_sources = by_dir[entry.dir] or {}
local module_results, all_results = lookup_module_results(ctx, dir_sources) local module_results, all_results = lookup_module_results(ctx, dir_sources)
for _, r in ipairs(all_results) do for _, r in ipairs(all_results) do
all_results_for_summary[#all_results_for_summary + 1] = r all_results_for_summary[#all_results_for_summary + 1] = r
end end
if module_has_content(module_results) then if module_has_content(module_results) then
local out_path = ctx.out_root .. "/" .. entry.dir_basename .. ".annotations.txt" local out_path = ctx.out_root .. "/" .. dir_basename .. ".annotations.txt"
if not ctx.dry_run then duffle.write_file(out_path, render_module_report(dir, dir_sources, module_results))
duffle.write_file(out_path, render_module_report(entry.dir, dir_sources, module_results))
end
outputs[#outputs + 1] = { annotations_txt = out_path } outputs[#outputs + 1] = { annotations_txt = out_path }
else else
debug_log(" -> no content; skipping\n") debug_log(" -> no content; skipping\n")
@@ -466,7 +463,7 @@ function M.run(ctx)
end end
end end
if not ctx.dry_run and #all_results_for_summary > 0 then if #all_results_for_summary > 0 then
local summary_path = ctx.out_root .. "/annotation_validation.txt" local summary_path = ctx.out_root .. "/annotation_validation.txt"
duffle.write_file(summary_path, render_project_report(all_results_for_summary)) duffle.write_file(summary_path, render_project_report(all_results_for_summary))
outputs[#outputs + 1] = { summary_txt = summary_path } outputs[#outputs + 1] = { summary_txt = summary_path }
+228 -146
View File
@@ -1,12 +1,12 @@
--- passes/scan_source.lua — Source pre-scan pass (the "mega entity" pass). --- passes/scan_source.lua — Source pre-scan pass (the "mega entity" pass).
--- ---
--- Single source-walk pass that produces the fat `SourceScan` payload consumed by all downstream passes. Walks each `ctx.sources` entry once, --- Single source-walk pass that produces the fat `SourceScan` payload consumed by all downstream passes. Walks each corpus source record once,
--- extracting every construct type the metaprograms need: --- extracting every construct type the metaprograms need:
--- ---
--- MipsAtom_ (kind = "atom", with optional atom_info inner) --- MipsAtom_ (kind = "atom", with optional atom_info inner)
--- MipsAtomComp_ (kind = "comp_bare") --- MipsAtomComp_ (kind = "comp_bare")
--- MipsAtomComp_Proc_ (kind = "comp_proc", body inside last {}) --- MipsAtomComp_Proc_ (kind = "comp_proc", body inside last {})
--- atom_dbg_skip_over (whole-atom/component debug-step marker; following declaration disambiguates) --- atom_dbg_skip — bare whole-atom/component debug-step marker; following declaration disambiguates
--- MipsCode code_<name> (kind = "raw_atom", offsets pass only) --- MipsCode code_<name> (kind = "raw_atom", offsets pass only)
--- typedef Struct_(Binds_X) { fields } --- typedef Struct_(Binds_X) { fields }
--- #pragma mac_X tape_atom words=N + _Pragma("...") --- #pragma mac_X tape_atom words=N + _Pragma("...")
@@ -42,33 +42,24 @@ local parse_enum_int_literal
--- @field binds BindsEntry[] -- typedef Struct_(Binds_X) { fields } (fields pre-parsed) --- @field binds BindsEntry[] -- typedef Struct_(Binds_X) { fields } (fields pre-parsed)
--- @field atom_infos AtomInfoEntry[] -- MipsAtom_(name) atom_info(...) (sub-calls pre-parsed) --- @field atom_infos AtomInfoEntry[] -- MipsAtom_(name) atom_info(...) (sub-calls pre-parsed)
--- @field macros MacroEntry[] -- #pragma mac_X tape_atom words=N + _Pragma("...") --- @field macros MacroEntry[] -- #pragma mac_X tape_atom words=N + _Pragma("...")
--- @field skip_over SkipOverScan -- atom/component debug-step markers + resolved declaration associations --- @field debug_skip_markers DebugSkipMarker[] -- raw marker evidence for annotation validation; `debug_skip` lives on the declaration record itself
--- @field types table<string, RegTypeDefault> -- atom_dbg_reg_default(R_X, <type>) declarations --- @field types table<string, RegTypeDefault> -- atom_dbg_reg_default(R_X, <type>) declarations
--- @field atom_views table<string, AtomViewEntry> -- MipsAtom_(name) -> {binds_name, reg_type_overrides, info_line} --- @field atom_views table<string, AtomViewEntry> -- MipsAtom_(name) -> {binds_name, reg_type_overrides, info_line}
--- @field atom_ctxs table<string, AtomCtxEntry> -- MipsAtom_(name) -> {rbind_atom, info_line, source} (atom_ctx(...) call sites) --- @field atom_ctxs table<string, AtomCtxEntry> -- MipsAtom_(name) -> {rbind_atom, info_line, source} (atom_ctx(...) call sites)
--- @field atom_phases table<string, AtomPhaseGroup> -- phase_label -> {atoms = {atom_name1, atom_name2, ...}} (atom_phase(...) tags) --- @field atom_phases table<string, AtomPhaseGroup> -- phase_label -> {atoms = {atom_name1, atom_name2, ...}} (atom_phase(...) tags)
--- @field line_of fun(pos: integer): integer -- shared LineIndex closure --- @field line_of fun(pos: integer): integer -- shared LineIndex closure
--- @class SkipOverScan --- @class DebugSkipMarker
--- @field atoms table<string, SkipOverAssociation> --- @field marker_kind string -- exact marker ident read from source. Only "atom_dbg_skip" (bare) is positive; any other ident reaches the unrelated fallback and is never associated with a declaration.
--- @field components table<string, SkipOverAssociation> --- @field marker_line integer -- line of the marker ident start
--- @field markers SkipOverMarker[] --- @field marker_pos integer -- byte position of the marker ident start (the comment walker anchors here)
--- @field is_bare boolean -- true iff marker_kind == "atom_dbg_skip" AND has_parens == false (the only positive form)
--- @class SkipOverMarker --- @field has_parens boolean -- true iff a `(...)` follows the marker ident (diagnostic-only)
--- @field marker_kind string -- exact marker ident (always "atom_dbg_skip_over") --- @field args string|nil -- trimmed args inside the `(...)` (nil when has_parens is false)
--- @field marker_line integer --- @field pending boolean -- true while awaiting the following declaration
--- @field marker_pos integer --- @field superseded_by_marker_line integer|nil -- set when a newer marker bumped this one out of the pending slot
--- @field after_paren integer --- @field target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed (nil if no declaration ever followed)
--- @field args string|nil -- trimmed marker args"" --- @field proc_prelude boolean|nil -- true after the marker crossed an `FI_` prelude and awaits `MipsAtomComp_Proc_`
--- @field has_parens boolean
--- @field pending boolean
--- @field superseded_by_marker_line integer|nil
--- @field target_name string|nil -- stripped declaration name once observed
--- @field target_raw_name string|nil -- source-written declaration name once observed
--- @field target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed
--- @field declaration_line integer|nil
--- @field declaration_pos integer|nil
--- @field proc_prelude boolean|nil -- marker has crossed FI_ and awaits MipsAtomComp_Proc_
--- @class RegTypeDefault --- @class RegTypeDefault
--- @field name string -- "R_TapePtr" (the register ident; without the value part) --- @field name string -- "R_TapePtr" (the register ident; without the value part)
@@ -96,12 +87,6 @@ local parse_enum_int_literal
--- @field reg_type_overrides table<string, RegTypeOverride> -- "R_T0" -> override --- @field reg_type_overrides table<string, RegTypeOverride> -- "R_T0" -> override
--- @field info_line integer -- line of the atom_info call --- @field info_line integer -- line of the atom_info call
--- @class SkipOverAssociation
--- @field marker_line integer
--- @field declaration_line integer
--- @field kind string
--- @field marker SkipOverMarker
--- @class SourceFile --- @class SourceFile
--- @field path string -- absolute path to the source file --- @field path string -- absolute path to the source file
--- @field text string -- the full source text --- @field text string -- the full source text
@@ -117,7 +102,6 @@ local parse_enum_int_literal
--- @field project_root string --- @field project_root string
--- @field upstream table<string, table> --- @field upstream table<string, table>
--- @field flags table --- @field flags table
--- @field dry_run boolean
--- @field verbose boolean --- @field verbose boolean
--- @class PassResult --- @class PassResult
@@ -134,8 +118,8 @@ local parse_enum_int_literal
--- @field raw_name string -- un-stripped name (for components: with ac_ prefix) --- @field raw_name string -- un-stripped name (for components: with ac_ prefix)
--- @field ident_pos integer -- position of the MipsAtom_/MipsAtomComp_ ident start --- @field ident_pos integer -- position of the MipsAtom_/MipsAtomComp_ ident start
--- @field after_paren integer -- position past the closing paren --- @field after_paren integer -- position past the closing paren
--- @field args string|nil -- populated by components pass (backward lookup) --- @field debug_skip boolean -- true when an `atom_dbg_skip` bare marker immediately precedes this declaration (sole-owner stamp; see push_debug_skip_marker)
--- @field comment string|nil -- populated by components pass (backward lookup) --- @field declaration_comment string|nil -- populated by the scanner (backward walk past the marker, captures contiguous `/* */` or `//` block)
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Local helpers (shared by per-form parsers) -- Local helpers (shared by per-form parsers)
@@ -167,8 +151,10 @@ local function strip_ac_prefix(raw_name)
end end
-- Preserve a source marker until the following declaration parser observes it. -- Preserve a source marker until the following declaration parser observes it.
local function push_skip_over_marker(out, marker) -- The scanner is the sole owner of marker recognition, placement association, declaration comment attachment, and canonical `debug_skip` fields.
local markers = out.skip_over.markers -- Raw marker evidence lives in `out.debug_skip_markers` for annotation validation; the declaration record carries the resolved `debug_skip` boolean directly.
local function push_debug_skip_marker(out, marker)
local markers = out.debug_skip_markers
local prior = markers[#markers] local prior = markers[#markers]
if prior and prior.pending then if prior and prior.pending then
prior.pending = false prior.pending = false
@@ -198,55 +184,150 @@ local function find_body_braces(source, after_paren, fallback)
return body, after_brace, brace + 1 return body, after_brace, brace + 1
end end
-- Walk backward from `start_pos` capturing contiguous `/* */` block(s) and
-- `//` line(s) that immediately precede it. The caller (preceding_declaration_comment)
-- supplies `start_pos` so the walker does not need to detect marker shape or prelude layout.
-- The scanner already knows the marker_pos + decl ident_pos and threads that knowledge forward.
--
-- The walker captures:
-- - Block comment close `*/` followed by walking back to `/*`.
-- - `//` line comments (the line containing the current non-ws position starts with `//`).
-- It stops at the first non-ws char that does not begin a comment block or line.
-- Empty string if no comment is adjacent.
-- @param source string
-- @param start_pos integer -- exclusive upper bound for the captured block
-- @return string
local function preceding_comment_walk_backward(source, start_pos)
local pieces = {}
local scan_pos = start_pos
while scan_pos > 0 do
local non_ws = scan_pos - 1
while non_ws > 0 do
local ch = source:sub(non_ws, non_ws)
if ch == " " or ch == "\t" or ch == "\n" or ch == "\r" then
non_ws = non_ws - 1
else
break
end
end
if non_ws == 0 then break end
if non_ws >= 2 and source:sub(non_ws - 1, non_ws) == "*/" then
-- Block comment close: walk back over `/*` candidates.
local prefix = source:sub(1, non_ws - 1)
local open_at = nil
for scan = #prefix - 1, 1, -1 do
if prefix:sub(scan, scan + 1) == "/*" then
open_at = scan
break
end
end
if not open_at then break end
local block_start = open_at
while block_start > 1 do
local ch = source:sub(block_start - 1, block_start - 1)
if ch ~= " " and ch ~= "\t" then break end
block_start = block_start - 1
end
table.insert(pieces, 1, source:sub(block_start, non_ws))
scan_pos = block_start
else
-- Line comment check: walk back from non_ws to the most recent `\n`
-- (or position 1) and inspect the resulting line. This handles both
-- `// foo\n<marker>` (non_ws ends on `o`) and `// foo\r\n<marker>`.
local line_start = non_ws
while line_start > 1 and source:sub(line_start - 1, line_start - 1) ~= "\n" do
line_start = line_start - 1
end
local line = source:sub(line_start, non_ws)
if line:sub(1, 2) ~= "//" then break end
table.insert(pieces, 1, line)
scan_pos = line_start - 1
end
end
if #pieces == 0 then return "" end
return table.concat(pieces, "\n")
end
-- Resolve the start position for the declaration-comment walk.
-- When a debug-skip marker is pending, the walker must start from the position immediately before the marker ident
-- (so it walks backward past the marker text and any `FI_ MipsAtom ac_X(args)` proc-prelude layout — neither of which is visible if we start from the declaration ident_pos).
-- When no marker is pending, the walker starts from the declaration ident_pos directly.
-- @param pending_marker DebugSkipMarker|nil
-- @param ident_pos integer -- declaration ident position
-- @return integer
local function comment_walk_start(pending_marker, ident_pos)
if pending_marker then
return pending_marker.marker_pos - 1
end
return ident_pos - 1
end
-- Attach the pending marker to the next declaration. -- Attach the pending marker to the next declaration.
-- The declaration form disambiguates whole atoms from components; -- The declaration form disambiguates whole atoms from components; the resolved `debug_skip` is stamped directly on the declaration record
-- unsupported declarations retain placement evidence for annotation.lua and populate neither lookup table. -- (sole-owner discipline; see push_debug_skip_marker).
local function associate_skip_over_marker(out, target_name, target_raw_name, target_kind, declaration_line, declaration_pos) --
local markers = out.skip_over.markers -- A marker is POSITIVE (stamps `debug_skip = true` on the declaration) iff:
-- marker_kind == "atom_dbg_skip" AND is_bare == true
-- Any other spelling or shape (parenthesized form, legacy name) is recorded as a raw marker for annotation validation but never stamps `debug_skip`.
-- @param out SourceScan
-- @param target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed
-- @return boolean|nil -- true iff the marker is the positive bare form
local function attach_debug_skip_marker(out, target_kind)
local markers = out.debug_skip_markers
local marker = markers[#markers] local marker = markers[#markers]
if not (marker and marker.pending) then return end if not (marker and marker.pending) then return nil end
marker.pending = false marker.pending = false
marker.target_name = target_name
marker.target_raw_name = target_raw_name
marker.target_kind = target_kind marker.target_kind = target_kind
marker.declaration_line = declaration_line
marker.declaration_pos = declaration_pos
if not (marker.has_parens and marker.args == "") then return end if marker.marker_kind == "atom_dbg_skip" and marker.is_bare then
return true
local association = {
marker_line = marker.marker_line,
declaration_line = declaration_line,
kind = target_kind,
marker = marker,
}
if target_kind == "atom" then
out.skip_over.atoms[target_name] = association
elseif target_kind == "comp_bare" or target_kind == "comp_proc" then
out.skip_over.components[target_name] = association
end end
return nil
end end
-- Register a parsed atom entry in `out.atoms` and link its skip-over marker. -- Register a parsed atom entry in `out.atoms`. Stamps the resolved `debug_skip` boolean
-- Captures the shared 8-field shape used by MipsAtom_, MipsAtomComp_, MipsAtomComp_Proc_. -- on the record when a positive bare `atom_dbg_skip` marker is pending.
local function register_atom(out, kind, declaration_line, name, body, body_off, raw_name, pos, after_paren) -- Captures the shared shape used by MipsAtom_, MipsAtomComp_, MipsAtomComp_Proc_.
local function register_atom(out, kind, declaration_line, name, body, body_off, raw_name, pos, after_paren, source)
-- Capture the pending marker BEFORE attaching so the walker can anchor the backward comment walk on the marker's marker_pos
-- (which is the correct anchor even when an `FI_ MipsAtom ac_X(args)` proc-prelude separates the marker from the declaration).
local pending_marker = nil
local markers = out.debug_skip_markers
local m = markers[#markers]
if m and m.pending then pending_marker = m end
local positive = attach_debug_skip_marker(out, kind)
local comment = ""
if kind == "comp_bare" or kind == "comp_proc" then
-- Scanner-owned declaration-comment attachment.
-- The walker does not need to detect marker shape.
-- A pending_marker record (or the declaration ident_pos fallback) supplies the anchor position.
local start_pos = comment_walk_start(pending_marker, pos)
comment = preceding_comment_walk_backward(source, start_pos)
end
out.atoms[#out.atoms + 1] = { out.atoms[#out.atoms + 1] = {
line = declaration_line, name = name, body = body, body_off = body_off, line = declaration_line,
kind = kind, raw_name = raw_name, name = name,
ident_pos = pos, after_paren = after_paren, body = body,
body_off = body_off,
kind = kind,
raw_name = raw_name,
ident_pos = pos,
after_paren = after_paren,
debug_skip = positive == true,
declaration_comment = comment,
} }
associate_skip_over_marker(out, name, raw_name, kind, declaration_line, pos)
end end
-- Register a parsed raw-atom entry in `out.raw_atoms` and link its skip-over marker. -- Register a parsed raw-atom entry in `out.raw_atoms`.
-- Captures the 5-field shape used by MipsCode (the raw-atom form; offsets pass only). -- Captures the 5-field shape used by MipsCode (the raw-atom form; offsets pass only).
local function register_raw_atom(out, declaration_line, name, body, body_off, raw_name, pos, marker_kind) local function register_raw_atom(out, declaration_line, name, body, body_off, raw_name, pos)
out.raw_atoms[#out.raw_atoms + 1] = { out.raw_atoms[#out.raw_atoms + 1] = {
line = declaration_line, name = name, body = body, body_off = body_off, line = declaration_line, name = name, body = body, body_off = body_off,
kind = "raw_atom", raw_name = raw_name, kind = "raw_atom", raw_name = raw_name,
} }
associate_skip_over_marker(out, name, raw_name, marker_kind, declaration_line, pos)
end end
-- Parse a `Type*` chain (zero or more `*` separated by optional whitespace) followed by the type ident. -- Parse a `Type*` chain (zero or more `*` separated by optional whitespace) followed by the type ident.
@@ -355,7 +436,7 @@ end
-- Parse the `Enum_(<underlying>, <name>) { <body> }` body for entries. -- Parse the `Enum_(<underlying>, <name>) { <body> }` body for entries.
-- Captures one field per named enumerator with the shape { name, value }. -- Captures one field per named enumerator with the shape { name, value }.
-- The value is the integer literal parsed from the source via the canonical `parse_enum_int_literal`. -- The value is the integer literal parsed from the source via `parse_enum_int_literal`.
local function parse_enum_body_fields(body) local function parse_enum_body_fields(body)
return walk_body_fields(body, function(entry_name, name_end, after_name) return walk_body_fields(body, function(entry_name, name_end, after_name)
local value local value
@@ -653,20 +734,13 @@ local function scan_atom_info_subcalls(info_inner, info_line)
} }
end end
local SUBCALL_HANDLERS = { local SUBCALL_HANDLERS = {
-- scan: atom_bind(<Binds_X>) atom_bind = function(sub_inner) binds = duffle.trim(sub_inner) end, -- scan: atom_bind(<Binds_X>)
atom_bind = function(sub_inner) binds = duffle.trim(sub_inner) end, atom_reads = function(sub_inner, info_line) rw_handler(sub_inner, info_line, "atom_reads") end, -- scan: atom_reads(<R_X [atom_type(<T>)], ...>)
-- scan: atom_reads(<R_X [atom_type(<T>)], ...>) atom_writes = function(sub_inner, info_line) rw_handler(sub_inner, info_line, "atom_writes") end, -- scan: atom_writes(<R_X [atom_type(<T>)], ...>)
atom_reads = function(sub_inner, info_line) rw_handler(sub_inner, info_line, "atom_reads") end, atom_view = function(sub_inner) view_binds = duffle.trim(sub_inner) end, -- scan: atom_view(<Binds_X>)
-- scan: atom_writes(<R_X [atom_type(<T>)], ...>) atom_reg_types = reg_types_handler, -- scan: atom_reg_types(<R_X>, <T>)
atom_writes = function(sub_inner, info_line) rw_handler(sub_inner, info_line, "atom_writes") end, atom_ctx = function(sub_inner, info_line) ident_handler(sub_inner, info_line, "ctx_atom_name") end, -- scan: atom_ctx(<atom_name>)
-- scan: atom_view(<Binds_X>) atom_phase = function(sub_inner, info_line) ident_handler(sub_inner, info_line, "phase_label") end, -- scan: atom_phase(<label>)
atom_view = function(sub_inner) view_binds = duffle.trim(sub_inner) end,
-- scan: atom_reg_types(<R_X>, <T>)
atom_reg_types = reg_types_handler,
-- scan: atom_ctx(<atom_name>)
atom_ctx = function(sub_inner, info_line) ident_handler(sub_inner, info_line, "ctx_atom_name") end,
-- scan: atom_phase(<label>)
atom_phase = function(sub_inner, info_line) ident_handler(sub_inner, info_line, "phase_label") end,
} }
local sub_pos = 1 local sub_pos = 1
@@ -1007,7 +1081,7 @@ end
-- pos -- position of the construct's leading ident (e.g., `M` of `MipsAtom_`) -- pos -- position of the construct's leading ident (e.g., `M` of `MipsAtom_`)
-- ident_end -- position past the leading ident (where the `(` should be) -- ident_end -- position past the leading ident (where the `(` should be)
-- line_of -- closure over LineIndex(source) for 1-based line lookups -- line_of -- closure over LineIndex(source) for 1-based line lookups
-- out -- the SourceScan out table (mutated in place: out.atoms / out.raw_atoms / out.binds / out.atom_infos / out.macros / out.skip_over) -- out -- the SourceScan out table (mutated in place: out.atoms / out.raw_atoms / out.binds / out.atom_infos / out.macros / out.debug_skip_markers)
-- returns -- new position after the construct -- returns -- new position after the construct
-- --
-- All parsers read source-as-written via the duffle primitives (skip_ws_and_cmt / read_parens / read_braces / read_balanced). -- All parsers read source-as-written via the duffle primitives (skip_ws_and_cmt / read_parens / read_braces / read_balanced).
@@ -1016,34 +1090,45 @@ end
-- --
-- Adding a new construct = 1 row in DECL_PARSERS + 1 parser function. The scan_source() loop never needs editing. -- Adding a new construct = 1 row in DECL_PARSERS + 1 parser function. The scan_source() loop never needs editing.
--- Parse an empty debug-skip marker and retain its raw placement evidence. --- Parse a `atom_dbg_skip` marker and record its raw placement evidence.
--- The marker_kind is the source ident itself (e.g. `atom_dbg_skip_over`). ---
--- The dispatch table maps each ident to this same function; --- Positive path: the BARE form (`atom_dbg_skip` followed by whitespace + a supported declaration)
--- the marker_kind is derived from the source so future idents route through the same row. --- stamps the `debug_skip` field on the immediately-following declaration record via `attach_debug_skip_marker`. `is_bare`
--- is set true only when `marker_kind == "atom_dbg_skip"` and there are no parens.
---
--- Diagnostic-only path: a following `(...)` is recorded as an invalid parenthesized-form marker so the annotation rule can emit a precise "parenthesized form" diagnostic.
--- The parenthesized form stays diagnostic; the bare form alone carries the runtime stamp.
--- @param source string --- @param source string
--- @param pos integer --- @param pos integer
--- @param ident_end integer --- @param ident_end integer
--- @param line_of fun(pos: integer): integer --- @param line_of fun(pos: integer): integer
--- @param out SourceScan --- @param out SourceScan
--- @return integer --- @return integer -- source cursor position to resume from
local function parse_skip_over_marker(source, pos, ident_end, line_of, out) local function parse_dbg_skip_marker(source, pos, ident_end, line_of, out)
local marker_kind = source:sub(pos, ident_end - 1)
-- Diagnostic-only detection of an invalid following `(...)`.
-- The cursor is advanced past the `()` either way to keep token order coherent for the next scan iteration.
local marker_end = ident_end
local open_paren = duffle.skip_ws_and_cmt(source, ident_end) local open_paren = duffle.skip_ws_and_cmt(source, ident_end)
local marker = { local has_parens = false
marker_kind = source:sub(pos, ident_end - 1), local args = nil
marker_line = line_of(pos),
marker_pos = pos,
after_paren = ident_end,
args = nil,
has_parens = false,
}
if source:sub(open_paren, open_paren) == "(" then if source:sub(open_paren, open_paren) == "(" then
local inner, after_paren = duffle.read_parens(source, open_paren) local inner, after_paren = duffle.read_parens(source, open_paren)
marker.after_paren = after_paren marker_end = after_paren
marker.args = duffle.trim(inner) has_parens = true
marker.has_parens = true args = duffle.trim(inner)
end end
push_skip_over_marker(out, marker)
return marker.after_paren push_debug_skip_marker(out, {
marker_kind = marker_kind,
marker_line = line_of(pos),
marker_pos = pos,
is_bare = (marker_kind == "atom_dbg_skip") and (not has_parens),
has_parens = has_parens,
args = args,
})
return marker_end
end end
-- Parse `atom_dbg_reg_default(R_X, <type>...)`; -- Parse `atom_dbg_reg_default(R_X, <type>...)`;
@@ -1139,7 +1224,7 @@ local function parse_mips_atom(source, pos, ident_end, line_of, out)
local body, after_brace, body_off = find_body_braces(source, brace_search_pos, open_paren + 1) local body, after_brace, body_off = find_body_braces(source, brace_search_pos, open_paren + 1)
if not body then return after_brace end if not body then return after_brace end
if raw_name and raw_name ~= "" then if raw_name and raw_name ~= "" then
register_atom(out, "atom", line_of(pos), raw_name, body, body_off, raw_name, pos, after_paren) register_atom(out, "atom", line_of(pos), raw_name, body, body_off, raw_name, pos, after_paren, source)
end end
return after_brace return after_brace
@@ -1162,7 +1247,7 @@ local function parse_mips_atom_comp(source, pos, ident_end, line_of, out)
local body, after_brace, body_off = find_body_braces(source, after_paren, open_paren + 1) local body, after_brace, body_off = find_body_braces(source, after_paren, open_paren + 1)
if not body then return after_brace end if not body then return after_brace end
local name = strip_ac_prefix(raw_name) local name = strip_ac_prefix(raw_name)
register_atom(out, "comp_bare", line_of(pos), name, body, body_off, raw_name, pos, after_paren) register_atom(out, "comp_bare", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
return after_brace return after_brace
end end
@@ -1196,7 +1281,7 @@ local function parse_mips_atom_comp_proc(source, pos, ident_end, line_of, out)
-- Position of body[1] in source = open_paren + 1 (start of inner) + last_brace_pos + 1 (past '{'). -- Position of body[1] in source = open_paren + 1 (start of inner) + last_brace_pos + 1 (past '{').
local body_off = open_paren + 2 + last_brace_pos local body_off = open_paren + 2 + last_brace_pos
register_atom(out, "comp_proc", line_of(pos), name, body, body_off, raw_name, pos, after_paren) register_atom(out, "comp_proc", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
return after_paren return after_paren
end end
@@ -1218,7 +1303,7 @@ local function parse_mips_code(source, pos, ident_end, line_of, out)
local atom_name = next_ident:sub(6) local atom_name = next_ident:sub(6)
local body, after_brace, body_off = find_body_braces(source, next_after, ident_end) local body, after_brace, body_off = find_body_braces(source, next_after, ident_end)
if not body then return after_brace end if not body then return after_brace end
register_raw_atom(out, line_of(pos), atom_name, body, body_off, atom_name, pos, "unrelated") register_raw_atom(out, line_of(pos), atom_name, body, body_off, atom_name, pos)
return after_brace return after_brace
end end
@@ -1317,7 +1402,7 @@ end
--- 4. `typedef <type> TSet_(<name>);` duffle TSet_ convention. --- 4. `typedef <type> TSet_(<name>);` duffle TSet_ convention.
--- Strips TSet_ wrapper; adds to type_name_registry (kind="typedef") with underlying_type=<type>. --- Strips TSet_ wrapper; adds to type_name_registry (kind="typedef") with underlying_type=<type>.
--- ---
--- All four shapes also associate an "unrelated" skip-over marker (the existing behavior — typedef declarations don't carry atom_dbg_skip_over). --- All four shapes also attach an "unrelated" debug-skip marker (the existing behavior — typedef declarations don't carry atom_dbg_skip).
--- @param source string --- @param source string
--- @param pos integer --- @param pos integer
--- @param ident_end integer --- @param ident_end integer
@@ -1338,7 +1423,7 @@ local function parse_typedef_binds(source, pos, ident_end, line_of, out)
local body, after_brace = find_body_braces(source, after_paren, open_paren + 1) local body, after_brace = find_body_braces(source, after_paren, open_paren + 1)
if not body then return after_brace end if not body then return after_brace end
register_struct_type(body, name, pos, line_of, out) register_struct_type(body, name, pos, line_of, out)
associate_skip_over_marker(out, name, name, "unrelated", line_of(pos), pos) attach_debug_skip_marker(out, "unrelated")
return after_brace return after_brace
-- ── Shape 2: `typedef Enum_(<underlying>, <name>) { <body> } <alias>;` -- ── Shape 2: `typedef Enum_(<underlying>, <name>) { <body> } <alias>;`
@@ -1354,7 +1439,7 @@ local function parse_typedef_binds(source, pos, ident_end, line_of, out)
local body, after_brace = find_body_braces(source, after_paren, open_paren + 1) local body, after_brace = find_body_braces(source, after_paren, open_paren + 1)
if not body then return after_brace end if not body then return after_brace end
register_enum_type(underlying, name, body, pos, line_of, out) register_enum_type(underlying, name, body, pos, line_of, out)
associate_skip_over_marker(out, name, name, "unrelated", line_of(pos), pos) attach_debug_skip_marker(out, "unrelated")
return after_brace return after_brace
end end
@@ -1390,7 +1475,7 @@ local function parse_typedef_binds(source, pos, ident_end, line_of, out)
-- Empty underlying span is acceptable; the TSet_ wrapper itself -- Empty underlying span is acceptable; the TSet_ wrapper itself
-- encodes the alias identity (per the duffle TSet_ convention). -- encodes the alias identity (per the duffle TSet_ convention).
register_typedef_alias("", tset_name, pos, line_of, out) register_typedef_alias("", tset_name, pos, line_of, out)
associate_skip_over_marker(out, tset_name, tset_name, "unrelated", line_of(pos), pos) attach_debug_skip_marker(out, "unrelated")
return after_paren return after_paren
end end
@@ -1433,7 +1518,7 @@ local function parse_typedef_binds(source, pos, ident_end, line_of, out)
local underlying_span = source:sub(after_typedef, tset_pos - 1) local underlying_span = source:sub(after_typedef, tset_pos - 1)
local underlying = duffle.trim(underlying_span) local underlying = duffle.trim(underlying_span)
register_typedef_alias(underlying, tset_arg, pos, line_of, out) register_typedef_alias(underlying, tset_arg, pos, line_of, out)
associate_skip_over_marker(out, tset_arg, tset_arg, "unrelated", line_of(pos), pos) attach_debug_skip_marker(out, "unrelated")
return tset_arg_end or (semi_pos + 1) return tset_arg_end or (semi_pos + 1)
end end
@@ -1442,7 +1527,7 @@ local function parse_typedef_binds(source, pos, ident_end, line_of, out)
local underlying_span = source:sub(after_typedef, last_ident_pos - 1) local underlying_span = source:sub(after_typedef, last_ident_pos - 1)
local underlying = duffle.trim(underlying_span) local underlying = duffle.trim(underlying_span)
register_typedef_alias(underlying, last_ident, pos, line_of, out) register_typedef_alias(underlying, last_ident, pos, line_of, out)
associate_skip_over_marker(out, last_ident, last_ident, "unrelated", line_of(pos), pos) attach_debug_skip_marker(out, "unrelated")
return last_ident_end return last_ident_end
end end
@@ -1630,7 +1715,9 @@ local DECL_PARSERS = {
MipsAtom_ = parse_mips_atom, MipsAtom_ = parse_mips_atom,
MipsAtomComp_ = parse_mips_atom_comp, MipsAtomComp_ = parse_mips_atom_comp,
MipsAtomComp_Proc_ = parse_mips_atom_comp_proc, MipsAtomComp_Proc_ = parse_mips_atom_comp_proc,
atom_dbg_skip_over = parse_skip_over_marker, -- `atom_dbg_skip` is the only debug-skip parser entry. Every other
-- identifier follows the ordinary unrelated-token path; there is no alias.
atom_dbg_skip = parse_dbg_skip_marker,
atom_dbg_reg_default = parse_atom_dbg_reg_default, atom_dbg_reg_default = parse_atom_dbg_reg_default,
MipsCode = parse_mips_code, MipsCode = parse_mips_code,
typedef = parse_typedef_binds, typedef = parse_typedef_binds,
@@ -1639,6 +1726,9 @@ local DECL_PARSERS = {
enum = parse_enum, enum = parse_enum,
} }
-- Only the bare `atom_dbg_skip` marker reaches `parse_dbg_skip_marker`.
-- Unknown identifiers follow the same unrelated-token path as every other unsupported source token.
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- The single source walker -- The single source walker
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -1649,7 +1739,7 @@ local DECL_PARSERS = {
--- @param source_file string|nil -- absolute source path (forwarded into AliasEntry.source_file) --- @param source_file string|nil -- absolute source path (forwarded into AliasEntry.source_file)
--- @param code_macros table|nil -- cross-source `R_*_Code` registry; nil = local-only --- @param code_macros table|nil -- cross-source `R_*_Code` registry; nil = local-only
--- @param code_macro_bodies table|nil -- cross-source raw RHS body table; nil = local-only --- @param code_macro_bodies table|nil -- cross-source raw RHS body table; nil = local-only
--- @return table -- SourceScan { atoms, raw_atoms, binds, atom_infos, macros, skip_over, line_of, register_alias_registry, _code_macros, _code_macro_bodies } --- @return table -- SourceScan { atoms, raw_atoms, binds, atom_infos, macros, debug_skip_markers, line_of, register_alias_registry, _code_macros, _code_macro_bodies }
local function scan_source(source, source_file, code_macros, code_macro_bodies) local function scan_source(source, source_file, code_macros, code_macro_bodies)
local line_of = duffle.LineIndex(source) local line_of = duffle.LineIndex(source)
local out = { local out = {
@@ -1658,11 +1748,9 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
binds = {}, binds = {},
atom_infos = {}, atom_infos = {},
macros = {}, macros = {},
skip_over = { -- Raw marker evidence for annotation validation. The `debug_skip` boolean
atoms = {}, -- is stamped on the declaration record itself; the projection lives on AtomEntry.debug_skip.
components = {}, debug_skip_markers = {},
markers = {},
},
types = {}, types = {},
atom_views = {}, atom_views = {},
line_of = line_of, line_of = line_of,
@@ -1709,26 +1797,27 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
if parser then if parser then
pos = parser(source, pos, ident_end, line_of, out) pos = parser(source, pos, ident_end, line_of, out)
else else
-- A component-procedure declaration has an FI_ signature before MipsAtomComp_Proc_; keep the marker pending across that prelude. -- Unsupported identifiers follow the unrelated-token path. If a
-- Any other identifier begins an unrelated declaration/construct and consumes the marker so it cannot drift to a later atom. -- pending marker is still open, consume it so it cannot drift to a
local markers = out.skip_over.markers -- later declaration. Unsupported identifiers never create marker records.
local markers = out.debug_skip_markers
local marker = markers[#markers] local marker = markers[#markers]
if marker and marker.pending then if marker and marker.pending then
if ident == "FI_" then if ident == "FI_" then
marker.proc_prelude = true marker.proc_prelude = true
elseif not marker.proc_prelude then elseif not marker.proc_prelude then
associate_skip_over_marker(out, ident, ident, "unrelated", line_of(pos), pos) attach_debug_skip_marker(out, "unrelated")
end end
end end
pos = ident_end pos = ident_end
end end
else else
local markers = out.skip_over.markers local markers = out.debug_skip_markers
local marker = markers[#markers] local marker = markers[#markers]
if marker and marker.pending and marker.proc_prelude then if marker and marker.pending and marker.proc_prelude then
local c = source:sub(pos, pos) local c = source:sub(pos, pos)
if c == "{" or c == ";" then if c == "{" or c == ";" then
associate_skip_over_marker(out, c, c, "unrelated", line_of(pos), pos) attach_debug_skip_marker(out, "unrelated")
end end
end end
pos = pos + 1 pos = pos + 1
@@ -1749,8 +1838,8 @@ end
-- Corpus merge — first-wins lookup identity + typed collisions -- Corpus merge — first-wins lookup identity + typed collisions
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- These helpers run ONCE per `M.run` invocation, after every per-source scan has attached `src.scan`. -- These helpers run ONCE per `M.run` invocation, after every per-source scan has attached `src.scan`.
-- They merge per-source scans into the canonical `ctx.shared.corpus.*` registries. -- They merge per-source scans into the `ctx.shared.corpus.*` registries.
-- The corpus is the source of truth; `src.scan` keeps the source-local projection for the duration of the run but the cross-source visibility lives on `corpus`. -- `src.scan` keeps the source-local projection for the duration of the run; the cross-source visibility lives on `corpus`.
-- Build a deterministic site record (path + line) from a per-source entry. -- Build a deterministic site record (path + line) from a per-source entry.
-- Falls back to the placeholder when an entry lacks a recorded source file or line. -- Falls back to the placeholder when an entry lacks a recorded source file or line.
@@ -1859,8 +1948,8 @@ local function phase_shape(entry)
end end
-- Merge a new declaration site into a registry following the first-wins discipline. -- Merge a new declaration site into a registry following the first-wins discipline.
-- * first declaration: Entry becomes the canonical corpus entry (entry.sites initialized). -- * first declaration: Entry becomes the corpus entry (entry.sites initialized).
-- * identical subsequent: Append the new site to entry.sites (no collision). -- * identical subsequent: Append the new site to entry.sites.
-- * conflicting shape: Keep first entry, append ONE typed collision record with shape diff. -- * conflicting shape: Keep first entry, append ONE typed collision record with shape diff.
local function merge_named_with_sites(registry, name, new_entry, site, collisions, kind, shape_fn) local function merge_named_with_sites(registry, name, new_entry, site, collisions, kind, shape_fn)
if registry[name] == nil then if registry[name] == nil then
@@ -1889,8 +1978,8 @@ local function merge_named_with_sites(registry, name, new_entry, site, collision
} }
end end
-- Merge per-source scans into the canonical corpus registries. -- Merge per-source scans into the corpus registries.
-- Iterates `corpus.source_order` (not `ctx.sources`) — the corpus is the source of truth. -- Iterates `corpus.source_order` (not `ctx.sources`).
-- Each source owns only its `src.scan`; the corpus owns the cross-source lookup tables. -- Each source owns only its `src.scan`; the corpus owns the cross-source lookup tables.
local function merge_corpus_registries(corpus) local function merge_corpus_registries(corpus)
-- Ensure every expected corpus table exists (the fixture_ctx seeds most of these, -- Ensure every expected corpus table exists (the fixture_ctx seeds most of these,
@@ -2003,8 +2092,7 @@ local M = {}
--- No output files; this is a pure in-memory pre-processing pass. --- No output files; this is a pure in-memory pre-processing pass.
--- ---
--- Runs in 5 phases. --- Runs in 5 phases.
--- Resolve: Resolve the canonical source order from `ctx.shared.corpus.source_order`. --- Resolve: Source order from `ctx.shared.corpus.source_order` (the corpus owns it; the check below enforces the invariant).
--- The canonical corpus is the SOLE source of truth; no `ctx.sources` alias is consulted and no per-source fallback synthesis is performed.
--- Pass 1a: `scan_source_pre_pass` over every source, populating LOCAL `code_macros` AND LOCAL `code_macro_bodies` tables. --- Pass 1a: `scan_source_pre_pass` over every source, populating LOCAL `code_macros` AND LOCAL `code_macro_bodies` tables.
--- The bodies table holds the raw post-`=` text of every `#define R_*_Code` line (cross-source) --- The bodies table holds the raw post-`=` text of every `#define R_*_Code` line (cross-source)
--- so the chain walker can fall back when the defining `#define` lives in a different source than the chain call site. --- so the chain walker can fall back when the defining `#define` lives in a different source than the chain call site.
@@ -2013,8 +2101,8 @@ local M = {}
--- Pass 2: The full `scan_source(source, source_file, code_macros, code_macro_bodies)` walk per source. The per-source `src.scan` payload includes the source-local registries --- Pass 2: The full `scan_source(source, source_file, code_macros, code_macro_bodies)` walk per source. The per-source `src.scan` payload includes the source-local registries
--- (register_alias_registry, type_name_registry, atom_views, atom_ctxs, atom_phases, binds, atoms, atom_infos, ...). --- (register_alias_registry, type_name_registry, atom_views, atom_ctxs, atom_phases, binds, atoms, atom_infos, ...).
--- Strip: Strip `src.scan._code_macros`, `src.scan._code_macro_bodies`, and the `_source_file` pointer. --- Strip: Strip `src.scan._code_macros`, `src.scan._code_macro_bodies`, and the `_source_file` pointer.
--- The LOCAL tables `code_macros` and `code_macro_bodies` go out of scope here; they MUST NOT appear on `ctx.shared`, `ctx.shared.corpus`, or any `src.scan` after this point. --- The LOCAL tables `code_macros` and `code_macro_bodies` stay confined to this function; they go out of scope on return.
--- Merge: Iterate `ctx.shared.corpus.source_order` in declared order. For every source's local registry, first-wins lookup identity (entry from the first declaration site becomes the canonical corpus entry); --- Merge: Iterate `ctx.shared.corpus.source_order` in declared order. For every source's local registry, first-wins lookup identity (entry from the first declaration site becomes the corpus entry);
--- identical shapes coalesce by appending the declaration site; conflicting shapes keep the first lookup entry and append ONE typed collision record with shape diff. --- identical shapes coalesce by appending the declaration site; conflicting shapes keep the first lookup entry and append ONE typed collision record with shape diff.
--- Populate `register_alias_registry`, `type_name_registry`, `binds_by_name`, `atoms_by_name`, `atom_views`, `atom_ctxs`, `atom_phases`. --- Populate `register_alias_registry`, `type_name_registry`, `binds_by_name`, `atoms_by_name`, `atom_views`, `atom_ctxs`, `atom_phases`.
--- `atom_infos` ALWAYS appends every record (preserving source order + duplicates for annotation evidence). --- `atom_infos` ALWAYS appends every record (preserving source order + duplicates for annotation evidence).
@@ -2023,14 +2111,12 @@ local M = {}
--- @return PassResult --- @return PassResult
function M.run(ctx) function M.run(ctx)
-- The cross-source _code_macros / _code_macro_bodies tables are LOCAL to this run. -- The cross-source _code_macros / _code_macro_bodies tables are LOCAL to this run.
-- They are shared across source scans ONLY long enough to resolve cross-source R_*_Code chains, then DISCARDED. -- They live across source scans only long enough to resolve cross-source R_*_Code chains, then go out of scope on M.run return.
-- They MUST NOT appear on ctx.shared, ctx.shared.corpus, or any src.scan after -- The Lua GC reclaims them; nothing here survives onto ctx.shared, ctx.shared.corpus, or any src.scan.
-- this function returns.
local code_macros = {} local code_macros = {}
local code_macro_bodies = {} local code_macro_bodies = {}
-- Resolve the canonical source list. The corpus owns the authoritative source_order; no legacy alias is consulted and no per-source fallback synthesis is performed. -- Canonical-corpus check (see the docstring Resolve phase). The corpus is the only source of source_order.
-- A context without `ctx.shared.corpus` is rejected with an explicit canonical-corpus message so callers migrate to the canonical context (no production compatibility layer).
ctx.shared = ctx.shared or {} ctx.shared = ctx.shared or {}
local corpus = ctx.shared.corpus local corpus = ctx.shared.corpus
if not corpus or type(corpus.source_order) ~= "table" then if not corpus or type(corpus.source_order) ~= "table" then
@@ -2070,20 +2156,16 @@ function M.run(ctx)
src.scan._code_macro_bodies = nil src.scan._code_macro_bodies = nil
src.scan._source_file = nil src.scan._source_file = nil
end end
-- Pre-tokenize each atom body once (plex: single source of truth). -- Pre-tokenize each atom body once (plex: cache lives in duffle.lua; downstream passes read from `atom.body_tokens` instead of calling `split_top_level_commas` / `tokenize_body` independently).
-- Downstream passes (offsets, word-counts, components, static-analysis) read from `atom.body_tokens` instead of calling `split_top_level_commas` / `tokenize_body` independently. -- Re-access is O(1) thanks to the memoization.
-- The tokens are memoized in duffle.lua's cache, so re-access is O(1).
for _, atom in ipairs(src.scan.atoms) do atom.body_tokens = duffle.tokenize_body(atom.body) end for _, atom in ipairs(src.scan.atoms) do atom.body_tokens = duffle.tokenize_body(atom.body) end
for _, atom in ipairs(src.scan.raw_atoms or {}) do atom.body_tokens = duffle.tokenize_body(atom.body) end for _, atom in ipairs(src.scan.raw_atoms or {}) do atom.body_tokens = duffle.tokenize_body(atom.body) end
end end
-- Merge per-source scans into the canonical corpus registries. -- Merge per-source scans into the corpus registries (see merge_corpus_registries for first-wins + collision discipline).
-- First-wins lookup identity + collision discipline (see merge_corpus_registries).
-- The corpus is always present; no conditional / fallback path.
merge_corpus_registries(corpus) merge_corpus_registries(corpus)
-- code_macros and code_macro_bodies go out of scope here; their references are not captured on corpus, ctx.shared, or any src.scan. -- code_macros and code_macro_bodies are function-local; the GC reclaims them on M.run return.
-- The Lua GC reclaims them on M.run return.
return { outputs = {}, errors = {}, warnings = {} } return { outputs = {}, errors = {}, warnings = {} }
end end
+66 -88
View File
@@ -1,21 +1,23 @@
--- passes/static_analysis.lua — Per-atom static-analysis checks. --- passes/static_analysis.lua — Per-atom static-analysis checks.
--- ---
--- Ownership: `ctx.shared.corpus` is the canonical merged registry; per-source fallback synthesis is rejected.
--- `atom.paths` supplies the emitted and analysis projections consumed by this pass.
---
--- Per-atom rules: --- Per-atom rules:
--- 1. transfer_hazards: A single forward walker (`analyze_hardware_relations`) reads `atom.paths.word_events` --- 1. transfer_hazards: A single forward walker (`analyze_hardware_relations`) reads `atom.paths.word_events` once per atom.
--- once per atom. For each emitted word event it (a) inspects pending CPU/COP0/COP2/GTE relations against --- For each emitted word event it (a) inspects pending CPU/COP0/COP2/GTE relations against the event as CONSUMER
--- the event as CONSUMER (recording a hazard on `atom.paths.hazards` when the producer→consumer gap is below --- (recording a hazard on `atom.paths.hazards` when the producer→consumer gap is below the required retire-slot count),
--- the required retire-slot count), (b) applies the event's GPR value effects (`duffle.INSTRUCTION_GPR_EFFECTS`) --- (b) applies the event's GPR value effects (`duffle.INSTRUCTION_GPR_EFFECTS`) to `atom.paths.forward_state.gpr_values`,
--- to `atom.paths.forward_state.gpr_values`, applies bounded constant propagation, and stages --- applies bounded constant propagation, and stages matching relation rows as PRODUCERS (with `destination_match` filters, e.g. for the IRGB fan-out).
--- matching relation rows as PRODUCERS (with `destination_match` filters, e.g. for the IRGB fan-out). The --- The `transfer_hazards` CHECK_RULES reader projects `atom.paths.hazards` into per-atom findings.
--- `transfer_hazards` CHECK_RULES reader projects `atom.paths.hazards` into per-atom findings without --- The reader does NOT re-walk source; this is the per-check purity contract.
--- re-walking source. The walker runs once per atom before the per-atom dispatch; the reader runs inside --- The walker runs once per atom before the per-atom dispatch; the reader runs inside the same dispatch.
--- the same dispatch.
--- 2. control_transfer_delay_slot_use: For every emitted branch/jump/call encoder in `duffle.CONTROL_TRANSFER_DELAY_SLOT_POLICIES` --- 2. control_transfer_delay_slot_use: For every emitted branch/jump/call encoder in `duffle.CONTROL_TRANSFER_DELAY_SLOT_POLICIES`
--- (the six `branch_*` encoders plus `jump` / `jump_reg` / `jump_link` / `call_reg` / `call_addr`), --- (the six `branch_*` encoders plus `jump` / `jump_reg` / `jump_link` / `call_reg` / `call_addr`),
--- inspect the next emitted event in `atom.paths.word_events`. --- inspect the next emitted event in `atom.paths.word_events`.
--- Emit an `info`-severity finding when the successor is `nop` or absent (the next emitted word IS the hardware delay slot). --- Emit an `info`-severity finding when the successor is `nop` or absent (the next emitted word IS the hardware delay slot).
--- `jump_reg(R_AtomJmp)` is suppressed by policy (the fixed `mac_yield()` handshake). --- `jump_reg(R_AtomJmp)` is suppressed by policy (the fixed `mac_yield()` handshake).
--- `nop2` needs no special case: `expand_word_events` emits two `nop` events for it, so the first expansion is the hardware delay slot. --- `nop2` needs no special case: emission-model emits two `nop` events for it, so the first expansion is the hardware delay slot.
--- `atom_label` also needs no special case (zero events). --- `atom_label` also needs no special case (zero events).
--- 3. mac_yield uniformity: Every atom body must contain exactly one `mac_yield()` call (control transfer pattern). --- 3. mac_yield uniformity: Every atom body must contain exactly one `mac_yield()` call (control transfer pattern).
--- 4. Binding handoff: Every `atom_bind(Binds_X)` must reference a `typedef Struct_(Binds_X) { ... }` declaration. --- 4. Binding handoff: Every `atom_bind(Binds_X)` must reference a `typedef Struct_(Binds_X) { ... }` declaration.
@@ -28,12 +30,8 @@
--- must be in `corpus.register_alias_registry`. --- must be in `corpus.register_alias_registry`.
--- 9. atom_type_consistency: Every `reg_type_overrides[R_X].type_name` must resolve in `corpus.type_name_registry`. --- 9. atom_type_consistency: Every `reg_type_overrides[R_X].type_name` must resolve in `corpus.type_name_registry`.
--- 10. binds_no_substruct_deref: Every `load_word(R_A, R_B, O_(Type, Field))` and `store_word(...)` in every atom body must reference a leaf scalar --- 10. binds_no_substruct_deref: Every `load_word(R_A, R_B, O_(Type, Field))` and `store_word(...)` in every atom body must reference a leaf scalar
--- (pointer-to-struct counts as leaf; nested struct members do NOT). --- (pointer-to-struct counts as leaf; nested struct members fail the leaf test).
--- ---
--- `enum_alias_membership` already iterates `ai.reads` and `ai.writes` against
--- `corpus.register_alias_registry`, so a duplicate precedence-class check
--- only produced duplicate findings. The CHECK_RULES row and the helper
--- function are gone; no public/private surface retains that name.)
--- ---
--- Findings carry an explicit `kind` ("error" / "warning" / "info"). --- Findings carry an explicit `kind` ("error" / "warning" / "info").
--- The renderer maintains three independent severity collections; `info` is never folded into warnings. --- The renderer maintains three independent severity collections; `info` is never folded into warnings.
@@ -113,7 +111,6 @@ local OUTPUT_EXTENSION = ".static_analysis.txt"
--- @field project_root string --- @field project_root string
--- @field upstream table<string, table> --- @field upstream table<string, table>
--- @field flags table --- @field flags table
--- @field dry_run boolean
--- @field verbose boolean --- @field verbose boolean
--- @class PassResult --- @class PassResult
@@ -304,7 +301,7 @@ end
-- Only the matching row stages (the non-matching row is ignored for that event). -- Only the matching row stages (the non-matching row is ignored for that event).
-- --
-- After the walker runs, the `transfer_hazards` CHECK_RULES reader (`check_transfer_hazards`) copies every entry on `atom.paths.hazards` into the per-atom findings list. -- After the walker runs, the `transfer_hazards` CHECK_RULES reader (`check_transfer_hazards`) copies every entry on `atom.paths.hazards` into the per-atom findings list.
-- The reader does NOT re-walk source or re-classify tokens; it is a pure projection of the walker's output. -- The first `transfer_hazards` reader comment above records the projection contract.
-- --
-- The walker is called once before the CHECK_RULES per-atom dispatch (see `validate()`); -- The walker is called once before the CHECK_RULES per-atom dispatch (see `validate()`);
-- The reader runs as part of the same CHECK_RULES dispatch so its findings land in `findings` alongside the other checks. -- The reader runs as part of the same CHECK_RULES dispatch so its findings land in `findings` alongside the other checks.
@@ -326,7 +323,7 @@ local function is_cop2_consumer_of(consumer_event, destination, producer_rel)
for _, pos in ipairs(args) do for _, pos in ipairs(args) do
if pos == destination then return true end if pos == destination then return true end
end end
-- Match via the command's input set: the consumer's encoder resolves to a canonical `gte_cmdw_*` -- Match via the command's input set: the consumer encoder resolves to a `gte_cmdw_*`
-- short form whose `duffle.GTE_COMMAND_INPUTS` entry includes the destination (or a fan-out target). -- short form whose `duffle.GTE_COMMAND_INPUTS` entry includes the destination (or a fan-out target).
local aliases = duffle.GTE_COMMAND_ALIASES or {} local aliases = duffle.GTE_COMMAND_ALIASES or {}
local canonical = aliases[consumer_token] or consumer_token local canonical = aliases[consumer_token] or consumer_token
@@ -535,7 +532,7 @@ local function apply_gpr_effects(ev_ident, ev_args, forward_state)
end end
end end
-- Look up the canonical alias of a GTE command ident. -- Look up the alias of a GTE command ident.
-- Defaults to the input ident so unknown idents surface rather than silently inheriting a 0-cycle command input set. -- Defaults to the input ident so unknown idents surface rather than silently inheriting a 0-cycle command input set.
local function canonical_command(ident) local function canonical_command(ident)
local aliases = duffle.GTE_COMMAND_ALIASES or {} local aliases = duffle.GTE_COMMAND_ALIASES or {}
@@ -714,10 +711,9 @@ local function analyze_hardware_relations(atom)
local ev_line = ev.body_line or ev.line or ev.def_line or 0 local ev_line = ev.body_line or ev.line or ev.def_line or 0
local ev_source = ev.def_path or ev.source or "" local ev_source = ev.def_path or ev.source or ""
local ev_args = ev.args or {} local ev_args = ev.args or {}
-- `word_events` use `i` as the 0-based word index across the entire expansion); -- `word_events` use `i` as the 0-based word index across the entire expansion.
-- The legacy `duffle.expand_word_events` walker emits `word`. -- Default to 0 if the field is absent (the producer's own word).
-- Either is accepted; unknown defaults to 0 (the producer's own word). local ev_word = ev.i or 0
local ev_word = ev.i or ev.word or 0
-- Read a Status source, then apply current-event GPR writes, then consume a pending CU2 transition at the first relevant COP2 use. -- Read a Status source, then apply current-event GPR writes, then consume a pending CU2 transition at the first relevant COP2 use.
-- Both operations are part of this one event walk. -- Both operations are part of this one event walk.
@@ -940,7 +936,7 @@ end
-- --
-- The single forward walker `analyze_hardware_relations` (defined above) has already populated `atom.paths.hazards`. -- The single forward walker `analyze_hardware_relations` (defined above) has already populated `atom.paths.hazards`.
-- This check copies every entry on that list into the per-atom `findings` table. -- This check copies every entry on that list into the per-atom `findings` table.
-- It does NOT re-walk source / re-classify tokens; it is a pure projection of the walker's output. -- The first `transfer_hazards` reader comment above records the projection contract.
-- --
-- The walker also populates `atom.paths.relations` (one entry per satisfied-or-violated relation touch) and `atom.paths.forward_state` (the GPR-value lattice). -- The walker also populates `atom.paths.relations` (one entry per satisfied-or-violated relation touch) and `atom.paths.forward_state` (the GPR-value lattice).
-- Neither of those is rendered as a finding here; bounded-value rules and LWC2 unknown edges share on top of the same forward walker and adds additional readers. -- Neither of those is rendered as a finding here; bounded-value rules and LWC2 unknown edges share on top of the same forward walker and adds additional readers.
@@ -962,7 +958,7 @@ end
-- --
-- The forward walker stages post-command latch relations on `atom.paths.hazards` with `relation_id = "command_latch_input"`. -- The forward walker stages post-command latch relations on `atom.paths.hazards` with `relation_id = "command_latch_input"`.
-- This reader filters those entries and re-emits them under the `gte_input_latch` check name so the test contract can target them independently of the transfer_hazards check. -- This reader filters those entries and re-emits them under the `gte_input_latch` check name so the test contract can target them independently of the transfer_hazards check.
-- The reader does NOT re-walk source tokens or build its own pending state; it is a pure projection of the walker's output. -- The first `transfer_hazards` reader comment above records the projection contract.
-- ───────────────────────────────────────────────────────────────────────── -- ─────────────────────────────────────────────────────────────────────────
local function check_gte_input_latch(atom, _pipe_ctx, findings) local function check_gte_input_latch(atom, _pipe_ctx, findings)
@@ -992,7 +988,7 @@ end
-- A subsequent MFC2 (or any encoder that reads a C2 register) that picks the WRONG register for the active role emits a `result_role_mismatch` warning. -- A subsequent MFC2 (or any encoder that reads a C2 register) that picks the WRONG register for the active role emits a `result_role_mismatch` warning.
-- For example, reading `C2_SXY0` after RTPS is wrong: the `latest_screen_xy` role is `C2_SXY2`. -- For example, reading `C2_SXY0` after RTPS is wrong: the `latest_screen_xy` role is `C2_SXY2`.
-- --
-- The reader does NOT re-walk source tokens; it consumes `forward_state.post_command_roles` and `atom.paths.word_events` only. -- The first `transfer_hazards` reader comment above records the projection contract.
-- ───────────────────────────────────────────────────────────────────────── -- ─────────────────────────────────────────────────────────────────────────
local function check_gte_result_position(atom, _pipe_ctx, findings) local function check_gte_result_position(atom, _pipe_ctx, findings)
@@ -1032,12 +1028,12 @@ local function check_gte_result_position(atom, _pipe_ctx, findings)
-- For each word event whose encoder is `gte_mv_from_data_r`, look up the register being read in `forward_state.post_command_roles`. -- For each word event whose encoder is `gte_mv_from_data_r`, look up the register being read in `forward_state.post_command_roles`.
-- If a role is set, the reader's register must match the role's register (the registered "latest_<role>" target). -- If a role is set, the reader's register must match the role's register (the registered "latest_<role>" target).
for _, ev in ipairs(events) do for _, ev in ipairs(events) do
local ev_ident = ev.encoder or ev.ident local ev_ident = ev.encoder
if ev_ident == "gte_mv_from_data_r" then if ev_ident == "gte_mv_from_data_r" then
local args = ev.args or {} local args = ev.args or {}
local reg = args[2] local reg = args[2]
-- Find any post-command `latest_screen_xy` role entry recorded by a prior command. -- Find any post-command `latest_screen_xy` role entry recorded by a prior command.
-- The newest projected screen coordinate is recorded under the command's canonical name. -- The newest projected screen coordinate is recorded under the command name.
-- Reading from C2_SXY0 (the older projection slot) when a `latest_screen_xy` role was set to C2_SXY2 by RTPS / RTPT is a semantic mismatch. -- Reading from C2_SXY0 (the older projection slot) when a `latest_screen_xy` role was set to C2_SXY2 by RTPS / RTPT is a semantic mismatch.
local latest_screen_xy_entry = nil local latest_screen_xy_entry = nil
for r, e in pairs(forward.post_command_roles or {}) do for r, e in pairs(forward.post_command_roles or {}) do
@@ -1085,10 +1081,10 @@ end
-- (the nop is needed to retire the relation, even if it can be replaced by independent useful work). -- (the nop is needed to retire the relation, even if it can be replaced by independent useful work).
-- * `modeled-redundant`: no modeled relation is pending immediately before the nop (the nop is a redundant hazard). -- * `modeled-redundant`: no modeled relation is pending immediately before the nop (the nop is a redundant hazard).
-- --
-- Branch/jump delay-slot NOPs are NOT classified by this check (they are exclusively owned by `control_transfer_delay_slot_use`). -- Branch/jump delay-slot NOPs belong to `control_transfer_delay_slot_use`, so this check leaves them unclassified.
-- The fixed `mac_yield()` handshake (`jump_reg(R_AtomJmp), nop`) is preserved as suppressed. -- The fixed `mac_yield()` handshake (`jump_reg(R_AtomJmp), nop`) is preserved as suppressed.
-- --
-- The reader does NOT re-walk source tokens; it consumes `forward_state.pending` snapshots and `atom.paths.word_events`. -- The first `transfer_hazards` reader comment above records the projection contract.
-- ───────────────────────────────────────────────────────────────────────── -- ─────────────────────────────────────────────────────────────────────────
local function check_hazard_nop_use(atom, _pipe_ctx, findings) local function check_hazard_nop_use(atom, _pipe_ctx, findings)
@@ -1101,14 +1097,14 @@ local function check_hazard_nop_use(atom, _pipe_ctx, findings)
local pending_snapshot = {} local pending_snapshot = {}
local prev_ev = nil local prev_ev = nil
for event_idx, ev in ipairs(events) do for event_idx, ev in ipairs(events) do
local ev_ident = ev.encoder or ev.ident or "" local ev_ident = ev.encoder or ""
local ev_args = ev.args or {} local ev_args = ev.args or {}
local ev_word = ev.i or ev.word or 0 local ev_word = ev.i or 0
-- Classify the nop BEFORE its event is applied to the pending state. -- Classify the nop BEFORE its event is applied to the pending state.
if ev_ident == "nop" and prev_ev ~= nil then if ev_ident == "nop" and prev_ev ~= nil then
-- Skip BD-slot nops: they are exclusively owned by control_transfer_delay_slot_use. -- Skip BD-slot nops: they are exclusively owned by control_transfer_delay_slot_use.
local prev_ident = prev_ev.encoder or prev_ev.ident or "" local prev_ident = prev_ev.encoder or ""
local prev_args = prev_ev.args or {} local prev_args = prev_ev.args or {}
local bd_policies = duffle.CONTROL_TRANSFER_DELAY_SLOT_POLICIES or {} local bd_policies = duffle.CONTROL_TRANSFER_DELAY_SLOT_POLICIES or {}
local is_bd_slot = false local is_bd_slot = false
@@ -1246,8 +1242,8 @@ end
-- ───────────────────────────────────────────────────────────────────────── -- ─────────────────────────────────────────────────────────────────────────
-- Check #1c: control-transfer delay-slot use. -- Check #1c: control-transfer delay-slot use.
-- --
-- Reads `atom.paths.word_events` (the semantic emitted-word stream from `duffle.expand_word_events`). -- Reads `atom.paths.word_events` (the semantic emitted-word stream from `passes/emission_model.lua`).
-- For each event whose `ident` is in `duffle.CONTROL_TRANSFER_DELAY_SLOT_POLICIES`, inspect the next emitted event in the SAME `events` array. -- For each event whose `encoder` is in `duffle.CONTROL_TRANSFER_DELAY_SLOT_POLICIES`, inspect the next emitted event in the SAME `events` array.
-- The next event is the hardware delay-slot word (the duffle pipeline already absorbs the BD-slot into the branch's cost in `analyze_atom_paths`. -- The next event is the hardware delay-slot word (the duffle pipeline already absorbs the BD-slot into the branch's cost in `analyze_atom_paths`.
-- This check observes, it does not reschedule. -- This check observes, it does not reschedule.
-- --
@@ -1260,7 +1256,7 @@ end
-- --
-- `pipe_ctx` is unused; the uniform `(atom, pipe_ctx, findings)` signature is preserved so the check plugs into -- `pipe_ctx` is unused; the uniform `(atom, pipe_ctx, findings)` signature is preserved so the check plugs into
-- the existing CHECK_RULES dispatch without modifying the per-atom loop or analyze_atom_paths. -- the existing CHECK_RULES dispatch without modifying the per-atom loop or analyze_atom_paths.
-- `expand_word_events` already normalizes `nop2` to two `nop` events and `atom_label` to zero events, so no special-case branching is needed for either. -- `passes/emission_model` already normalizes `nop2` to two `nop` events and `atom_label` to zero events, so no special-case branching is needed for either.
-- ───────────────────────────────────────────────────────────────────────── -- ─────────────────────────────────────────────────────────────────────────
local function check_control_transfer_delay_slot_use(atom, pipe_ctx, findings) local function check_control_transfer_delay_slot_use(atom, pipe_ctx, findings)
@@ -1408,8 +1404,8 @@ end
--- 2. Body MUST contain an `add_ui_self(R_TapePtr, S_(Binds_X))` (or equivalent advance by the struct's byte count). Missing = error. --- 2. Body MUST contain an `add_ui_self(R_TapePtr, S_(Binds_X))` (or equivalent advance by the struct's byte count). Missing = error.
--- 3. atom_bind(Binds_X) where Binds_X doesn't exist = error. --- 3. atom_bind(Binds_X) where Binds_X doesn't exist = error.
--- Per-atom: Verify the atom body reads every field of its `Binds_X` from R_TapePtr and advances R_TapePtr by S_(Binds_X). --- Per-atom: Verify the atom body reads every field of its `Binds_X` from R_TapePtr and advances R_TapePtr by S_(Binds_X).
--- Takes `(atom, pipe_ctx, findings)`; `pipe_ctx` carries the cross-atom --- Takes `(atom, pipe_ctx, findings)`; `pipe_ctx` carries the cross-atom `info_by_atom` + `binds_index` tables
--- `info_by_atom` + `binds_index` tables (built once by validate() before the per-atom loop). --- (built once by validate() before the per-atom loop).
--- `validate()` owns per-atom iteration; this function evaluates one atom. --- `validate()` owns per-atom iteration; this function evaluates one atom.
local function check_abi_handoff(atom, pipe_ctx, findings) local function check_abi_handoff(atom, pipe_ctx, findings)
local info = pipe_ctx.info_by_atom[atom.name] local info = pipe_ctx.info_by_atom[atom.name]
@@ -1665,9 +1661,8 @@ local function analyze_atom_paths(atom)
local succ, term = successors(tok_idx) local succ, term = successors(tok_idx)
if term then if term then
-- Terminator: record the path's cycle sum. -- Terminator: record the path's cycle sum.
-- We do NOT add the terminator token to `visited` a path ends here, so a different path that -- The terminator token stays out of `visited`, so another path reaching the same terminator remains a distinct path.
-- ALSO reaches this terminator is a legitimate new path (not a loop). -- Marking it visited would flag those legitimate paths as loops.
-- If we marked it visited, subsequent paths that reach the same terminator would be incorrectly flagged as loops.
path_count = path_count + 1 path_count = path_count + 1
if new_acc < cycles_min then cycles_min = new_acc end if new_acc < cycles_min then cycles_min = new_acc end
if new_acc > cycles_max then cycles_max = new_acc end if new_acc > cycles_max then cycles_max = new_acc end
@@ -1710,7 +1705,7 @@ end
--- (deduplicated across atoms so the warning section doesn't get spammed with N copies of "macro X not in duffle.INSTRUCTION_LATENCY"). --- (deduplicated across atoms so the warning section doesn't get spammed with N copies of "macro X not in duffle.INSTRUCTION_LATENCY").
--- Per-atom: emit one finding per unknown macro seen, deduplicated across atoms --- Per-atom: emit one finding per unknown macro seen, deduplicated across atoms
--- (so the warning section doesn't get spammed with N copies of "macro X not in duffle.INSTRUCTION_LATENCY"). --- (so the warning section doesn't get spammed with N copies of "macro X not in duffle.INSTRUCTION_LATENCY").
--- Reuses `analyze_atom_paths`'s per-atom unknown_macros discovery (it's the canonical place that walks tokens and computes per-token cycle costs). --- Reuses `analyze_atom_paths`'s per-atom unknown_macros discovery, which walks tokens and computes per-token cycle costs.
local function check_per_atom_cycle_budget(atom, pipe_ctx, findings) local function check_per_atom_cycle_budget(atom, pipe_ctx, findings)
local p = atom.paths or {} local p = atom.paths or {}
for _, name in ipairs(p.unknown_macros or {}) do for _, name in ipairs(p.unknown_macros or {}) do
@@ -1742,9 +1737,9 @@ end
-- The rule is intentionally permissive because the production `code/duffle/` and `code/gte_hello/` -- The rule is intentionally permissive because the production `code/duffle/` and `code/gte_hello/`
-- sources use R_* aliases in atom_reads / atom_writes that may not yet be opted in via the bare `atom_reg` marker. -- sources use R_* aliases in atom_reads / atom_writes that may not yet be opted in via the bare `atom_reg` marker.
-- R_TapePtr / R_AtomJmp / R_PrimCursor / R_FaceCursor / R_VertBase / R_OtBase ARE opted in. -- R_TapePtr / R_AtomJmp / R_PrimCursor / R_FaceCursor / R_VertBase / R_OtBase ARE opted in.
-- Raw C-ABI aliases like R_T0..R_T3 are intentionally NOT auto-included (per the prototype principle: -- Raw C-ABI aliases like R_T0..R_T3 require explicit opt-in; the prototype keeps wave-context registration explicit.
-- no auto-include of wave-context; explicit opt-in only). Warnings keep the build green -- no auto-include of wave-context; explicit opt-in only).
-- and report aliases that need explicit registration. -- Warnings keep the build green and report aliases that need explicit registration.
local function check_enum_alias_membership(_src, pipe_ctx, findings) local function check_enum_alias_membership(_src, pipe_ctx, findings)
local reg_registry = pipe_ctx.register_alias_registry or {} local reg_registry = pipe_ctx.register_alias_registry or {}
@@ -1842,7 +1837,7 @@ end
-- the `<Field>` MUST resolve to a leaf scalar of `<Type>`. A "leaf scalar" is: -- the `<Field>` MUST resolve to a leaf scalar of `<Type>`. A "leaf scalar" is:
-- * a non-struct field with `pointer_depth >= 1` (pointer-to-struct IS a leaf — the field is a pointer; the pointee is unrelated), OR -- * a non-struct field with `pointer_depth >= 1` (pointer-to-struct IS a leaf — the field is a pointer; the pointee is unrelated), OR
-- * a non-struct field whose type_name resolves to a typedef / enum / builtin in `type_name_registry`. -- * a non-struct field whose type_name resolves to a typedef / enum / builtin in `type_name_registry`.
-- A nested struct member (pointer_depth == 0 AND type_name resolves to a `kind = "struct"` registry entry) is NOT a leaf scalar and is flagged. -- A nested struct member (pointer_depth == 0 and type_name resolves to a `kind = "struct"` registry entry) fails the leaf-scalar test.
-- The check also flags fields whose Type has no `fields` table (typedefs and enums don't have fields — any Field reference against them is bogus) -- The check also flags fields whose Type has no `fields` table (typedefs and enums don't have fields — any Field reference against them is bogus)
-- and fields whose name doesn't appear in the resolved Type's fields array. -- and fields whose name doesn't appear in the resolved Type's fields array.
-- --
@@ -1864,7 +1859,7 @@ local function find_field_by_name(type_entry, field_name)
end end
-- True iff a (field, type_registry) pair is a leaf scalar (safe to dereference as a tape-payload field). -- True iff a (field, type_registry) pair is a leaf scalar (safe to dereference as a tape-payload field).
-- Pointer-to-X is always leaf; non-pointer struct members are NOT leaf. -- Pointer-to-X is always a leaf; non-pointer struct members fail the leaf test.
local function is_field_leaf(field, type_registry) local function is_field_leaf(field, type_registry)
if field.pointer_depth and field.pointer_depth > 0 then if field.pointer_depth and field.pointer_depth > 0 then
return true return true
@@ -1921,10 +1916,6 @@ local function check_binds_no_substruct_deref(_src, pipe_ctx, findings)
end end
end end
-- Because enum_alias_membership (Check #8) already iterates ai.reads / ai.writes against corpus.register_alias_registry.
-- The duplicate row produced redundant findings for the same off-registry register.
-- Neither a check function nor a CHECK_RULES row retains the name.
-- If a future regression reintroduces either, the test_canonical_corpus.lua grep sweep will surface it.
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- CHECK_RULES — data-driven check dispatch (Muratori: data over control flow) -- CHECK_RULES — data-driven check dispatch (Muratori: data over control flow)
@@ -1935,7 +1926,7 @@ end
-- per_atom(atom, pipe_ctx, findings) — runs once per atom inside validate()'s single loop -- per_atom(atom, pipe_ctx, findings) — runs once per atom inside validate()'s single loop
-- post(pipe_ctx, findings) — runs once after all per-atom calls complete -- post(pipe_ctx, findings) — runs once after all per-atom calls complete
-- per_macro(macro, wc, findings) — runs once per TAPE_WORDS / _Pragma macro declaration -- per_macro(macro, wc, findings) — runs once per TAPE_WORDS / _Pragma macro declaration
-- per_skip_marker(marker, pipe_ctx, findings) — runs once per src.scan.skip_over.markers entry -- per_skip_marker(marker, pipe_ctx, findings) — runs once per src.scan.debug_skip_markers entry
-- per_source(src, pipe_ctx, findings) — runs once per source AFTER the per-atom loop completes -- per_source(src, pipe_ctx, findings) — runs once per source AFTER the per-atom loop completes
-- (registry-driven rule; same CHECK_RULES table) -- (registry-driven rule; same CHECK_RULES table)
-- Each check is one table row and one `check_*` function. -- Each check is one table row and one `check_*` function.
@@ -1961,11 +1952,11 @@ local CHECK_RULES = {
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- Build the corpus-wide pipe_ctx ONCE per pass run. --- Build the corpus-wide pipe_ctx ONCE per pass run.
--- Reads the merged `corpus.*` registries (canonical cross-source lookups), and the corpus-wide `atom_infos` list (preserving source order + duplicates). --- Reads the merged `corpus.*` registries and the corpus-wide `atom_infos` list (preserving source order + duplicates).
--- The corpus is the source of truth; per-source scans retain body / declaration ownership via `src.scan` and the per-source `atoms` / `atom_infos` projections. --- The corpus supplies shared registries; `src.scan` and per-source projections retain body and declaration ownership.
--- ---
--- Ownership: A context without `ctx.shared.corpus` is rejected with an explicit canonical-corpus message. --- A context without `ctx.shared.corpus` is rejected with an explicit corpus message.
--- No per-source fallback synthesis is performed; callers MUST construct a canonical ctx through `build_ctx`. --- Callers construct the context through `build_ctx`.
--- @param ctx PassCtx --- @param ctx PassCtx
--- @return PipeCtx --- @return PipeCtx
local function build_corpus_pipe_ctx(ctx) local function build_corpus_pipe_ctx(ctx)
@@ -1976,9 +1967,9 @@ local function build_corpus_pipe_ctx(ctx)
.. "no per-source fallback is supported)", 0) .. "no per-source fallback is supported)", 0)
end end
-- The pipe_ctx views REFERENCE the corpus tables directly (no copies). -- The pipe_ctx views REFERENCE the corpus tables directly (no copies).
-- Every consumer of these fields observes mutations via the canonical corpus without independently mutable registry construction. -- Every consumer observes mutations through the corpus tables directly.
return { return {
-- Cross-source lookup tables (canonical corpus projections). -- Cross-source lookup tables.
register_alias_registry = corpus.register_alias_registry or {}, register_alias_registry = corpus.register_alias_registry or {},
type_name_registry = corpus.type_name_registry or {}, type_name_registry = corpus.type_name_registry or {},
atom_views = corpus.atom_views or {}, atom_views = corpus.atom_views or {},
@@ -1995,8 +1986,9 @@ end
local function validate(ctx, src, corpus_pipe_ctx) local function validate(ctx, src, corpus_pipe_ctx)
local scan = src.scan local scan = src.scan
-- Read the canonical corpus word_counts for the defensive fallback path below -- Read the corpus word_counts for the per-atom pipeline
-- (test-only: Focused tests that bypass emission-model and feed static_analysis still need word_events produced by the legacy `duffle.expand_word_events` walker). -- (`atom.paths.word_events` is the emitted projection).
local corpus = (ctx.shared and ctx.shared.corpus) or {} local corpus = (ctx.shared and ctx.shared.corpus) or {}
-- Read atoms + binds + atom_infos from the pre-scanned SourceScan payload. -- Read atoms + binds + atom_infos from the pre-scanned SourceScan payload.
@@ -2037,7 +2029,7 @@ local function validate(ctx, src, corpus_pipe_ctx)
register_alias_registry = corpus_pipe_ctx.register_alias_registry, register_alias_registry = corpus_pipe_ctx.register_alias_registry,
type_name_registry = corpus_pipe_ctx.type_name_registry, type_name_registry = corpus_pipe_ctx.type_name_registry,
} }
-- Shared cross-source component-body index is owned by the canonical corpus -- Shared cross-source component-body index is owned by the corpus
-- (`corpus.component_body_index`, populated by `passes/components.lua`). -- (`corpus.component_body_index`, populated by `passes/components.lua`).
-- Per-atom checks consume the corpus-owned index directly. -- Per-atom checks consume the corpus-owned index directly.
pipe_ctx.component_body_index = (corpus and corpus.component_body_index) or {} pipe_ctx.component_body_index = (corpus and corpus.component_body_index) or {}
@@ -2050,36 +2042,25 @@ local function validate(ctx, src, corpus_pipe_ctx)
--- ---
--- Body, token, and emission projections come from here (`paths.tokens = body_tokens`, `paths.line_in_body = build_body_line_index` `paths.word_events` --- Body, token, and emission projections come from here (`paths.tokens = body_tokens`, `paths.line_in_body = build_body_line_index` `paths.word_events`
--- and related fields are owned by `passes/emission_model.lua` pass (per-atom emission projection). --- and related fields are owned by `passes/emission_model.lua` pass (per-atom emission projection).
--- This pass reads: `paths.tokens`, `paths.line_in_body` ` paths.items`, `paths.word_events` from the canonical projection, --- This pass reads: `paths.tokens`, `paths.line_in_body`, `paths.items`, `paths.word_events` from the emitted projection,
--- then computes `paths.tok_class`, `paths.cycles_min/max`, `paths.branches`, `paths.paths`, `paths.has_loops`, `paths.unknown_macros` --- then computes `paths.tok_class`, `paths.cycles_min/max`, `paths.branches`, `paths.paths`, `paths.has_loops`, `paths.unknown_macros` via `classify_tokens` + `analyze_atom_paths`.
--- via `classify_tokens` + `analyze_atom_paths`.
--- No re-walk of body text or body_tokens happens here. --- No re-walk of body text or body_tokens happens here.
--- ---
--- Isolated component checks may supply a component body directly. --- Canonical contract: `atom.paths` and `atom.paths.word_events` MUST be populated by `passes/emission_model.run(ctx)` before this pass runs.
--- Such inputs may omit an emission projection and require this pass to populate `paths.word_events` through `duffle.expand_word_events`. --- The `atom.paths.word_events` projection is owned by the emission-model pass; static-analysis reads it directly.
--- Normal callers run emission-model first.
local findings = {} local findings = {}
for _, a in ipairs(atoms) do for _, a in ipairs(atoms) do
a.paths = a.paths or {} if a.paths == nil then
error("static_analysis: a.paths is nil; emit emission-model first")
end
if a.paths.word_events == nil then
error("static_analysis: a.paths.word_events is nil; emit emission-model first")
end
-- `paths.tokens` / `paths.line_in_body` / `paths.items` / `paths.word_events` are populated by `passes/emission_model.lua`. -- `paths.tokens` / `paths.line_in_body` / `paths.items` / `paths.word_events` are populated by `passes/emission_model.lua`.
-- Supply tokens when no emission projection is present. -- Supply tokens when no emission projection is present.
if a.paths.tokens == nil then a.paths.tokens = a.body_tokens end if a.paths.tokens == nil then a.paths.tokens = a.body_tokens end
a.paths.tok_class = classify_tokens(a.paths.tokens) a.paths.tok_class = classify_tokens(a.paths.tokens)
-- Supply word events when no emission projection is present.
if a.paths.word_events == nil then
local body_entry = {
body_tokens = a.body_tokens,
body_off = a.body_off,
line_of = src.scan.line_of,
source = src.path,
declaration = a.line,
}
a.paths.word_events = duffle.expand_word_events(body_entry,
pipe_ctx.component_body_index,
corpus.word_counts or {})
end
-- analyze_atom_paths fills the *cycles / branches / has_loops / unknown_macros* fields of a.paths. -- analyze_atom_paths fills the *cycles / branches / has_loops / unknown_macros* fields of a.paths.
analyze_atom_paths(a) analyze_atom_paths(a)
@@ -2212,7 +2193,6 @@ local function emit_module_static_analysis_txt(ctx, dir, dir_sources, atoms, fin
-- Module basename = last component of `dir` ("code/duffle" -> "duffle"). -- Module basename = last component of `dir` ("code/duffle" -> "duffle").
local dir_basename = dir:match("([^/\\]+)$") or dir local dir_basename = dir:match("([^/\\]+)$") or dir
local out_path = ctx.out_root .. "/" .. dir_basename .. ".static_analysis.txt" local out_path = ctx.out_root .. "/" .. dir_basename .. ".static_analysis.txt"
if ctx.dry_run then return out_path end
duffle.ensure_dir(ctx.out_root) duffle.ensure_dir(ctx.out_root)
local lines = {} local lines = {}
@@ -2373,7 +2353,7 @@ local function emit_module_static_analysis_txt(ctx, dir, dir_sources, atoms, fin
end end
-- Module-level findings summary (across all sources). -- Module-level findings summary (across all sources).
-- Info is its own count; it is NOT lumped into warnings. -- Info has its own count; it remains separate from warnings.
local total_errs = #errors local total_errs = #errors
local total_warns = #warnings local total_warns = #warnings
local total_infos = #info local total_infos = #info
@@ -2412,13 +2392,11 @@ function M.run(ctx)
local warnings = {} local warnings = {}
-- `info` aggregates finding-level info across every source (the per-source validate() also -- `info` aggregates finding-level info across every source (the per-source validate() also
-- returns a `summaries` collection for scan/cycle rollups; -- returns a `summaries` collection for scan/cycle rollups;
-- those are NOT finding-level and never enter `info`). -- those are summary rows and never enter `info`).
local info = {} local info = {}
-- Build the corpus-wide pipe_ctx ONCE per pass run. -- Build the corpus-wide pipe_ctx ONCE per pass run.
-- The corpus owns the canonical cross-source registries; per-source scans -- The pipe_ctx is shared across every validate() invocation in this M.run so cross-source visibility is constant.
-- retain body / declaration ownership. The pipe_ctx is shared across every
-- validate() invocation in this M.run so cross-source visibility is constant.
local corpus_pipe_ctx = build_corpus_pipe_ctx(ctx) local corpus_pipe_ctx = build_corpus_pipe_ctx(ctx)
local corpus = ctx.shared.corpus local corpus = ctx.shared.corpus
+4 -5
View File
@@ -8,9 +8,9 @@
--- AFTER computing each current count from the just-built body + `corpus.word_counts`). --- AFTER computing each current count from the just-built body + `corpus.word_counts`).
--- ---
--- **Canonical contract**: --- **Canonical contract**:
--- * `ctx.shared.corpus.word_counts` is the canonical count table. --- * `ctx.shared.corpus.word_counts` is the count table.
--- * `corpus.word_counts` is the sole count table. Consumers read `corpus.word_counts` directly. --- * `corpus.word_counts` is the sole count table. Consumers read `corpus.word_counts` directly.
--- * `ctx.shared.components` and `ctx.shared.component_body_index` are NOT created by this pass (canonical projections only). --- * `ctx.shared.components` and `ctx.shared.component_body_index` are NOT created by this pass (projections only).
--- * No `.macs.h` recursive discovery (no `scan_dir`, no scan cache, no `_invalidate_scan_cache`). --- * No `.macs.h` recursive discovery (no `scan_dir`, no scan cache, no `_invalidate_scan_cache`).
--- ---
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex, --- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
@@ -49,7 +49,6 @@ local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
--- @field project_root string -- project root (e.g. "code/") --- @field project_root string -- project root (e.g. "code/")
--- @field upstream table<string, table> -- per-pass upstream outputs --- @field upstream table<string, table> -- per-pass upstream outputs
--- @field flags table -- CLI flags --- @field flags table -- CLI flags
--- @field dry_run boolean -- if true, compute but don't write
--- @field verbose boolean -- if true, log diagnostic info --- @field verbose boolean -- if true, log diagnostic info
--- @class PassResult --- @class PassResult
@@ -117,10 +116,10 @@ function M.run(ctx)
end end
-- 3. Load authored metadata. Generated .macs.h files are NOT scanned -- 3. Load authored metadata. Generated .macs.h files are NOT scanned
-- (the canonical pass computes their counts from the just-built bodies after disk emission; see passes/components.lua). -- (the pass computes their counts from the just-built bodies after disk emission; see passes/components.lua).
local wc = duffle.load_word_counts(ctx.metadata_path) local wc = duffle.load_word_counts(ctx.metadata_path)
-- 4. Assign the canonical count table. ONE assignment, no copy. The assignment creates no secondary alias. -- 4. Assign the count table. ONE assignment, no copy. The assignment creates no secondary alias.
corpus.word_counts = wc corpus.word_counts = wc
return { outputs = {}, errors = {}, warnings = {} } return { outputs = {}, errors = {}, warnings = {} }
+9 -113
View File
@@ -20,7 +20,7 @@
-- Bootstrap: load `duffle_paths.lua` via this script's own path. -- Bootstrap: load `duffle_paths.lua` via this script's own path.
-- Use `arg[0]` when this file is the entry script (`arg[0]` ends in "ps1_meta.lua"); -- Use `arg[0]` when this file is the entry script (`arg[0]` ends in "ps1_meta.lua");
-- fall back to `debug.getinfo(1, "S").source` when this file is being dofile()'d or require()'d (in which case `arg[0]` is the *caller's* path, not ours). -- fall back to `debug.getinfo(1, "S").source` when this file is being dofile()'d or require()'d (in which case `arg[0]` is the *caller's* path, not ours).
-- That single statement: (a) sets `package.path` + `package.cpath` (via cached `git rev-parse`), (b) at the bottom returns `require("duffle")`. -- That single statement: (a) sets `package.path` + `package.cpath`, (b) at the bottom returns `require("duffle")`.
-- So the dofile's return value is the duffle module. -- So the dofile's return value is the duffle module.
local _is_entry_script = arg and arg[0] and arg[0]:match("ps1_meta%.lua$") ~= nil local _is_entry_script = arg and arg[0] and arg[0]:match("ps1_meta%.lua$") ~= nil
local _bootstrap_src local _bootstrap_src
@@ -63,12 +63,6 @@ local PASS_FLAG_DISPATCH_KEY = "__pass__"
--- @field deps string[] -- names of upstream passes --- @field deps string[] -- names of upstream passes
--- @field groups string[]? -- OPTIONAL build-phase groups this pass is a root of --- @field groups string[]? -- OPTIONAL build-phase groups this pass is a root of
--- -- (e.g. { "pre-link" }, { "post-link" }); absent ⇒ dependency-only --- -- (e.g. { "pre-link" }, { "post-link" }); absent ⇒ dependency-only
--- @field desc string -- human description (used by --help + ASCII graph)
--- @field out PassOutput[] -- output paths (used by --dry-run + report)
--- @class PassOutput
--- @field kind string -- "header" | "report"
--- @field path_template string -- e.g. "<source_dir>/gen/<basename>.macs.h"
--- @class SourceFile --- @class SourceFile
--- @field path string -- absolute path to the source file --- @field path string -- absolute path to the source file
@@ -82,15 +76,9 @@ local PASS_FLAG_DISPATCH_KEY = "__pass__"
--- @field shared.corpus table -- canonical authored-source/project projection --- @field shared.corpus table -- canonical authored-source/project projection
--- @field out_root string -- output root (e.g. "build/gen") --- @field out_root string -- output root (e.g. "build/gen")
--- @field project_root string -- PS1 repository root --- @field project_root string -- PS1 repository root
--- @field upstream table<string, table> -- per-pass output accumulator
--- @field flags table -- CLI flags + per-pass stash --- @field flags table -- CLI flags + per-pass stash
--- @field dry_run boolean -- if true, compute but don't write
--- @field verbose boolean -- if true, log diagnostic info --- @field verbose boolean -- if true, log diagnostic info
--- @class PassOutputEntry
--- @field [string] string -- dynamic shape; key is the output kind
-- (e.g. "macs_h", "offsets_h", "errors_h", "annotations_txt", "static_analysis_txt", "summary_txt"), value is the path
--- @class Finding --- @class Finding
--- @field line integer -- source line (or 0 for pass-level) --- @field line integer -- source line (or 0 for pass-level)
--- @field msg string -- finding message --- @field msg string -- finding message
@@ -107,7 +95,6 @@ local PASS_FLAG_DISPATCH_KEY = "__pass__"
--- @field metadata string -- --metadata value --- @field metadata string -- --metadata value
--- @field out_root string -- --out-root value (default "build/gen") --- @field out_root string -- --out-root value (default "build/gen")
--- @field project_root string -- PS1 repository root (derived from metadata by default) --- @field project_root string -- PS1 repository root (derived from metadata by default)
--- @field dry_run boolean -- if true, compute but don't write
--- @field verbose boolean -- if true, log diagnostic info --- @field verbose boolean -- if true, log diagnostic info
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -125,46 +112,31 @@ local PASSES = {
["scan-source"] = { ["scan-source"] = {
module = "passes.scan_source", module = "passes.scan_source",
kind = "shared", deps = {}, kind = "shared", deps = {},
desc = "Walk each source once; produce the fat SourceScan payload for downstream passes",
out = {},
}, },
["word-counts"] = { ["word-counts"] = {
module = "passes.word_count_eval", module = "passes.word_count_eval",
kind = "shared", deps = {}, kind = "shared", deps = {},
desc = "Build the shared metadata table (metadata.h + .macs.h)",
out = {},
}, },
components = { components = {
module = "passes.components", module = "passes.components",
kind = "header-output", kind = "header-output",
deps = {"scan-source", "word-counts"}, deps = {"scan-source", "word-counts"},
desc = "Emit mac_X macros from MipsAtomComp_ declarations",
out = { { kind = "header", path_template = "<source_dir>/gen/<basename>.macs.h" } },
}, },
["emission-model"] = { ["emission-model"] = {
module = "passes.emission_model", module = "passes.emission_model",
kind = "validation", kind = "validation",
deps = {"components"}, deps = {"components"},
desc = "Build canonical per-atom words, markers, and invocation ancestry",
out = {},
}, },
annotation = { annotation = {
module = "passes.annotation", module = "passes.annotation",
kind = "validation", kind = "validation",
deps = {"scan-source", "word-counts"}, deps = {"scan-source", "word-counts"},
desc = "Validate atom DSL usage; emit errors.h + annotations.txt",
out = {
{ kind = "report", path_template = "<out_root>/<basename>.errors.h" },
{ kind = "report", path_template = "<out_root>/<basename>.annotations.txt" },
},
}, },
offsets = { offsets = {
module = "passes.offsets", module = "passes.offsets",
kind = "header-output", kind = "header-output",
deps = {"scan-source", "word-counts", "components", "emission-model"}, deps = {"scan-source", "word-counts", "components", "emission-model"},
groups = { "pre-link" }, groups = { "pre-link" },
desc = "Compute branch offsets for atom_label / atom_offset",
out = { { kind = "header", path_template = "<source_dir>/gen/<basename>.offsets.h" } },
}, },
["static-analysis"] = { ["static-analysis"] = {
module = "passes.static_analysis", module = "passes.static_analysis",
@@ -173,43 +145,23 @@ local PASSES = {
-- Report severity is independent from process exit policy. -- Report severity is independent from process exit policy.
kind = "diagnostic", kind = "diagnostic",
deps = {"scan-source", "word-counts", "components", "emission-model"}, deps = {"scan-source", "word-counts", "components", "emission-model"},
desc = "Static analysis: GTE pipeline-fill, mac_yield uniformity, ABI handoff, GPU port-store shape, per-atom cycle budget, type consistency",
out = { { kind = "report", path_template = "<out_root>/<basename>.static_analysis.txt" } },
}, },
["atoms-source-map"] = { ["atoms-source-map"] = {
module = "passes.atoms_source_map", module = "passes.atoms_source_map",
kind = "header-output", kind = "header-output",
deps = {"word-counts", "components", "emission-model"}, deps = {"word-counts", "components", "emission-model"},
desc = "Emit gen/<basename>.atoms.sourcemap.txt (per-.word C source line map for gdb debugging) AND gen/<basename>.atoms.provenance.txt (per-.word provenance; each word tagged with its call-site file:line and, when emitted by a mac_X(...) component invocation, the component's definition file:line). Consumed by passes/dwarf_injection.lua to synthesize DW_TAG_inlined_subroutine instances for source-level Step Into on component invocations.",
out = {
{ kind = "report", path_template = "<out_root>/<basename>.atoms.sourcemap.txt" },
{ kind = "report", path_template = "<out_root>/<basename>.atoms.provenance.txt" },
},
}, },
["dwarf-injection"] = { ["dwarf-injection"] = {
module = "passes.dwarf_injection", module = "passes.dwarf_injection",
kind = "shared", kind = "shared",
deps = {"scan-source", "atoms-source-map"}, deps = {"scan-source", "atoms-source-map"},
groups = { "post-link" }, groups = { "post-link" },
desc = "Inject per-atom .debug_line + .debug_aranges (F') + per-atom .debug_info subprogram + per-wave-context-reg .debug_info variables (G') into the ELF (post-link; writes 7 section .bin blobs plus one deterministic .gdbinit sidecar). (rbind composite) reads ctx.sources[i].scan to find atom_bind(Binds_X) atoms + their Binds_X struct fields; emits per-Binds_X DW_TAG_structure_type DIEs + per-rbind-atom DW_TAG_variable 'bind_args' DIEs with piece-chain DW_OP_bregN/DW_OP_piece location expressions.",
out = {
{ kind = "report", path_template = "<out_root>/<basename>.dwarf_line.bin" },
{ kind = "report", path_template = "<out_root>/<basename>.dwarf_aranges.bin" },
{ kind = "report", path_template = "<out_root>/<basename>.dwarf_rnglists.bin" },
{ kind = "report", path_template = "<out_root>/<basename>.dwarf_abbrev.bin" },
{ kind = "report", path_template = "<out_root>/<basename>.dwarf_info.bin" },
{ kind = "report", path_template = "<out_root>/<basename>.dwarf_str.bin" },
{ kind = "report", path_template = "<out_root>/<basename>.dwarf_loc.bin" },
{ kind = "report", path_template = "<out_root>/<basename>.gdbinit" },
},
}, },
report = { report = {
module = "passes.report", module = "passes.report",
kind = "report", kind = "report",
deps = {"annotation", "static-analysis"}, deps = {"annotation", "static-analysis"},
groups = { "pre-link" }, groups = { "pre-link" },
desc = "Render the per-project summary",
out = { { kind = "report", path_template = "<out_root>/annotation_validation.txt" } },
}, },
} }
@@ -314,13 +266,10 @@ USAGE:
PASS_FLAGS: PASS_FLAGS:
Pick a phase or one-or-more individual passes: Pick a phase or one-or-more individual passes:
--pre-link [phase; default] Run the pre-link group + transitive deps. --pre-link [phase; default] Run the pre-link group + transitive deps.
The root set is data-driven from each PASSES row's The root set is data-driven from each PASSES row's groups` field; no parallel name list is maintained.
`groups` field; no parallel name list is maintained.
--post-link [phase] Run the post-link group + transitive deps. --post-link [phase] Run the post-link group + transitive deps.
Requires --elf. Sets --gdb-runtime and --dwarf-injection Requires --elf. Sets --gdb-runtime and --dwarf-injection opt-in flags as well.
opt-in flags as well. --all Select every row of the PASSES table. Pass-local opt-in guards remain active, so --dwarf-injection still requires
--all Select every row of the PASSES table. Pass-local opt-in
guards remain active, so --dwarf-injection still requires
--elf and --gdb-runtime still requires a runtime emission. --elf and --gdb-runtime still requires a runtime emission.
Or pick any subset: Or pick any subset:
--scan-source Scan sources into the fat SourceScan payload --scan-source Scan sources into the fat SourceScan payload
@@ -329,23 +278,18 @@ PASS_FLAGS:
--validate Run atom annotation DSL validation --validate Run atom annotation DSL validation
--offsets Generate <module>/gen/<basename>.offsets.h --offsets Generate <module>/gen/<basename>.offsets.h
--atoms-source-map Generate <basename>.atoms.sourcemap.txt per source --atoms-source-map Generate <basename>.atoms.sourcemap.txt per source
--dwarf-injection [opt-in] Select the post-link dwarf-injection pass + set the --dwarf-injection [opt-in] Select the post-link dwarf-injection pass + set the opt-in flag. Requires --elf.
opt-in flag. Requires --elf.
--static-analysis Static analysis: GTE pipeline-fill, mac_yield, ABI handoff, cycle budget --static-analysis Static analysis: GTE pipeline-fill, mac_yield, ABI handoff, cycle budget
--report Render per-project summary --report Render per-project summary
COMMON_FLAGS: COMMON_FLAGS:
--unity-root FILE Unity source root: load root + direct quoted authored --unity-root FILE Unity source root: load root + direct quoted authored includes only. Mutually exclusive with --source.
includes only. Mutually exclusive with --source. --source FILE Exact source file to process (repeatable, never expands includes). Mutually exclusive with --unity-root.
--source FILE Exact source file to process (repeatable, never expands
includes). Mutually exclusive with --unity-root.
--metadata PATH Path to metadata.h (required) --metadata PATH Path to metadata.h (required)
--out-root DIR Output root for reports (default: build/gen) --out-root DIR Output root for reports (default: build/gen)
--project-root DIR PS1 repository root (default: derived from --project-root DIR PS1 repository root (default: derived from <repo>/code/duffle/word_count.metadata.h)
<repo>/code/duffle/word_count.metadata.h)
--gdb-runtime Also emit <out_root>/gdb_tape_atoms_runtime.gdb (post-link, requires --elf) --gdb-runtime Also emit <out_root>/gdb_tape_atoms_runtime.gdb (post-link, requires --elf)
--elf PATH Path to linked .elf (for --gdb-runtime / --dwarf-injection) --elf PATH Path to linked .elf (for --gdb-runtime / --dwarf-injection)
--dry-run Print dep order (alphabetical); exit 0 without running
--verbose Print per-pass debug output --verbose Print per-pass debug output
--help Show this help and exit --help Show this help and exit
@@ -391,7 +335,6 @@ FLAG_HANDLERS["--help"] = function(args)
os.exit(0) os.exit(0)
end end
FLAG_HANDLERS["--dry-run"] = function(args) args.dry_run = true end
FLAG_HANDLERS["--verbose"] = function(args) args.verbose = true end FLAG_HANDLERS["--verbose"] = function(args) args.verbose = true end
FLAG_HANDLERS["--source"] = function(args, argv, arg_idx) FLAG_HANDLERS["--source"] = function(args, argv, arg_idx)
local value, value_idx = require_flag_value(argv, arg_idx, "--source") local value, value_idx = require_flag_value(argv, arg_idx, "--source")
@@ -476,7 +419,6 @@ local function parse_args(argv)
metadata = nil, metadata = nil,
out_root = DEFAULT_OUT_ROOT, out_root = DEFAULT_OUT_ROOT,
project_root = nil, project_root = nil,
dry_run = false,
verbose = false, verbose = false,
} }
@@ -557,7 +499,7 @@ local function build_ctx(args)
or normalized_project_root:sub(1, 1) == "/" or normalized_project_root:sub(1, 1) == "/"
if not project_root_is_absolute then if not project_root_is_absolute then
-- canonical_path_key validates ordinary relative paths and rejects -- canonical_path_key validates ordinary relative paths and rejects
-- drive-relative paths before the legacy display-path helper is used. -- drive-relative paths before the absolute-path rewrite is performed.
duffle.canonical_path_key(normalized_project_root) duffle.canonical_path_key(normalized_project_root)
project_root = duffle.normalize_path(duffle.to_absolute_path(normalized_project_root)) project_root = duffle.normalize_path(duffle.to_absolute_path(normalized_project_root))
else else
@@ -654,11 +596,9 @@ local function build_ctx(args)
local ctx = { local ctx = {
metadata_path = args.metadata, metadata_path = args.metadata,
shared = { corpus = corpus }, shared = { corpus = corpus },
upstream = {},
out_root = args.out_root, out_root = args.out_root,
project_root = corpus.project_root, project_root = corpus.project_root,
flags = args.flags or {}, flags = args.flags or {},
dry_run = args.dry_run,
verbose = args.verbose, verbose = args.verbose,
} }
@@ -756,42 +696,10 @@ local function topo_sort(passes, requested_set)
return order return order
end end
-- ════════════════════════════════════════════════════════════════════════════
-- ASCII dep graph renderer (Decision 6 in the spec)
-- ════════════════════════════════════════════════════════════════════════════
-- ════════════════════════════════════════════════════════════════════════════
-- Topological dep-order printer (used by --dry-run).
-- Re-render the PASSES graph manually in `docs/guide_metaprogram_ssdl.md` if you need an updated visual;
-- The canonical ASCII view there is regenerated by hand whenever PASSES rows change.
-- ════════════════════════════════════════════════════════════════════════════
local function render_dep_order(passes, closed)
local lines = {}
lines[#lines + 1] = "[ps1_meta] Resolved dependency order (closed under deps):"
for pass_idx, name in ipairs(closed) do
local p = passes[name]
local deps_str = (#p.deps == 0) and "(no deps)" or
"(deps: " .. table.concat(p.deps, ", ") .. ")"
lines[#lines + 1] = string.format(" %d. %-22s %-45s [%s]",
pass_idx, name, deps_str, p.kind)
end
return table.concat(lines, "\n") .. "\n"
end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Main Orchestrator -- Main Orchestrator
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- (internal) Push a pass's outputs + warnings into `ctx.upstream[name]` for downstream passes to consume.
--- @param ctx PassCtx
--- @param pass_name string
--- @param result PassResult
local function accumulate_pass_result(ctx, pass_name, result)
ctx.upstream[pass_name] = ctx.upstream[pass_name] or {}
for _, out in ipairs(result.outputs or {}) do table.insert(ctx.upstream[pass_name], out) end
for _, warn in ipairs(result.warnings or {}) do table.insert(ctx.upstream[pass_name], warn) end
end
--- (internal) If the pass's kind is in PASS_KIND_STOP_ON_ERROR and it reported errors, write each error to stderr. --- (internal) If the pass's kind is in PASS_KIND_STOP_ON_ERROR and it reported errors, write each error to stderr.
--- Returns true if any validation errors were reported. --- Returns true if any validation errors were reported.
--- @param pass_name string --- @param pass_name string
@@ -812,14 +720,11 @@ end
--- @param order string[] --- @param order string[]
--- @return boolean -- true if any validation errors were reported --- @return boolean -- true if any validation errors were reported
local function dispatch_passes(ctx, order) local function dispatch_passes(ctx, order)
ctx.shared = ctx.shared or {}
local had_errors = false local had_errors = false
for _, pass_name in ipairs(order) do for _, pass_name in ipairs(order) do
local pass = PASSES[pass_name] local pass = PASSES[pass_name]
-- io.stderr:write(string.format("[ps1_meta] %-22s running\n", pass_name))
local mod = require(pass.module) local mod = require(pass.module)
local result = mod.run(ctx) local result = mod.run(ctx)
accumulate_pass_result(ctx, pass_name, result)
if report_validation_errors(pass_name, pass, result) then if report_validation_errors(pass_name, pass, result) then
had_errors = true had_errors = true
end end
@@ -837,13 +742,6 @@ local function main(argv)
local requested = args.requested_set local requested = args.requested_set
local closed = topo_sort(PASSES, requested) local closed = topo_sort(PASSES, requested)
-- --dry-run: print the closed dep order and exit OK.
-- (The hand-rendered PASSES graph lives in docs/guide_metaprogram_ssdl.md; see Decision 6.)
if args.dry_run then
io.write(render_dep_order(PASSES, closed))
os.exit(EXIT_OK)
end
local had_errors = dispatch_passes(ctx, closed) local had_errors = dispatch_passes(ctx, closed)
if had_errors then os.exit(EXIT_VALIDATION_ERRORS) end if had_errors then os.exit(EXIT_VALIDATION_ERRORS) end
end) end)
@@ -857,11 +755,9 @@ local function main(argv)
end end
-- Module export for in-process consumers (tests that dofile this script). -- Module export for in-process consumers (tests that dofile this script).
-- The closed dep-order printer + the `PASSES` table are exposed so a test can observe the resolved dep order for synthetic PASSES tables without spawning a subprocess.
-- The conditional `main(...)` call below only fires when this file is invoked as the entry script (arg[0] ends in "ps1_meta.lua"); -- The conditional `main(...)` call below only fires when this file is invoked as the entry script (arg[0] ends in "ps1_meta.lua");
-- in dofile() mode (test's arg[0] does not match), main() is skipped and the chunk returns `_M` to the caller. -- in dofile() mode (test's arg[0] does not match), main() is skipped and the chunk returns `_M` to the caller.
local _M = { local _M = {
render_dep_order = render_dep_order,
PASSES = PASSES, PASSES = PASSES,
PASS_KIND_STOP_ON_ERROR = PASS_KIND_STOP_ON_ERROR, PASS_KIND_STOP_ON_ERROR = PASS_KIND_STOP_ON_ERROR,
parse_args = parse_args, parse_args = parse_args,