--- duffle_isa.lua — encoder / GTE / hardware tables. local M = {} -- Section 7: domain tables -- ════════════════════════════════════════════════════════════════════════════ -- atom_info sub-calls: atom_bind, atom_reads, atom_writes, atom_view, atom_reg_types, atom_ctx, atom_phase. M.TAPE_ATOM_MACROS = { ["atom_info"] = { kind = "info", binds = false }, } -- Empty C macros that prefix the next encoder. Zero words. -- BdSlot_ nop is one nop word. The marker is not the BD instruction. M.DELAY_MARKERS = { ["GteDelay_"] = true, ["LdSlot_"] = true, ["BdSlot_"] = true, ["DmaSlot_"] = true, } -- One row per encoder. Old table names are load-time views (build_isa_views). M.INSTRUCTION = { ["BdSlot_"] = { cycles = 0, kind = "marker", }, ["LdSlot_"] = { cycles = 0, kind = "marker", }, ["add_s"] = { cycles = 1, kind = "alu", }, ["add_si"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, }, ["add_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, }, ["add_u_self"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, value = { dest = 1, op = "add_u", sources = { 1, 2 }, }, }, ["add_ui"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, value = { dest = 1, immediate = 3, op = "add_ui", source = 2, }, }, ["add_ui_self"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = { 1 }, imm = { { arg = 2, signed = true, width = 16, }, }, value = { dest = 1, immediate = 2, op = "add_ui", source = 1, }, }, ["and"] = { cycles = 1, kind = "alu", }, ["and_i"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, width = 16, }, }, value = { dest = 1, immediate = 3, op = "and_i", source = 2, }, }, ["and_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, }, ["atom_bind"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, }, ["atom_info"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, }, ["atom_label"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, }, ["atom_offset"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, }, ["atom_reads"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, }, ["atom_writes"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, }, ["branch_equal"] = { cycles = 2, kind = "branch", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, }, ["branch_ge_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, }, ["branch_gt_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, }, ["branch_le_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, }, ["branch_lt_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, }, ["branch_ne"] = { cycles = 2, kind = "branch", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, }, ["call_addr"] = { cycles = 2, kind = "call", reads = {}, writes = { 1 }, }, ["call_reg"] = { cycles = 2, kind = "call", reads = { 1 }, writes = { 2 }, }, ["div_s"] = { cycles = 35, kind = "alu", reads = { 1, 2 }, writes = {}, }, ["div_u"] = { cycles = 35, kind = "alu", reads = { 1, 2 }, writes = {}, }, ["gte_load_v0"] = { cycles = 2, kind = "cop2_xfer", reads = { 2 }, writes = {}, }, ["gte_load_v0v1v2"] = { cycles = 6, kind = "cop2_xfer", reads = { 2 }, writes = {}, }, ["gte_load_v1"] = { cycles = 2, kind = "cop2_xfer", reads = { 2 }, writes = {}, }, ["gte_load_v2"] = { cycles = 2, kind = "cop2_xfer", reads = { 2 }, writes = {}, }, ["gte_lw"] = { cycles = 1, kind = "load", reads = { 2 }, writes = {}, }, ["gte_lwc2"] = { cycles = 1, kind = "load", }, ["gte_mv_from_ctrl_r"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = { 1 }, }, ["gte_mv_from_data_r"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = { 1 }, }, ["gte_mv_to_ctrl_r"] = { cycles = 1, kind = "cop2_xfer", reads = { 1 }, writes = {}, }, ["gte_mv_to_data_r"] = { cycles = 1, kind = "cop2_xfer", reads = { 1 }, writes = {}, }, ["gte_stotz"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = {}, }, ["gte_stsxy3"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = {}, }, ["gte_sw"] = { cycles = 1, kind = "store", reads = { 2 }, writes = {}, }, ["gte_swc2"] = { cycles = 1, kind = "store", }, ["jump"] = { cycles = 2, kind = "jump", reads = {}, writes = {}, }, ["jump_link"] = { cycles = 2, kind = "call", reads = { 1 }, writes = { 2 }, }, ["jump_reg"] = { cycles = 2, kind = "jump", reads = { 1 }, writes = {}, suppress_arg1 = { R_AtomJmp = "fixed mac_yield handshake", }, }, ["jump_rel"] = { cycles = 2, kind = "branch", delay_slot = false, }, ["li_s"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, value = { dest = 1, immediate = 3, op = "add_ui", source = 2, }, }, ["load_byte"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, }, ["load_byte_u"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, }, ["load_half"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, }, ["load_half_u"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, }, ["load_imm"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, }, ["load_imm_1w"] = { cycles = 1, kind = "alu", reads = {}, writes = { 1 }, }, ["load_imm_1w_s0"] = { cycles = 1, kind = "alu", reads = {}, writes = { 1 }, }, ["load_imm_2w"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, }, ["load_imm_2w_addi_forced"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, }, ["load_imm_2w_ori_forced"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, }, ["load_ui"] = { cycles = 1, kind = "alu", reads = {}, writes = { 1 }, }, ["load_upper_i"] = { cycles = 1, kind = "alu", reads = {}, writes = { 1 }, imm = { { arg = 2, width = 16, }, }, value = { dest = 1, immediate = 2, op = "load_upper_i", }, }, ["load_word"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, }, ["mac_yield"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, }, ["mask_upper"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, }, ["mov_from_high"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, }, ["mov_from_low"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, }, ["mov_to_high"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = {}, }, ["mov_to_low"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = {}, }, ["mult_s"] = { cycles = 12, kind = "alu", reads = { 1, 2 }, writes = {}, }, ["mult_u"] = { cycles = 12, kind = "alu", reads = { 1, 2 }, writes = {}, }, ["nop"] = { cycles = 1, kind = "nop", reads = {}, writes = {}, }, ["nop2"] = { cycles = 2, kind = "nop", reads = {}, writes = {}, }, ["nor_u"] = { cycles = 1, kind = "alu", }, ["or_i"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, width = 16, }, }, value = { dest = 1, immediate = 3, op = "or_i", source = 2, }, }, ["or_i_self"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = { 1 }, imm = { { arg = 2, width = 16, }, }, value = { dest = 1, immediate = 2, op = "or_i", source = 1, }, }, ["or_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, }, ["or_u_self"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, value = { dest = 1, op = "or", sources = { 1, 2 }, }, }, ["set_lt_s"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, }, ["set_lt_si"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, }, ["set_lt_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, }, ["set_lt_ui"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, }, ["shift_aright"] = { cycles = 1, kind = "alu", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, }, ["shift_aright_var"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, }, ["shift_lleft"] = { cycles = 1, kind = "alu", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, }, ["shift_lleft_self"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = { 1 }, imm = { { arg = 2, width = 5, }, }, value = { dest = 1, immediate = 2, op = "shift_lleft", source = 1, }, }, ["shift_lleft_var"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, }, ["shift_lright"] = { cycles = 1, kind = "alu", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, }, ["slt_s"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, }, ["slt_si"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, }, ["slt_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, }, ["slt_ui"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, }, ["store_byte"] = { cycles = 1, kind = "store", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, }, ["store_half"] = { cycles = 1, kind = "store", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, }, ["store_word"] = { cycles = 1, kind = "store", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, }, ["sub_s"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, }, ["sub_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, }, ["sys_mov_from_cop0"] = { cycles = 1, kind = "cop0_xfer", reads = {}, writes = { 1 }, }, ["sys_mov_to_cop0"] = { cycles = 1, kind = "cop0_xfer", reads = { 1 }, writes = {}, }, ["xor_i"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, width = 16, }, }, value = { dest = 1, immediate = 3, op = "xor_i", source = 2, }, }, ["xor_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, }, } -- One row per GTE command. Alias cycle numbers live here, not on INSTRUCTION. M.GTE_COMMAND = { ["gte_cmdw_avsz3"] = { aliases = { "gte_avg_sort_z3", "gte_avsz3", "gte_cmdw_avg_sort_z3" }, cycles = 5, inputs = { "C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3", "gte_cr_ZSF3" }, outputs = { { register = "C2_OTZ", role = "otz", }, }, latch = { { register = "C2_OTZ", required = 4, }, }, }, ["gte_cmdw_avsz4"] = { aliases = { "gte_avg_sort_z4", "gte_avsz4", "gte_cmdw_avg_sort_z4" }, cycles = 6, inputs = { "C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3", "gte_cr_ZSF4" }, outputs = { { register = "C2_OTZ", role = "otz", }, }, latch = { { register = "C2_OTZ", required = 4, }, }, }, ["gte_cmdw_gpf"] = { aliases = {}, cycles = 5, inputs = { "C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3" }, outputs = { { register = "C2_MAC1", role = "mac_result", }, { register = "C2_MAC2", role = "mac_result", }, { register = "C2_MAC3", role = "mac_result", }, { register = "C2_IR1", role = "latest_color", }, { register = "C2_IR2", role = "latest_color", }, { register = "C2_IR3", role = "latest_color", }, }, latch = { { register = "C2_MAC1", required = 4, }, { register = "C2_MAC2", required = 4, }, { register = "C2_MAC3", required = 4, }, { register = "C2_IR1", required = 4, }, { register = "C2_IR2", required = 4, }, { register = "C2_IR3", required = 4, }, }, }, ["gte_cmdw_mvmva"] = { aliases = {}, cycles = 8, inputs = { "C2_VXY0", "C2_VZ0", "C2_VXY1", "C2_VZ1", "C2_VXY2", "C2_VZ2", "C2_IR1", "C2_IR2", "C2_IR3", "gte_cr_RT11", "gte_cr_RT12", "gte_cr_RT13", "gte_cr_RT21", "gte_cr_RT22", "gte_cr_RT23", "gte_cr_RT31", "gte_cr_RT32", "gte_cr_RT33", "gte_cr_TRX", "gte_cr_TRY", "gte_cr_TRZ" }, outputs = { { register = "C2_IR1", role = "latest_color", }, { register = "C2_IR2", role = "latest_color", }, { register = "C2_IR3", role = "latest_color", }, }, latch = { { register = "C2_IR1", required = 4, }, { register = "C2_IR2", required = 4, }, { register = "C2_IR3", required = 4, }, }, }, ["gte_cmdw_nclip"] = { aliases = { "gte_nclip" }, cycles = 8, inputs = { "C2_SXY0", "C2_SXY1", "C2_SXY2" }, outputs = { { register = "C2_SZ3", role = "mac_result", }, }, latch = { { register = "C2_SZ3", required = 4, }, }, }, ["gte_cmdw_op"] = { aliases = { "gte_cmdw_outer_product", "gte_cmdw_wedge" }, cycles = 6, inputs = {}, outputs = { { register = "C2_IR1", role = "latest_color", }, { register = "C2_IR2", role = "latest_color", }, { register = "C2_IR3", role = "latest_color", }, }, latch = { { register = "C2_IR1", required = 4, }, { register = "C2_IR2", required = 4, }, { register = "C2_IR3", required = 4, }, }, }, ["gte_cmdw_rtps"] = { aliases = { "gte_cmdw_rotate_translate_perspective_single", "gte_rtps" }, cycles = 15, inputs = { "C2_VXY0", "C2_VZ0", "C2_VXY1", "C2_VZ1", "C2_VXY2", "C2_VZ2", "C2_RGB", "C2_OTZ", "C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3", "C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3", "gte_cr_RT11", "gte_cr_RT12", "gte_cr_RT13", "gte_cr_RT21", "gte_cr_RT22", "gte_cr_RT23", "gte_cr_RT31", "gte_cr_RT32", "gte_cr_RT33", "gte_cr_TRX", "gte_cr_TRY", "gte_cr_TRZ", "gte_cr_OFX", "gte_cr_OFY", "gte_cr_H", "gte_cr_DQA", "gte_cr_DQB" }, outputs = { { register = "C2_SXY2", role = "latest_screen_xy", }, { register = "C2_SZ2", role = "latest_screen_z", }, { register = "C2_OTZ", role = "otz", }, { register = "C2_IR0", role = "latest_color", }, }, latch = { { register = "C2_SXY2", required = 4, }, { register = "C2_SZ2", required = 4, }, { register = "C2_OTZ", required = 4, }, { register = "C2_IR0", required = 4, }, }, }, ["gte_cmdw_rtpt"] = { aliases = { "gte_cmdw_rotate_translate_perspective_triple", "gte_rtpt" }, cycles = 23, inputs = { "C2_VXY0", "C2_VZ0", "C2_VXY1", "C2_VZ1", "C2_VXY2", "C2_VZ2", "C2_RGB", "C2_OTZ", "C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3", "C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3", "gte_cr_RT11", "gte_cr_RT12", "gte_cr_RT13", "gte_cr_RT21", "gte_cr_RT22", "gte_cr_RT23", "gte_cr_RT31", "gte_cr_RT32", "gte_cr_RT33", "gte_cr_TRX", "gte_cr_TRY", "gte_cr_TRZ", "gte_cr_OFX", "gte_cr_OFY", "gte_cr_H", "gte_cr_DQA", "gte_cr_DQB" }, outputs = { { register = "C2_SXY0", role = "screen_xy[0]", }, { register = "C2_SXY1", role = "screen_xy[1]", }, { register = "C2_SXY2", role = "latest_screen_xy", }, { register = "C2_SZ3", role = "latest_screen_z", }, { register = "C2_OTZ", role = "otz", }, }, latch = { { register = "C2_SXY0", required = 4, }, { register = "C2_SXY1", required = 4, }, { register = "C2_SXY2", required = 4, }, { register = "C2_SZ3", required = 4, }, { register = "C2_OTZ", required = 4, }, }, }, ["gte_cmdw_sqr"] = { aliases = {}, cycles = 5, inputs = { "C2_IR1", "C2_IR2", "C2_IR3" }, outputs = { { register = "C2_MAC1", role = "mac_result", }, { register = "C2_MAC2", role = "mac_result", }, { register = "C2_MAC3", role = "mac_result", }, { register = "C2_IR1", role = "latest_color", }, { register = "C2_IR2", role = "latest_color", }, { register = "C2_IR3", role = "latest_color", }, }, latch = { { register = "C2_MAC1", required = 4, }, { register = "C2_MAC2", required = 4, }, { register = "C2_MAC3", required = 4, }, { register = "C2_IR1", required = 4, }, { register = "C2_IR2", required = 4, }, { register = "C2_IR3", required = 4, }, }, }, } function M.instr(ident) return M.INSTRUCTION[ident] end function M.gte_canon(ident) return M.ALIAS_TO_CANONICAL[ident] or ident end function M.gte(ident) return M.GTE_COMMAND[M.gte_canon(ident)] end local function build_isa_views() M.ALIAS_TO_CANONICAL = {} for canon, row in pairs(M.GTE_COMMAND) do M.ALIAS_TO_CANONICAL[canon] = canon for _, alias in ipairs(row.aliases or {}) do M.ALIAS_TO_CANONICAL[alias] = canon end end M.INSTRUCTION_LATENCY = {} M.INSTRUCTION_GPR_EFFECTS = {} M.IMMEDIATE_FIELD_WIDTHS = {} M.GPR_VALUE_RULES = {} M.CONTROL_TRANSFER_DELAY_SLOT_POLICIES = {} for name, row in pairs(M.INSTRUCTION) do M.INSTRUCTION_LATENCY[name] = row.cycles if row.reads or row.writes then M.INSTRUCTION_GPR_EFFECTS[name] = { reads = row.reads or {}, writes = row.writes or {}, } end if row.imm then M.IMMEDIATE_FIELD_WIDTHS[name] = row.imm end if row.value then M.GPR_VALUE_RULES[name] = row.value end if (row.kind == "branch" or row.kind == "jump" or row.kind == "call") and row.delay_slot ~= false then M.CONTROL_TRANSFER_DELAY_SLOT_POLICIES[name] = { family = row.kind, suppress_arg1 = row.suppress_arg1, } end end M.GTE_COMMAND_ALIASES = {} M.GTE_COMMAND_INPUTS = {} M.GTE_COMMAND_OUTPUTS = {} M.GTE_COMMAND_LATCH_WINDOWS = {} for canon, row in pairs(M.GTE_COMMAND) do M.GTE_COMMAND_ALIASES[canon] = canon M.INSTRUCTION_LATENCY[canon] = row.cycles M.INSTRUCTION_GPR_EFFECTS[canon] = { reads = {}, writes = {} } for _, alias in ipairs(row.aliases or {}) do M.GTE_COMMAND_ALIASES[alias] = canon M.INSTRUCTION_LATENCY[alias] = row.cycles M.INSTRUCTION_GPR_EFFECTS[alias] = { reads = {}, writes = {} } end M.GTE_COMMAND_INPUTS[canon] = row.inputs M.GTE_COMMAND_OUTPUTS[canon] = row.outputs M.GTE_COMMAND_LATCH_WINDOWS[canon] = row.latch end end build_isa_views() --- GTE control-register alias groups. --- Aliases within a group write to the same C2 control-register slot on real silicon --- (the silicon double-maps some C2 slots across multiple PSX SDK / libgte conventions). --- Aliases across groups write to distinct C2 slots. --- --- Cross-alias writes inside one atom body, or across the wave-context boundary, --- silently clobber each other. The `check_gte_cr_alias_writes` check warns about --- each pair per source. See `docs/gte_reference.md` §"Control-register alias table" --- for the silicon rationale and the libgte outer-product convention. M.GTE_CR_ALIAS_GROUPS = { { 24, { "gte_cr_RBK", "gte_cr_OFX" } }, -- background R vs screen offset X { 25, { "gte_cr_GBK", "gte_cr_OFY" } }, -- background G vs screen offset Y { 26, { "gte_cr_BBK", "gte_cr_H" } }, -- background B vs projection plane distance H } -- Packed RT slots named by the gte.h packed-slot comment. -- first must be written before second. M.GTE_PACKED_SLOT_RELATIONS = { { slot = 2, first = "gte_cr_RT13", second = "gte_cr_RT22" }, } -- Operand-class table for the COP2->GPR load-delay check. -- Maps each emitting-token ident to the set of GPR operand positions it reads. -- Covers the current encoder vocabulary (`code/duffle/mips.h` + `code/duffle/gte.h`); add rows here as new encoders land. -- -- Semantics: -- * A "GPR operand position" is the textual slot in the macro's argument list, 1-based; e.g. `load_word(rt, base, off)` has -- positional operands 1 (rt), 2 (base), 3 (off). The table reads operands 1 + 2 + 3 to find what GPRs the macro touches. -- * The check tracks one entry per destination GPR per MFC2 / CFC2 event. -- A subsequent event counts as a "use" iff any of its read operand positions reference that destination GPR's ident (e.g. `R_T0`). -- * Branch delay slots are out of scope (MIPS control-flow; tracked separately). M.OPERAND_READ_POSITIONS = { -- CPU ALU with one or two GPR operands. Reads every GPR operand. ["add_ui"] = {1, 2}, ["li_s"] = {1, 2}, -- rt (write), imm16 (immediate) ["add_ui_self"] = {1}, ["add_si"] = {1, 2}, ["add_u"] = {1, 2, 3}, ["add_u_self"] = {1, 2}, ["sub_s"] = {1, 2, 3}, ["sub_u"] = {1, 2, 3}, ["and_i"] = {1, 2}, ["and"] = {1, 2, 3}, ["or_i"] = {1, 2}, ["or_i_self"] = {1}, ["or"] = {1, 2, 3}, ["or_self"] = {1, 2}, ["xor_i"] = {1, 2}, ["xor"] = {1, 2, 3}, ["slt_s"] = {1, 2, 3}, ["slt_u"] = {1, 2, 3}, ["slt_si"] = {1, 2}, ["slt_ui"] = {1, 2}, ["mult_s"] = {1, 2}, ["mult_u"] = {1, 2}, ["div_s"] = {1, 2}, ["div_u"] = {1, 2}, -- Shifts: shift_lleft(rd, rt, shamt); the rt operand is the value, rd is dest. ["shift_lleft"] = {1, 2}, ["shift_lright"] = {1, 2}, ["shift_aright"] = {1, 2}, ["shift_lleft_self"] = {1}, -- Loads: load_word(rt, base, off); the rt operand is the destination (it's written, not read) and base + off are non-GPR operands. -- The check treats the rt operand as a write, so the read-positions table for `load_*` is empty. ["load_word"] = {}, ["load_half_u"] = {}, ["load_byte_u"] = {}, ["load_half"] = {}, ["load_byte"] = {}, ["load_upper_i"] = {}, ["load_ui"] = {}, -- Stores write to memory; base + rt operands are non-read for load-delay purposes. ["store_word"] = {}, ["store_half"] = {}, ["store_byte"] = {}, -- Branches read rs (+ rt for beq/bne). The branch delay slot is out of scope. ["branch_equal"] = {1, 2}, ["branch_ne"] = {1, 2}, ["branch_le_zero"] = {1}, ["branch_lt_zero"] = {1}, ["branch_ge_zero"] = {1}, ["branch_gt_zero"] = {1}, -- Jumps / link: jr / jalr read rs only (the target). RD is the destination link. ["jump_reg"] = {1}, ["jump_link"] = {1}, ["call_reg"] = {1}, ["call_addr"] = {}, ["jump"] = {}, -- mask_upper is a 2-word macro: shift_lleft then shift_lright. The first reads rt. ["mask_upper"] = {1, 2}, -- move from/to HI/LO. ["mov_from_high"] = {}, ["mov_from_low"] = {}, ["mov_to_high"] = {1}, ["mov_to_low"] = {1}, -- GTE transfers / loads / stores / commands: the relevant table values live in the check itself. -- `gte_mv_to_*` writes its rt operand; `gte_mv_from_*` writes its rt operand; `gte_*` commands are atomic-from-the-CPU-POV -- once they issue (the CPU holds until the command completes, so load-delay violations don't surface here). ["gte_mv_from_data_r"] = {}, ["gte_mv_from_ctrl_r"] = {}, ["gte_mv_to_data_r"] = {}, ["gte_mv_to_ctrl_r"] = {}, ["gte_lw"] = {}, ["gte_sw"] = {}, ["shift_lleft_var"] = {1, 2, 3}, -- rd, rt, rs (variable shift amount) ["shift_aright_var"] = {1, 2, 3}, } -- GP0 packet sizes (total words including the 1-word tag) per GP0 cmd byte. -- Per PSX-SPX `docs/psx-spx/docs/graphicsprocessingunitgpu.md` §"GPU Render Polygon Commands": -- Each polygon command's word count = 1 (tag/cmd) + per-vertex (vertex + optional color + optional UV). -- F3: cmd + 3 vertices = 4 words; +1 tag = 5 -- F4: cmd + 4 vertices = 5 words; +1 tag = 6 -- G3: cmd + 3×(color + vertex) = 6 words; +1 tag = 7 -- G4: cmd + 4×(color + vertex) = 8 words; +1 tag = 9 -- FT3: cmd + tpage + clut + 3×(vertex + UV) = 7 words; +1 tag = 8 -- FT4: cmd + tpage + clut + 4×(vertex + UV) = 9 words; +1 tag = 10 -- GT3: cmd + tpage + clut + 3×(color + vertex + UV) = 9 words; +1 tag = 10 -- GT4: cmd + tpage + clut + 4×(color + vertex + UV) = 12 words; +1 tag = 13 -- -- Cross-checked against code/duffle/gp.h struct sizes + the set_poly_* macros -- (which encode "len" = "words after tag"): -- set_poly_f3(p) -> set_len(p, 4) -> 5 total GP0 0x20 -- set_poly_ft3(p) -> set_len(p, 7) -> 8 total GP0 0x24 -- set_poly_f4(p) -> set_len(p, 5) -> 6 total GP0 0x28 -- set_poly_ft4(p) -> set_len(p, 9) -> 10 total GP0 0x2C -- set_poly_g3(p) -> set_len(p, 6) -> 7 total GP0 0x30 -- set_poly_gt3(p) -> set_len(p, 9) -> 10 total GP0 0x34 -- set_poly_g4(p) -> set_len(p, 8) -> 9 total GP0 0x38 -- set_poly_gt4(p) -> set_len(p, 12) -> 13 total GP0 0x3C M.GP0_CMD_SIZE = { [0x20] = 5, -- Poly_F3 [0x24] = 8, -- Poly_FT3 [0x28] = 6, -- Poly_F4 [0x2C] = 10, -- Poly_FT4 [0x30] = 7, -- Poly_G3 [0x34] = 10, -- Poly_GT3 [0x38] = 9, -- Poly_G4 [0x3C] = 13, -- Poly_GT4 } -- Shape suffix (after `ac_format_` / `mac_format_` prefix) -> GP0 cmd byte. -- Lets the static-analysis check derive the cmd byte from a macro name like `mac_format_g4_color` -> `g4` -> 0x38 -> 9 expected words. M.GP0_CMD_BY_SHAPE = { ["f3"] = 0x20, ["ft3"] = 0x24, ["f4"] = 0x28, ["ft4"] = 0x2C, ["g3"] = 0x30, ["gt3"] = 0x34, ["g4"] = 0x38, ["gt4"] = 0x3C, } M.UNKNOWN_INSTRUCTION_CYCLES = 1 -- Hardware-relation policy table. -- -- The forward walker in `passes/static_analysis.lua::analyze_hardware_relations` reads every emitted word_event, matches its `encoder` against `row.token`, and: -- * stages the event as a producer in `atom.paths.forward_state`; or -- * matches it as a consumer against pending producers and records a hazard on `atom.paths.hazards` when the gap is below `visibility.required`. -- -- Each row is the contract for one CPU-to-coprocessor transfer semantic (the coprocessor-to-CPU path mirrors the same shape). -- The `reads` / `writes` sub-tables carry the argument positions the analyzer inspects: -- * `writes.arg` is the destination operand (the producer's effect); the analyzer stages this register as a pending producer. -- * `reads` (when present) lists the operand positions the same token reads back from hardware; for MTC2 / CTC2 the producer reads the GPR source it is loading from. -- The `fanout_to` field (MTC2-IRGB row only) tells the consumer-match logic which downstream COP2 registers are transitively updated by the write. -- -- Visibility semantics: -- * `kind = "post_producer_words"` means the consumer observes the producer's effect after `required` independent emitted words that are -- strictly between the producer and the consumer. The producer's own emitted slot is implicit (it counts as the slot of issue, not toward -- `required`) — per the PSX-SPX rule: "Store delays are counted in numbers of clock cycles (not in numbers of opcodes). For 3 cycle delay, -- one must usually insert 3 cached opcodes (or one uncached opcode)." -- * `required` is the minimum count of intervening emitted words between producer and consumer. -- `required = 0` permits the consumer on the very next slot; `required < 0` would place the consumer on the same slot as the producer -- and is reserved for future "self-retires" relations. -- -- Evidence: -- * `evidence.confidence` is one of `"exact"`, `"conservative"`, `"unknown"`. The severity comes from `violation_kind`; -- A hardware measurement that the vendor caveats may still classify as `"conservative"` even when the underlying timing is numerically known. -- * `evidence.source` is the upstream reference (file + line range) the row is sourced from. New rows must carry this citation. -- -- Consumers: -- * passes/static_analysis.lua::analyze_hardware_relations (forward walker). -- * passes/static_analysis.lua::transfer_hazards CHECK_RULES reader (renders hazards onto `findings`). -- This table is consumed by the hardware-relation analyzer and hazard renderer. M.HARDWARE_RELATIONS = { -- CPU → COP2 data register (MTC2). The ordinary default is 2 cached words between producer and consumer (cpuspecifications.md:407-419). { id = "mtc2_gpr_visibility", semantic = "MTC2", token = "gte_mv_to_data_r", direction = "gpr_to_cop2_data", reads = { domain = "gpr", arg = 1 }, writes = { domain = "cop2.data", arg = 2 }, visibility = { kind = "post_producer_words", required = 2 }, evidence = { confidence = "exact", source = "cpuspecifications.md:407-419", }, violation_kind = "error", }, -- CPU → COP2 data register when the destination is C2_IRGB (data 28). -- C2_IRGB drives the IR1/IR2/IR3 color-conversion fan-out, which extends the propagation delay to 3 cached words. -- `destination_match = "C2_IRGB"` is the row's filter; the analyzer consults this when the producer's destination operand equals "C2_IRGB". -- C2_ORGB (data 29) is read-only and is never classified as a writable fan-out destination. { id = "mtc2_irgb_visibility", semantic = "MTC2", token = "gte_mv_to_data_r", direction = "gpr_to_cop2_data", reads = { domain = "gpr", arg = 1 }, writes = { domain = "cop2.data", arg = 2 }, destination_match = "C2_IRGB", fanout_to = { "C2_IR1", "C2_IR2", "C2_IR3" }, visibility = { kind = "post_producer_words", required = 3 }, evidence = { confidence = "exact", source = "cpuspecifications.md:407-419", }, violation_kind = "error", }, -- CPU → COP2 control register (CTC2). Ordinary minimum 2; -- no IRGB-style fan-out exists for control registers (per spec §3.6: only C2_IRGB has the 3-cycle fan-out on the data side). { id = "ctc2_gpr_visibility", semantic = "CTC2", token = "gte_mv_to_ctrl_r", direction = "gpr_to_cop2_control", reads = { domain = "gpr", arg = 1 }, writes = { domain = "cop2.ctrl", arg = 2 }, visibility = { kind = "post_producer_words", required = 2 }, evidence = { confidence = "exact", source = "cpuspecifications.md:407-419", }, violation_kind = "error", }, -- COP2 data → GPR (MFC2). One cached slot between the transfer and the first GPR consumer; -- the GPR is not updated until the instruction AFTER the MFC2 completes (geometrytransformationenginegte.md:29-32). { id = "mfc2_gpr_visibility", semantic = "MFC2", token = "gte_mv_from_data_r", direction = "cop2_data_to_gpr", reads = { domain = "cop2.data", arg = 2 }, writes = { domain = "gpr", arg = 1 }, visibility = { kind = "post_producer_words", required = 1 }, evidence = { confidence = "exact", source = "geometrytransformationenginegte.md:29-32", }, violation_kind = "error", }, -- COP2 control → GPR (CFC2). Same delay as MFC2 (cpuspecifications.md treats the two load-from-COP2 paths symmetrically). { id = "cfc2_gpr_visibility", semantic = "CFC2", token = "gte_mv_from_ctrl_r", direction = "cop2_control_to_gpr", reads = { domain = "cop2.ctrl", arg = 2 }, writes = { domain = "gpr", arg = 1 }, visibility = { kind = "post_producer_words", required = 1 }, evidence = { confidence = "exact", source = "cpuspecifications.md:382-419", }, violation_kind = "error", }, -- COP0 control → GPR (MFC0). -- One cached slot; the analyzer treats `sys_mov_from_cop0(rt, 12)` (the SR/CU2 transfer) as the same shape as the COP2 load-delay path. -- The semantic-level SR/CU2 transition models the load delay; -- SR.CU2 bounded-value propagation is modeled separately). { id = "mfc0_gpr_visibility", semantic = "MFC0", token = "sys_mov_from_cop0", direction = "cop0_control_to_gpr", reads = { domain = "cop0.ctrl", arg = 2 }, writes = { domain = "gpr", arg = 1 }, visibility = { kind = "post_producer_words", required = 1 }, evidence = { confidence = "exact", source = "cpuspecifications.md:171-178", }, violation_kind = "error", }, -- Memory -> COP2 data register (LWC2). -- The memory-side timing is not measured by the vendored GTE latch experiment, so this relation has no numeric retirement threshold. -- The LWC2 destination has TWO retirement regimes (per PSX-SPX): -- * GTE-command consumer (`gte_cmdw_*`): the GTE pipeline LATCHES the LWC2 result, so a `gte_cmdw_*` -- in the very next slot uses the latched value. Gap = 0 is allowed. (Per `docs/psx-spx/docs/gtepipelinetimings.md:271-274`.) -- * Any other consumer: standard MIPS load delay applies. Gap = 1 required. (Per `docs/psx-spx/docs/cpuspecifications.md:407-419`.) -- Two separate relations so the walker can dispatch by consumer type and emit different severities -- (the GTE-command path is `info` because the latch is intentional; the non-GTE-consumer path is `error` because the missing nop is a real bug). { id = "lwc2_to_gte_command", semantic = "LWC2_to_GTE", token = "gte_lw", direction = "memory_to_cop2_data", reads = { domain = "memory", arg = 2 }, writes = { domain = "cop2.data", arg = 1 }, required = 0, -- GTE-command consumer: gap = 0 OK (latched). evidence = { confidence = "measured", source = "gtepipelinetimings.md:271-274", }, violation_kind = "info", clear_on_consumer = true, }, { id = "lwc2_to_other_consumer", semantic = "LWC2_to_other", token = "gte_lw", direction = "memory_to_cop2_data", reads = { domain = "memory", arg = 2 }, writes = { domain = "cop2.data", arg = 1 }, required = 1, -- Non-GTE-consumer: standard MIPS load delay. evidence = { confidence = "inferred", source = "cpuspecifications.md:407-419", }, violation_kind = "error", clear_on_consumer = true, }, -- COP2 data register -> memory (SWC2). A read of C2 state, not a CPU-to-COP2 write. -- The policy row stays in for direction/provenance; staging it as a later command-input producer is suppressed. { id = "swc2_memory_write", semantic = "SWC2", token = "gte_sw", direction = "cop2_data_to_memory", reads = { domain = "cop2.data", arg = 1 }, writes = { domain = "memory", arg = 2 }, visibility = { kind = "none", required = 0 }, evidence = { confidence = "exact", source = "cpuspecifications.md:79", }, violation_kind = "info", stage = false, }, -- MTC0 Status/SR.CU2. The ordinary COP0 store has no general store-delay relation; -- this row feeds the dedicated CU2 transition logic in the same forward walk and is therefore not staged in `pending`. { id = "mtc0_cu2_visibility", semantic = "MTC0", token = "sys_mov_to_cop0", direction = "gpr_to_cop0_status", reads = { domain = "gpr", arg = 1 }, writes = { domain = "cop0.status", arg = 2 }, status_register = 12, visibility = { kind = "post_producer_words", required = 2 }, evidence = { confidence = "conservative", source = "cpuspecifications.md:543,625-628", }, violation_kind = "warning", stage = false, cu2_transition = true, }, } -- Bounded Status/SR.CU2 transition policy. -- The value lattice and the transition consumer both read this immutable row; no second value pass is permitted. -- The source says the enable/disable transition takes "2 clock cycles or so", so the boundary is conservative rather than exact. M.CU2_TRANSITION_POLICY = { status_register = 12, enable_bit = 0x40000000, required = 2, visibility_kind = "post_producer_words", evidence = { confidence = "conservative", source = "cpuspecifications.md:543,625-628", }, } return M