Author SHA1 Message Date
ed 86fe189b4e Moving definitions to use dedicated scratch register. 2026-08-18 09:26:49 -04:00
ed da007d342e Reviewing. Successfuly reworked register allocation for tape runs. 2026-08-18 00:26:36 -04:00
ed 5a4bfb1224 Collapse of atom 6-9 into a single atom (finaly). Generalized cross product atom proc and atom component. Still working on normalize_v3s4. 2026-08-17 18:18:18 -04:00
ed d4795cf9de Extract out a cross roduct component. 2026-08-17 10:34:43 -04:00
ed e79c364b40 reducing cross product atom procs to a single one in gte for once. 2026-08-17 01:05:58 -04:00
ed 18b1d5a04b remove outdated comments. 2026-08-16 12:07:09 -04:00
ed 581b00b960 wip: going over all codepaths. 2026-08-16 10:35:32 -04:00
ed 3faccfc283 more reviewing, thinking about atom bundles... 2026-08-16 01:22:53 -04:00
ed 1a0d417649 lua metaprogram improvmeents 2026-08-15 22:24:26 -04:00
ed 3301826f5c reviewed: resolve_look_at__populate_proc 2026-08-15 21:34:55 -04:00
ed d9b9241e2c resolve_look_at__cross_uz_ux_to_up_proc reviewed 2026-08-15 19:53:08 -04:00
ed a16c727db2 updates to lua program to furhter support new constructs and correct report errors. 2026-08-15 19:52:56 -04:00
ed 8a825a59c7 Add RegUse_ support to the lua metaprogram. Ideated further on type mapping atom comonents to their base component op (math distinctions annotated in the asm). 2026-08-15 15:51:06 -04:00
ed f8b28be02e Lua metaprogram support for RegUse_ (needs review) 2026-08-15 11:54:51 -04:00
ed ffc66052f8 Curating duffle, preparing to update metaprogram for latest atom asm ideation. Reviewing the resolve_look_at atoms further... 2026-08-15 11:21:28 -04:00
ed 7764612325 add install extension script 2026-08-15 01:19:45 -04:00
ed 1a5b618484 done with this theming stuff for now. 2026-08-15 01:15:43 -04:00
ed d23b6a2a36 messing around. 2026-08-14 22:43:27 -04:00
ed 7ec778a68e more theme stuff 2026-08-14 21:48:52 -04:00
ed 9ca865d5db update license to zip for now...
not sure what the standalone repo is going to be yet, but it will be relatively permissive since this is prototype/educational setup
2026-08-14 19:43:03 -04:00
ed 764ded4557 initial plugin setup for syntax highlighting in vscode... 2026-08-14 18:57:37 -04:00
ed 67a84d34f3 oops: endregion 2026-08-14 13:41:45 -04:00
ed baaff12f33 Ideating on "RegUse_" patterned structs for describe register allocatins to mips atom proc. 2026-08-14 12:38:00 -04:00
ed b695056b9a finished reviewing normalize_v3s4 for now 2026-08-14 03:45:34 -04:00
ed 3a4d6304dd static analysis: immeidate field awarenss 2026-08-14 01:22:54 -04:00
ed a535d381ed remove encoding masks from gp (unnecessary, hides errors) 2026-08-14 01:22:36 -04:00
ed c447bfa877 fixes to the reg file allocator, exploring... 2026-08-14 00:43:19 -04:00
ed d88e0d0487 remove mask from mips and gte instruction encoders. missing math changes. 2026-08-13 23:39:35 -04:00
ed 9a6eca6047 more review, made a register file allocator (drafted, kinda iffy, want todo comp-time as well). 2026-08-13 23:39:03 -04:00
ed 5c9c61720f Redesign: Not making local var in MipsAtom_Proc_ or MipsAtomComp_Proc_ have sym tied to proc name. Adjusted parser as well base do that. 2026-08-13 21:42:06 -04:00
ed b8e31123e4 editing/reading. 2026-08-13 21:22:29 -04:00
ed ea3e30a11e oops 2026-08-13 20:51:45 -04:00
ed 37f4712237 gutting nosiy comments. Looking into some atom components.. 2026-08-13 19:55:17 -04:00
62 changed files with 8372 additions and 5258 deletions
+26
View File
@@ -0,0 +1,26 @@
# Cozy and Windy
Editor theme ported from the Rider scheme of the same name.
It colors the editor surface, C/C++ syntax, and tape-atom DSL keywords
emitted by `local.tape-atom-syntax`. It does not change workbench chrome.
## Install
```powershell
cd C:\projects\Pikuma\ps1\.vscode\cozy-and-windy
npm run package
code --install-extension .\cozy-and-windy-0.1.0.vsix --force
```
Reload the window. Select **Cozy and Windy** as the color theme, or set
`workbench.colorTheme` to `Cozy and Windy` in the PS1 workspace settings.
Keep `local.tape-atom-syntax` installed. This theme colors those token
types; it does not classify them.
## Inspect
Open `hello_camera.atom.c` and run **Developer: Inspect Editor Tokens and Scopes**
on `MipsAtom_`, an atom name, `atom_info`, `R_PrimCursor`, a `gte_*` call,
and a `mac_*` call.
Binary file not shown.
+25
View File
@@ -0,0 +1,25 @@
{
"name": "cozy-and-windy",
"displayName": "Cozy and Windy",
"description": "Editor theme ported from the Rider Cozy and Windy scheme. Colors C/C++ and tape-atom DSL keywords.",
"publisher": "local",
"version": "0.1.0",
"engines": {
"vscode": "^1.80.0"
},
"categories": [
"Themes"
],
"scripts": {
"package": "npx --yes @vscode/vsce@3.6.1 package --allow-missing-repository --skip-license --out cozy-and-windy-0.1.0.vsix"
},
"contributes": {
"themes": [
{
"label": "Cozy and Windy",
"uiTheme": "vs-dark",
"path": "./themes/cozy-and-windy-color-theme.json"
}
]
}
}
@@ -0,0 +1,132 @@
{
"name": "Cozy and Windy",
"type": "dark",
"semanticHighlighting": true,
"colors": {
// 121212
// 111212
// 211f1e
// 191817
"editor.background": "#191817",
"editor.foreground": "#dfc6ba",
"editor.lineHighlightBackground": "#1c1c1c",
"editor.selectionBackground": "#164371",
"editor.selectionForeground": "#c8c8c8",
"editorLineNumber.foreground": "#43c3c3",
"editorLineNumber.activeForeground": "#00fff4",
"editorIndentGuide.background1": "#181818",
"editorIndentGuide.activeBackground1": "#202020",
"editorRuler.foreground": "#505050",
"editorGutter.background": "#211f1e",
"editorBracketMatch.background": "#3b514d",
"editor.foldBackground": "#0c0c0c6a",
"editor.wordHighlightBackground": "#211f1e4d",
"editor.wordHighlightStrongBackground": "#303030",
"editorCursor.foreground": "#00fff4",
"editorWhitespace.foreground": "#181818",
// "editorLineHighlightBorder": "#1c1c1c",
"editorWidget.background": "#211f1e",
"editorSuggestWidget.background": "#2c334b",
"editorHoverWidget.background": "#2c334b"
},
"semanticTokenColors": {
"comment": { "foreground": "#868686", "fontStyle": "italic" },
"keyword": { "foreground": "#d8bd5b" },
"string": { "foreground": "#d46a54" },
"number": { "foreground": "#b5cea8" },
"operator": { "foreground": "#be8e78" },
"class": { "foreground": "#54a4d6" },
"struct": { "foreground": "#54a4d6" },
"enum": { "foreground": "#54a4d6" },
"type": { "foreground": "#54a4d6" },
"interface": { "foreground": "#7984ab" },
"function": { "foreground": "#cccab5" },
// "function": { "foreground": "#6090a9" },
"method": { "foreground": "#6090a9" },
"variable": { "foreground": "#bc966c" },
"parameter": { "foreground": "#ce8365" },
"property": { "foreground": "#acb8c8" },
"*.static": { "foreground": "#9e95c6" },
"macro": { "foreground": "#5ea852" },
"namespace": { "foreground": "#8e8e8e" },
"typeParameter": { "foreground": "#b8d7a3" },
"enumMember": { "foreground": "#a373b0" },
"label": { "foreground": "#c8c8c8", "fontStyle": "bold" },
"tapeAtomKeyword": { "foreground": "#d8bd5b", "fontStyle": "bold" },
"tapeAtomName": { "foreground": "#b1b7d6", "fontStyle": "bold" },
"tapeComponentKeyword": { "foreground": "#d68a36", "fontStyle": "bold" },
"tapeComponentName": { "foreground": "#b1b7d6", "fontStyle": "bold" },
"tapeAnnotation": { "foreground": "#d8bd5b" },
"tapeBindType": { "foreground": "#54a4d6" },
"tapePhase": { "foreground": "#b8d7a3", "fontStyle": "italic" },
"tapeLabel": { "foreground": "#959595", "fontStyle": "bold" },
// "tapeCpuInstruction": { "foreground": "#6d9aa0" },
// "tapeCpuInstruction": { "foreground": "#cf7539" },
// "tapeCpuInstruction": { "foreground": "#d16b3a" },
"tapeCpuInstruction": { "foreground": "#d5895a" },
"tapeGteInstruction": { "foreground": "#988bcb" },
"tapeGpuInstruction": { "foreground": "#bf7dac" },
"tapeComponentInstruction": { "foreground": "#8baa5d" },
// "tapeGprRegister": { "foreground": "#92d4d9" },
"tapeGprRegister": { "foreground": "#a2bfa8" },
"tapeCop2Register": { "foreground": "#945cd9" },
"tapeDuffleType": { "foreground": "#54a4d6" },
"tapeAttribute": { "foreground": "#73a07c" },
"tapeGprRegister.tapeRead": { "foreground": "#5bb8b0", "fontStyle": "italic" },
"tapeGprRegister.tapeWrite": { "foreground": "#2d8f8c", "fontStyle": "bold" },
"tapeCop2Register.tapeRead": { "foreground": "#b08ae0", "fontStyle": "italic" },
"tapeCop2Register.tapeWrite": { "foreground": "#7b3ec4", "fontStyle": "bold" },
// "*.tapeAuto": { },
"tapeControlFlow": { "foreground": "#63d169", "fontStyle": "bold" },
"tapeDelaySlot": { "foreground": "#ff5647" }
},
"tokenColors": [
{ "scope": ["comment", "comment.block", "comment.line", "comment.block.documentation"], "settings": { "foreground": "#868686", "fontStyle": "italic" } },
{ "scope": ["keyword", "keyword.control", "keyword.other"], "settings": { "foreground": "#d8bd5b" } },
{ "scope": ["string", "string.quoted"], "settings": { "foreground": "#d46a54" } },
{ "scope": ["string.quoted.other"], "settings": { "foreground": "#d69d85" } },
{ "scope": ["constant.numeric"], "settings": { "foreground": "#b5cea8" } },
{ "scope": ["punctuation", "keyword.operator"], "settings": { "foreground": "#be8e78" } },
{ "scope": ["keyword.operator.overload"], "settings": { "foreground": "#b87e76" } },
{ "scope": ["entity.name.type", "entity.name.type.class", "entity.name.type.struct", "entity.name.type.enum"], "settings": { "foreground": "#54a4d6" } },
{ "scope": ["entity.name.type.interface"], "settings": { "foreground": "#7984ab" } },
{ "scope": ["entity.name.function"], "settings": { "foreground": "#cccab5" } },
{ "scope": ["entity.name.function.member"], "settings": { "foreground": "#6090a9" } },
{ "scope": ["variable.other.local"], "settings": { "foreground": "#bc966c" } },
{ "scope": ["variable.parameter"], "settings": { "foreground": "#ce8365" } },
{ "scope": ["variable.other.property"], "settings": { "foreground": "#acb8c8" } },
{ "scope": ["variable.other.constant"], "settings": { "foreground": "#9e95c6" } },
{ "scope": ["variable.other.global", "variable.other.defaultLibrary"], "settings": { "foreground": "#bf7dac" } },
{ "scope": ["support.type", "support.function"], "settings": { "foreground": "#8baa5d" } },
{ "scope": ["meta.preprocessor"], "settings": { "foreground": "#5ea852" } },
// { "scope": ["variable.parameter.preprocessor"], "settings": { "foreground": "#636363" } },
{ "scope": ["variable.parameter.preprocessor"], "settings": { "foreground": "#bc966c" } },
{ "scope": ["keyword.control.directive"], "settings": { "foreground": "#d68a36" } },
{ "scope": ["entity.name.namespace"], "settings": { "foreground": "#8e8e8e" } },
{ "scope": ["entity.name.type.parameter"], "settings": { "foreground": "#b8d7a3" } },
{ "scope": ["variable.other.enummember"], "settings": { "foreground": "#a373b0" } },
{ "scope": ["entity.name.type.concept"], "settings": { "foreground": "#76ff7d" } },
{ "scope": ["entity.name.type.dependent"], "settings": { "foreground": "#448b5a", "fontStyle": "bold" } },
{ "scope": ["entity.name.label"], "settings": { "foreground": "#c8c8c8", "fontStyle": "bold" } },
{ "scope": ["invalid"], "settings": { "foreground": "#ff5647" } },
{ "scope": ["keyword.codetag.todo"], "settings": { "foreground": "#c10000", "fontStyle": "bold italic" } },
{ "scope": ["keyword.control.duffle.atom"], "settings": { "foreground": "#d8bd5b", "fontStyle": "bold" } },
{ "scope": ["entity.name.function.duffle.atom"], "settings": { "foreground": "#cccab5", "fontStyle": "bold" } },
{ "scope": ["keyword.control.duffle.component"], "settings": { "foreground": "#d68a36", "fontStyle": "bold" } },
{ "scope": ["entity.name.function.duffle.component"], "settings": { "foreground": "#6090a9", "fontStyle": "bold" } },
{ "scope": ["support.function.duffle.annotation"], "settings": { "foreground": "#d8bd5b" } },
{ "scope": ["entity.name.type.duffle.bind"], "settings": { "foreground": "#54a4d6" } },
{ "scope": ["entity.name.tag.duffle.phase"], "settings": { "foreground": "#b8d7a3", "fontStyle": "italic" } },
{ "scope": ["entity.name.label.duffle.atom"], "settings": { "foreground": "#c8c8c8", "fontStyle": "bold" } },
{ "scope": ["support.function.duffle.cpu"], "settings": { "foreground": "#6d9aa0" } },
{ "scope": ["support.function.duffle.gte"], "settings": { "foreground": "#988bcb" } },
{ "scope": ["support.function.duffle.gpu"], "settings": { "foreground": "#bf7dac" } },
{ "scope": ["support.function.duffle.component"], "settings": { "foreground": "#8baa5d" } },
{ "scope": ["keyword.control.duffle.branch"], "settings": { "foreground": "#76ff7d", "fontStyle": "bold" } },
{ "scope": ["keyword.operator.duffle.delayslot"], "settings": { "foreground": "#ff5647" } },
{ "scope": ["variable.other.constant.duffle.gpr"], "settings": { "foreground": "#3fa8a6" } },
{ "scope": ["variable.other.constant.duffle.cop2"], "settings": { "foreground": "#945cd9" } },
{ "scope": ["storage.type.duffle.type"], "settings": { "foreground": "#54a4d6" } },
{ "scope": ["storage.modifier.duffle.attr"], "settings": { "foreground": "#73a07c" } }
]
}
+43
View File
@@ -0,0 +1,43 @@
# Package and install the local VS Code Insiders extensions under .vscode/.
# Usage:
# .\install_extensions.ps1
# .\install_extensions.ps1 -SkipPackage
param([switch] $SkipPackage)
$path_vscode = $PSScriptRoot
$code_insiders = "C:\apps\Microsoft VS Code Insiders\bin\code-insiders.cmd"
if (-not (test-path -literalpath $code_insiders)) {
$found = get-command code-insiders -erroraction silentlycontinue
if ($found) { $code_insiders = $found.source }
}
if (-not (test-path -literalpath $code_insiders)) { throw "code-insiders not found. Install VS Code Insiders or add it to PATH." }
$extensions = @(
(join-path $path_vscode "tape-atom-syntax"),
(join-path $path_vscode "cozy-and-windy")
)
foreach ($extension in $extensions) {
$package_json = join-path $extension "package.json"
if (-not (test-path -literalpath $package_json)) { throw "missing $package_json" }
$manifest = get-content -literalpath $package_json -raw | convertfrom-json
$vsix = join-path $extension ("{0}-{1}.vsix" -f $manifest.name, $manifest.version)
if (-not $SkipPackage) {
if (-not $manifest.scripts.package) { throw "$package_json has no scripts.package" }
write-host "packaging $($manifest.displayName) ($($manifest.name)@$($manifest.version))"
& npm --prefix $extension run package
if ($LASTEXITCODE -ne 0) { throw "npm run package failed for $extension" }
}
if (-not (test-path -literalpath $vsix)) { throw "missing $vsix" }
write-host "installing $vsix"
& $code_insiders --install-extension $vsix --force
if ($LASTEXITCODE -ne 0) { throw "install failed for $vsix" }
}
write-host "done. reload the Insiders window (Developer: Reload Window)."
Binary file not shown.
+222
View File
@@ -0,0 +1,222 @@
"use strict";
const { nearestCall } = require("./lexer");
const { mergeIndexes, scanSource } = require("./source-index");
const TOKEN_TYPES = [
"tapeAtomKeyword",
"tapeAtomName",
"tapeComponentKeyword",
"tapeComponentName",
"tapeAnnotation",
"tapeBindType",
"tapePhase",
"tapeLabel",
"tapeCpuInstruction",
"tapeControlFlow",
"tapeGteInstruction",
"tapeGpuInstruction",
"tapeComponentInstruction",
"tapeDelaySlot",
"tapeGprRegister",
"tapeCop2Register",
"tapeDuffleType",
"tapeAttribute",
"keyword",
"macro",
];
const TOKEN_MODIFIERS = ["declaration", "tapeRead", "tapeWrite", "tapeAuto"];
const TOKEN_TYPE_INDEX = new Map(TOKEN_TYPES.map((name, index) => [name, index]));
const TOKEN_MODIFIER_INDEX = new Map(TOKEN_MODIFIERS.map((name, index) => [name, index]));
const ATOM_KEYWORDS = new Set(["MipsAtom_", "MipsAtom_Proc_"]);
const COMPONENT_KEYWORDS = new Set(["MipsAtomComp_", "MipsAtomComp_Proc_"]);
const ANNOTATIONS = new Set([
"atom_info", "atom_bind", "atom_reads", "atom_writes", "atom_label",
"atom_offset", "atom_reg", "atom_type", "atom_ctx", "atom_phase",
"atom_auto_reg", "phase_auto_reg", "atom_dbg_skip",
]);
const DSL_KEYWORDS = new Set([
"FI_", "I_", "NI_", "Relative_", "Struct_", "Enum_", "Union_", "Array_",
"Slice_", "TypeR_", "TypeV_", "align_", "internal", "local_persist", "global",
"RO_", "LP_", "gknown", "expect_", "cexpr_",
"asm", "asm_words", "asm_rpins", "asm_clobber",
"O_", "S_", "C_", "T_", "tmpl", "glue", "r_", "v_", "tr_", "tv_",
"rgcc", "r_use", "r_set", "r_mod", "r_imm", "r_mem",
"u1_", "u2_", "u4_", "u8_", "s1_", "s2_", "s4_", "s8_",
"u1_r", "u2_r", "u4_r", "u8_r", "u1_v", "u2_v", "u4_v", "u8_v",
]);
const DELAY_SLOT_KEYWORDS = new Set(["LdSlot_", "BdSlot_", "DmaSlot_", "GteDelay_"]);
const CONTROL_FLOW_PREFIXES = /^(?:branch_|jump_|call_)/;
const ROLE_TO_TYPE = {
atomName: "tapeAtomName",
componentName: "tapeComponentName",
bindType: "tapeBindType",
duffleType: "tapeDuffleType",
gprRegister: "tapeGprRegister",
cop2Register: "tapeCop2Register",
};
function registerType(name, index) {
const kind = index.registers.get(name);
if (kind === "gpr" || /^R_[A-Za-z0-9_]+$/.test(name)) return "tapeGprRegister";
if (kind === "cop2" || /^(?:C2_|gte_cr_)[A-Za-z0-9_]+$/.test(name)) return "tapeCop2Register";
return null;
}
function instructionType(name, index) {
const domain = index.macros.get(name);
if (domain === "control") return "tapeControlFlow";
if (domain === "cpu") return "tapeCpuInstruction";
if (domain === "gte") return "tapeGteInstruction";
if (domain === "gpu") return "tapeGpuInstruction";
if (domain === "component") {
if (/^mac_gte_/.test(name)) return "tapeGteInstruction";
if (/^mac_gp/.test(name)) return "tapeGpuInstruction";
if (/^mac_/.test(name)) return "tapeComponentInstruction";
return "macro";
}
if (domain === "utility") return "macro";
if (/^gte_(?!cr_)/.test(name)) return "tapeGteInstruction";
if (/^gp[01]_/.test(name)) return "tapeGpuInstruction";
if (/^mac_gte_/.test(name)) return "tapeGteInstruction";
if (/^mac_gp/.test(name)) return "tapeGpuInstruction";
if (/^mac_/.test(name)) return "tapeComponentInstruction";
return null;
}
function modifierMask(modifiers) {
let mask = 0;
for (const modifier of modifiers) {
const index = TOKEN_MODIFIER_INDEX.get(modifier);
if (index !== undefined) mask |= (1 << index);
}
return mask;
}
function isRegUseAccess(tokens, tokenIndex) {
const prev = tokens[tokenIndex - 1];
if (!prev || prev.text !== ".") return false;
const prevPrev = tokens[tokenIndex - 2];
if (!prevPrev || prevPrev.kind !== "identifier") return false;
const next = tokens[tokenIndex + 1];
if (next && next.text === ".") return false;
if (prevPrev.text === "r") return true;
const prev3 = tokens[tokenIndex - 3];
const prev4 = tokens[tokenIndex - 4];
if (prev3 && prev3.text === "." && prev4 && prev4.kind === "identifier" && prev4.text === "r") return true;
return false;
}
function classifyDocument(source, filePath, workspaceIndex, shouldCancel = () => false) {
const scanned = scanSource(source, filePath);
const index = mergeIndexes(workspaceIndex, scanned.index);
const spans = [];
for (let tokenIndex = 0; tokenIndex < scanned.tokens.length; tokenIndex += 1) {
if (shouldCancel()) break;
const token = scanned.tokens[tokenIndex];
if (token.kind !== "identifier") continue;
let type = null;
let modifiers = [];
const declaration = scanned.declarations.get(token.start);
const context = nearestCall(scanned.contexts, tokenIndex);
if (declaration) {
type = ROLE_TO_TYPE[declaration.role] || null;
modifiers = declaration.modifiers.slice();
} else if (ATOM_KEYWORDS.has(token.text)) {
type = "tapeAtomKeyword";
} else if (COMPONENT_KEYWORDS.has(token.text)) {
type = "keyword";
} else if (ANNOTATIONS.has(token.text)) {
type = "tapeAnnotation";
} else if (context && context.callee === "atom_bind" && context.argIndex === 0) {
type = "tapeBindType";
} else if (context && context.callee === "atom_phase" && context.argIndex === 0) {
type = "tapePhase";
modifiers = ["declaration"];
} else if (context && context.callee === "atom_ctx" && context.argIndex === 0) {
type = "tapeAtomName";
} else if (context && context.callee === "atom_label" && context.argIndex === 0) {
type = "tapeLabel";
modifiers = ["declaration"];
} else if (context && context.callee === "atom_offset" && context.argIndex <= 1) {
type = "tapeLabel";
} else if (context && context.callee === "atom_reads") {
type = registerType(token.text, index);
if (type) modifiers = ["tapeRead"];
} else if (context && context.callee === "atom_writes") {
type = registerType(token.text, index);
if (type) modifiers = ["tapeWrite"];
} else if (context && context.callee === "atom_auto_reg") {
if (context.argIndex === 0) type = "tapeAtomName";
if (context.argIndex === 1) {
type = "tapeGprRegister";
modifiers = ["declaration", "tapeAuto"];
}
} else if (context && context.callee === "phase_auto_reg") {
if (context.argIndex === 0) type = "tapePhase";
if (context.argIndex === 1) {
type = "tapeGprRegister";
modifiers = ["declaration", "tapeAuto"];
}
}
if (!type && index.bindTypes.has(token.text)) type = "tapeBindType";
if (!type && DSL_KEYWORDS.has(token.text)) type = "keyword";
if (!type && index.types.has(token.text)) type = "tapeDuffleType";
if (!type && index.attributes.has(token.text)) type = "tapeAttribute";
if (!type) type = registerType(token.text, index);
if (!type && DELAY_SLOT_KEYWORDS.has(token.text)) type = "tapeDelaySlot";
if (!type) {
const domain = index.macros.get(token.text);
if (domain === "control" || (domain && CONTROL_FLOW_PREFIXES.test(token.text))) {
type = "tapeControlFlow";
}
}
if (!type && isRegUseAccess(scanned.tokens, tokenIndex)) type = "tapeGprRegister";
if (!type) type = instructionType(token.text, index);
if (!type && /^(?:Slice_|A[0-9]+_)/.test(token.text)) type = "tapeDuffleType";
if (!type && /_[RV]$/.test(token.text)) type = "tapeDuffleType";
if (!type && index.atoms.has(token.text)) type = "tapeAtomName";
if (!type && index.components.has(token.text)) type = "tapeComponentName";
if (!type && index.phases.has(token.text)) type = "tapePhase";
if (!type && index.labels.has(token.text)) type = "tapeLabel";
if (!type) continue;
spans.push({
text: token.text,
type,
typeIndex: TOKEN_TYPE_INDEX.get(type),
modifiers,
modifierMask: modifierMask(modifiers),
start: token.start,
length: token.end - token.start,
line: token.line,
character: token.character,
});
}
spans.sort((left, right) => left.start - right.start || left.length - right.length);
const nonOverlapping = [];
for (const span of spans) {
const previous = nonOverlapping[nonOverlapping.length - 1];
if (!previous || previous.start + previous.length <= span.start) nonOverlapping.push(span);
}
return { spans: nonOverlapping, errors: scanned.errors };
}
module.exports = {
TOKEN_MODIFIERS,
TOKEN_TYPES,
classifyDocument,
modifierMask,
};
+111
View File
@@ -0,0 +1,111 @@
"use strict";
const vscode = require("vscode");
const { TOKEN_MODIFIERS, TOKEN_TYPES, classifyDocument } = require("./classifier");
const { createIndex, mergeIndexes, scanSource } = require("./source-index");
const SOURCE_GLOB = "**/*.{c,h,cc,cpp,cxx,hh,hpp,hxx}";
const EXCLUDE_GLOB = "**/{gen,build,.slop_cache,toolchain,node_modules}/**";
const EXCLUDED_SEGMENTS = new Set(["gen", "build", ".slop_cache", "toolchain", "node_modules"]);
function isExcluded(uri) {
const segments = uri.fsPath.replaceAll("\\", "/").split("/");
return segments.some((segment) => EXCLUDED_SEGMENTS.has(segment));
}
function formatError(filePath, error) {
return `${filePath}:${error.offset}: ${error.kind}`;
}
async function activate(context) {
const output = vscode.window.createOutputChannel("Tape Atom DSL");
const emitter = new vscode.EventEmitter();
const legend = new vscode.SemanticTokensLegend(TOKEN_TYPES, TOKEN_MODIFIERS);
let workspaceIndex = createIndex();
let rebuildGeneration = 0;
let debounceHandle = null;
async function rebuildIndex() {
const generation = ++rebuildGeneration;
const files = await vscode.workspace.findFiles(SOURCE_GLOB, EXCLUDE_GLOB);
let nextIndex = createIndex();
for (const uri of files) {
if (generation !== rebuildGeneration) return;
if (isExcluded(uri)) continue;
try {
const bytes = await vscode.workspace.fs.readFile(uri);
const source = Buffer.from(bytes).toString("utf8");
const result = scanSource(source, uri.fsPath);
nextIndex = mergeIndexes(nextIndex, result.index);
for (const error of result.errors) output.appendLine(formatError(uri.fsPath, error));
} catch (error) {
output.appendLine(`${uri.fsPath}: ${error.stack || error.message || error}`);
}
}
if (generation !== rebuildGeneration) return;
workspaceIndex = nextIndex;
emitter.fire();
}
function scheduleRebuild(uri) {
if (uri && isExcluded(uri)) return;
if (debounceHandle !== null) clearTimeout(debounceHandle);
debounceHandle = setTimeout(() => {
debounceHandle = null;
rebuildIndex().catch((error) => output.appendLine(error.stack || String(error)));
}, 100);
}
const provider = {
onDidChangeSemanticTokens: emitter.event,
provideDocumentSemanticTokens(document, cancellationToken) {
try {
const result = classifyDocument(
document.getText(),
document.uri.fsPath,
workspaceIndex,
() => cancellationToken.isCancellationRequested
);
const builder = new vscode.SemanticTokensBuilder(legend);
for (const span of result.spans) {
if (cancellationToken.isCancellationRequested) break;
builder.push(span.line, span.character, span.length, span.typeIndex, span.modifierMask);
}
for (const error of result.errors) {
output.appendLine(formatError(document.uri.fsPath || document.uri.toString(), error));
}
return builder.build();
} catch (error) {
output.appendLine(`${document.uri}: ${error.stack || error.message || error}`);
return new vscode.SemanticTokensBuilder(legend).build();
}
},
};
const selector = [
{ language: "c", scheme: "file" },
{ language: "c", scheme: "untitled" },
{ language: "cpp", scheme: "file" },
{ language: "cpp", scheme: "untitled" },
];
const watcher = vscode.workspace.createFileSystemWatcher(SOURCE_GLOB);
context.subscriptions.push(
output,
emitter,
watcher,
watcher.onDidCreate(scheduleRebuild),
watcher.onDidChange(scheduleRebuild),
watcher.onDidDelete(scheduleRebuild),
vscode.languages.registerDocumentSemanticTokensProvider(selector, provider, legend),
{ dispose() { if (debounceHandle !== null) clearTimeout(debounceHandle); } }
);
await rebuildIndex();
}
function deactivate() {}
module.exports = { activate, deactivate };
+186
View File
@@ -0,0 +1,186 @@
"use strict";
function isIdentifierStart(code) {
return code === 95 ||
(code >= 65 && code <= 90) ||
(code >= 97 && code <= 122);
}
function isIdentifierContinue(code) {
return isIdentifierStart(code) || (code >= 48 && code <= 57);
}
function lex(source) {
if (typeof source !== "string") throw new TypeError("source must be a string");
const tokens = [];
const errors = [];
let offset = 0;
let line = 0;
let character = 0;
function advance() {
if (source[offset] === "\r" && source[offset + 1] === "\n") {
offset += 2;
line += 1;
character = 0;
return;
}
if (source[offset] === "\n") {
offset += 1;
line += 1;
character = 0;
return;
}
offset += 1;
character += 1;
}
function pushToken(kind, start, startLine, startCharacter) {
tokens.push({
kind,
text: source.slice(start, offset),
start,
end: offset,
line: startLine,
character: startCharacter,
});
}
while (offset < source.length) {
const ch = source[offset];
if (/\s/.test(ch)) {
advance();
continue;
}
if (ch === "/" && source[offset + 1] === "/") {
while (offset < source.length && source[offset] !== "\r" && source[offset] !== "\n") advance();
continue;
}
if (ch === "/" && source[offset + 1] === "*") {
const start = offset;
advance();
advance();
let closed = false;
while (offset < source.length) {
if (source[offset] === "*" && source[offset + 1] === "/") {
advance();
advance();
closed = true;
break;
}
advance();
}
if (!closed) errors.push({ kind: "unterminated-block-comment", offset: start });
continue;
}
if (ch === "\"" || ch === "'") {
const quote = ch;
const start = offset;
advance();
let closed = false;
while (offset < source.length) {
if (source[offset] === "\\") {
advance();
if (offset < source.length) advance();
continue;
}
if (source[offset] === quote) {
advance();
closed = true;
break;
}
if (source[offset] === "\n" || source[offset] === "\r") break;
advance();
}
if (!closed) errors.push({ kind: "unterminated-literal", offset: start });
continue;
}
const code = source.charCodeAt(offset);
if (isIdentifierStart(code)) {
const start = offset;
const startLine = line;
const startCharacter = character;
advance();
while (offset < source.length && isIdentifierContinue(source.charCodeAt(offset))) advance();
pushToken("identifier", start, startLine, startCharacter);
continue;
}
const start = offset;
const startLine = line;
const startCharacter = character;
advance();
pushToken("punctuation", start, startLine, startCharacter);
}
return { tokens, errors };
}
function buildCallContexts(tokens) {
const contexts = Array.from({ length: tokens.length }, () => []);
const calls = [];
const errors = [];
const stack = [];
for (let tokenIndex = 0; tokenIndex < tokens.length; tokenIndex += 1) {
const token = tokens[tokenIndex];
if (token.text === ")") {
if (stack.length === 0) {
errors.push({ kind: "unmatched-close-paren", offset: token.start });
} else {
const frame = stack.pop();
if (frame.callee !== null) calls.push({ ...frame, closeTokenIndex: tokenIndex });
}
}
contexts[tokenIndex] = stack
.filter((frame) => frame.callee !== null)
.map((frame) => ({
callee: frame.callee,
calleeTokenIndex: frame.calleeTokenIndex,
openTokenIndex: frame.openTokenIndex,
argIndex: frame.argIndex,
}));
if (token.text === "(") {
const previous = tokens[tokenIndex - 1];
const hasCallee = previous && previous.kind === "identifier";
stack.push({
callee: hasCallee ? previous.text : null,
calleeTokenIndex: hasCallee ? tokenIndex - 1 : -1,
openTokenIndex: tokenIndex,
argIndex: 0,
});
continue;
}
if (token.text === "," && stack.length > 0) {
const frame = stack[stack.length - 1];
if (frame.callee !== null) frame.argIndex += 1;
}
}
for (const frame of stack) {
errors.push({ kind: "unmatched-open-paren", offset: tokens[frame.openTokenIndex].start });
}
return { contexts, calls, errors };
}
function nearestCall(contexts, tokenIndex, callee) {
const entries = contexts[tokenIndex] || [];
for (let contextIndex = entries.length - 1; contextIndex >= 0; contextIndex -= 1) {
const entry = entries[contextIndex];
if (callee === undefined || entry.callee === callee) return entry;
}
return null;
}
module.exports = { buildCallContexts, lex, nearestCall };
+85
View File
@@ -0,0 +1,85 @@
{
"name": "atomasm-psx",
"displayName": "AtomAsm-PSX",
"description": "Semantic highlighting for the PS1 Tape/Atom MIPS macro DSL",
"publisher": "local",
"version": "0.3.0",
"engines": { "vscode": "^1.80.0" },
"categories": ["Programming Languages"],
"activationEvents": ["onLanguage:c", "onLanguage:cpp"],
"main": "./extension.js",
"files": [
"classifier.js",
"extension.js",
"lexer.js",
"source-index.js",
"syntaxes/tape_atom.tmLanguage.json"
],
"scripts": {
"test": "node --test test/*.test.js",
"package": "npx --yes @vscode/vsce@3.6.1 package --allow-missing-repository --skip-license --out atomasm-psx-0.3.0.vsix"
},
"contributes": {
"semanticTokenTypes": [
{ "id": "tapeAtomKeyword", "superType": "keyword", "description": "Tape atom declaration keyword" },
{ "id": "tapeAtomName", "superType": "function", "description": "Tape atom name" },
{ "id": "tapeComponentKeyword", "superType": "keyword", "description": "Tape atom component declaration keyword" },
{ "id": "tapeComponentName", "superType": "function", "description": "Tape atom component name" },
{ "id": "tapeAnnotation", "superType": "macro", "description": "Tape atom annotation" },
{ "id": "tapeBindType", "superType": "type", "description": "Tape bind structure type" },
{ "id": "tapePhase", "superType": "label", "description": "Tape atom phase" },
{ "id": "tapeLabel", "superType": "label", "description": "Tape atom branch label" },
{ "id": "tapeCpuInstruction", "superType": "macro", "description": "MIPS CPU instruction emitter" },
{ "id": "tapeControlFlow", "superType": "keyword", "description": "MIPS branch or jump instruction" },
{ "id": "tapeGteInstruction", "superType": "macro", "description": "GTE instruction emitter" },
{ "id": "tapeGpuInstruction", "superType": "macro", "description": "GPU command emitter" },
{ "id": "tapeComponentInstruction", "superType": "macro", "description": "Tape atom component invocation" },
{ "id": "tapeDelaySlot", "superType": "keyword", "description": "Load or branch delay slot annotation" },
{ "id": "tapeGprRegister", "superType": "variable", "description": "MIPS GPR alias" },
{ "id": "tapeCop2Register", "superType": "variable", "description": "COP2 data or control register alias" },
{ "id": "tapeDuffleType", "superType": "type", "description": "Duffle type or type constructor" },
{ "id": "tapeAttribute", "superType": "keyword", "description": "Duffle linkage or storage attribute" },
{ "id": "keyword", "description": "Standard keyword (DSL built-in macros)" },
{ "id": "macro", "description": "Standard macro (utility #define with no instruction domain)" }
],
"semanticTokenModifiers": [
{ "id": "tapeRead", "description": "Register declared in atom_reads" },
{ "id": "tapeWrite", "description": "Register declared in atom_writes" },
{ "id": "tapeAuto", "description": "Auto-allocated register" }
],
"semanticTokenScopes": [
{
"language": "c",
"scopes": {
"tapeAtomKeyword": ["keyword.control.duffle.atom"],
"tapeAtomName": ["entity.name.function.duffle.atom"],
"tapeComponentKeyword": ["keyword.control.duffle.component"],
"tapeComponentName": ["entity.name.function.duffle.component"],
"tapeAnnotation": ["support.function.duffle.annotation"],
"tapeBindType": ["entity.name.type.duffle.bind"],
"tapePhase": ["entity.name.tag.duffle.phase"],
"tapeLabel": ["entity.name.label.duffle.atom"],
"tapeCpuInstruction": ["support.function.duffle.cpu"],
"tapeControlFlow": ["keyword.control.duffle.branch"],
"tapeGteInstruction": ["support.function.duffle.gte"],
"tapeGpuInstruction": ["support.function.duffle.gpu"],
"tapeComponentInstruction": ["support.function.duffle.component"],
"tapeDelaySlot": ["keyword.operator.duffle.delayslot"],
"tapeGprRegister": ["variable.other.constant.duffle.gpr"],
"tapeCop2Register": ["variable.other.constant.duffle.cop2"],
"tapeDuffleType": ["storage.type.duffle.type"],
"tapeAttribute": ["storage.modifier.duffle.attr"],
"keyword": ["keyword"],
"macro": ["entity.name.function.preprocessor"]
}
}
],
"grammars": [
{
"scopeName": "tape_atom.injection",
"path": "./syntaxes/tape_atom.tmLanguage.json",
"injectTo": ["source.c", "source.cpp"]
}
]
}
}
+341
View File
@@ -0,0 +1,341 @@
"use strict";
const path = require("node:path");
const { buildCallContexts, lex, nearestCall } = require("./lexer");
const BASE_TYPES = [
"B1", "B2", "B4", "B8", "F4", "F8", "S1", "S2", "S4", "S8",
"U1", "U2", "U4", "U8", "MipsAtom", "MipsCode", "Reg",
];
const C_BUILTINS = new Set([
"void", "type", "char", "short", "int", "long", "float", "double",
"unsigned", "signed", "bool", "size_t", "uint8_t", "uint16_t", "uint32_t",
"int8_t", "int16_t", "int32_t",
]);
const BASE_ATTRIBUTES = [
"FI_", "I_", "NI_", "Relative_", "Struct_", "Enum_", "Union_", "Array_",
"Slice_", "TypeR_", "TypeV_", "align_", "internal", "local_persist", "global",
"RO_", "LP_", "gknown", "expect_", "cexpr_",
"asm", "asm_words", "asm_rpins", "asm_clobber",
"O_", "S_", "C_", "T_", "tmpl", "glue", "r_", "v_", "tr_", "tv_",
"rgcc", "r_use", "r_set", "r_mod", "r_imm", "r_mem",
"u1_", "u2_", "u4_", "u8_", "s1_", "s2_", "s4_", "s8_",
"u1_r", "u2_r", "u4_r", "u8_r", "u1_v", "u2_v", "u4_v", "u8_v",
];
function createIndex() {
return {
atoms: new Set(),
components: new Set(),
componentAliases: new Set(),
macros: new Map(),
registers: new Map(),
bindTypes: new Set(),
types: new Set(BASE_TYPES),
phases: new Set(),
labels: new Set(),
attributes: new Set(BASE_ATTRIBUTES),
componentCallees: new Map(),
};
}
function cloneIndex(source) {
const result = createIndex();
for (const key of ["atoms", "components", "componentAliases", "bindTypes", "types", "phases", "labels", "attributes"]) {
for (const value of source[key]) result[key].add(value);
}
for (const [name, domain] of source.macros) result.macros.set(name, domain);
for (const [name, domain] of source.registers) result.registers.set(name, domain);
for (const [name, callees] of source.componentCallees) result.componentCallees.set(name, callees.slice());
return result;
}
function mergeIndexes(...sources) {
const result = createIndex();
for (const source of sources) {
if (!source) continue;
for (const key of ["atoms", "components", "componentAliases", "bindTypes", "types", "phases", "labels", "attributes"]) {
for (const value of source[key]) result[key].add(value);
}
for (const [name, domain] of source.macros) {
const existing = result.macros.get(name);
if (!existing || domainRank(domain) >= domainRank(existing)) result.macros.set(name, domain);
}
for (const [name, domain] of source.registers) result.registers.set(name, domain);
for (const [name, callees] of source.componentCallees) {
const existing = result.componentCallees.get(name) || [];
result.componentCallees.set(name, existing.concat(callees));
}
}
return resolveComponentDomains(result);
}
function domainFromPath(filePath) {
const base = path.basename(filePath.replaceAll("\\", "/")).toLowerCase();
if (base === "mips.h") return "cpu";
if (base === "gte.h") return "gte";
if (base === "gp.h") return "gpu";
return null;
}
function prefixDomain(name) {
if (/^(?:branch_|jump_|call_)/.test(name)) return "control";
if (/^gte_(?!cr_)/.test(name) || name.startsWith("mac_gte_") || name.startsWith("ac_gte_")) return "gte";
if (/^gp[01]_/.test(name) || name.startsWith("mac_gp_") || name.startsWith("ac_gp_")) return "gpu";
return null;
}
function collectBraceIdentifiers(tokens, openBraceIndex) {
const names = [];
let depth = 0;
for (let tokenIndex = openBraceIndex; tokenIndex < tokens.length; tokenIndex += 1) {
if (tokens[tokenIndex].text === "{") depth += 1;
if (tokens[tokenIndex].text === "}") {
depth -= 1;
if (depth === 0) break;
}
if (tokens[tokenIndex].kind === "identifier") names.push(tokens[tokenIndex].text);
}
return names;
}
function resolveComponentDomains(index) {
const hardwareRank = { cpu: 1, gpu: 2, gte: 3, control: 4 };
let changed = true;
while (changed) {
changed = false;
for (const [alias, callees] of index.componentCallees) {
let best = index.macros.get(alias) || "component";
let bestRank = hardwareRank[best] || 0;
for (const callee of callees) {
const domain = prefixDomain(callee) || index.macros.get(callee);
const rank = hardwareRank[domain] || 0;
if (rank > bestRank) {
best = domain;
bestRank = rank;
}
}
if (bestRank > 0 && index.macros.get(alias) !== best) {
index.macros.set(alias, best);
changed = true;
}
}
}
return index;
}
function domainRank(domain) {
if (domain === "control") return 4;
if (domain === "cpu" || domain === "gte" || domain === "gpu") return 3;
if (domain === "component") return 2;
return 1;
}
function registerKind(name) {
if (/^R_[A-Za-z0-9_]+$/.test(name)) return "gpr";
if (/^(?:C2_|gte_cr_)[A-Za-z0-9_]+$/.test(name)) return "cop2";
return null;
}
function componentAlias(name) {
return name.startsWith("ac_") ? `mac_${name.slice(3)}` : null;
}
function findFunctionNameBefore(tokens, calleeTokenIndex) {
let closeIndex = calleeTokenIndex - 1;
while (closeIndex >= 0 && tokens[closeIndex].kind === "identifier" && tokens[closeIndex].text === "atom_dbg_skip") {
closeIndex -= 1;
}
if (!tokens[closeIndex] || tokens[closeIndex].text !== ")") return null;
let depth = 1;
for (let tokenIndex = closeIndex - 1; tokenIndex >= 0; tokenIndex -= 1) {
if (tokens[tokenIndex].text === ")") depth += 1;
if (tokens[tokenIndex].text === "(") depth -= 1;
if (depth !== 0) continue;
const name = tokens[tokenIndex - 1];
return name && name.kind === "identifier" ? name : null;
}
return null;
}
function scanSource(source, filePath) {
const lexical = lex(source);
const balanced = buildCallContexts(lexical.tokens);
const tokens = lexical.tokens;
const contexts = balanced.contexts;
const index = createIndex();
const declarations = new Map();
const domain = domainFromPath(filePath);
function mark(token, role, modifiers = ["declaration"]) {
declarations.set(token.start, { role, modifiers });
}
function addComponent(token) {
index.components.add(token.text);
mark(token, "componentName");
const alias = componentAlias(token.text);
if (alias) {
index.componentAliases.add(alias);
index.macros.set(alias, prefixDomain(alias) || prefixDomain(token.text) || "component");
}
}
function bindComponentCallees(alias, callees) {
if (!alias) return;
index.componentAliases.add(alias);
index.componentCallees.set(alias, callees);
if (!index.macros.has(alias)) index.macros.set(alias, "component");
}
for (let tokenIndex = 0; tokenIndex < tokens.length; tokenIndex += 1) {
const token = tokens[tokenIndex];
if (token.kind !== "identifier") continue;
const kind = registerKind(token.text);
if (kind) {
index.registers.set(token.text, kind);
if (tokens[tokenIndex + 1] && tokens[tokenIndex + 1].text === "=") {
mark(token, kind === "gpr" ? "gprRegister" : "cop2Register");
}
}
const context = nearestCall(contexts, tokenIndex);
if (context && context.argIndex === 0) {
if (context.callee === "MipsAtom_") {
index.atoms.add(token.text);
mark(token, "atomName");
}
if (context.callee === "MipsAtomComp_") addComponent(token);
if (context.callee === "atom_bind") index.bindTypes.add(token.text);
if (context.callee === "atom_phase" || context.callee === "phase_auto_reg") index.phases.add(token.text);
if (context.callee === "atom_label" || context.callee === "atom_offset") index.labels.add(token.text);
}
const isWrappedType = context && (
((context.callee === "Struct_" || context.callee === "Union_") && context.argIndex === 0) ||
(context.callee === "Enum_" && context.argIndex === 1)
);
if (isWrappedType) {
index.types.add(token.text);
mark(token, token.text.startsWith("Binds_") ? "bindType" : "duffleType");
if (token.text.startsWith("Binds_")) index.bindTypes.add(token.text);
}
if (context && context.callee === "atom_offset" && context.argIndex === 1) index.labels.add(token.text);
if (context && context.callee === "atom_auto_reg") {
if (context.argIndex === 0) index.atoms.add(token.text);
if (context.argIndex === 1) {
index.registers.set(token.text, "gpr");
mark(token, "gprRegister", ["declaration", "tapeAuto"]);
}
}
if (context && context.callee === "phase_auto_reg" && context.argIndex === 1) {
index.registers.set(token.text, "gpr");
mark(token, "gprRegister", ["declaration", "tapeAuto"]);
}
if (token.text === "define" && tokens[tokenIndex - 1] && tokens[tokenIndex - 1].text === "#") {
const name = tokens[tokenIndex + 1];
if (name && name.kind === "identifier" && name.line === token.line) {
if (/^(?:RegUse_|Struct_|Enum_|Union_|TypeR_|TypeV_|Relative_|Binds_)/.test(name.text)) {
index.types.add(name.text);
} else if (/^(?:ac_|mac_)/.test(name.text)) {
const alias = name.text.startsWith("ac_") ? componentAlias(name.text) : name.text;
const rest = [];
for (let restIndex = tokenIndex + 2; restIndex < tokens.length && tokens[restIndex].line === name.line; restIndex += 1) {
if (tokens[restIndex].kind === "identifier") rest.push(tokens[restIndex].text);
}
if (alias) {
index.componentAliases.add(alias);
index.macros.set(alias, prefixDomain(alias) || "component");
if (rest.length) index.componentCallees.set(alias, rest);
}
} else {
index.macros.set(name.text, domain || "utility");
}
}
}
if (token.text === "typedef") {
let endIndex = tokenIndex + 1;
let hasBrace = false;
let lastIdentifier = null;
while (endIndex < tokens.length && tokens[endIndex].text !== ";") {
if (tokens[endIndex].text === "{") hasBrace = true;
if (tokens[endIndex].kind === "identifier" && !C_BUILTINS.has(tokens[endIndex].text)) lastIdentifier = tokens[endIndex];
endIndex += 1;
}
if (!hasBrace && lastIdentifier && !C_BUILTINS.has(lastIdentifier.text)) {
index.types.add(lastIdentifier.text);
mark(lastIdentifier, "duffleType");
}
}
if (token.text === "MipsAtom_Proc_") {
const functionName = findFunctionNameBefore(tokens, tokenIndex);
if (functionName) {
const atomName = functionName.text.endsWith("_proc")
? functionName.text.slice(0, -5)
: functionName.text;
index.atoms.add(atomName);
index.atoms.add(functionName.text);
mark(functionName, "atomName");
}
}
if (token.text === "MipsAtomComp_Proc_") {
const functionName = findFunctionNameBefore(tokens, tokenIndex);
if (functionName) addComponent(functionName);
}
}
for (const call of balanced.calls) {
if (call.callee === "MipsAtomComp_") {
const name = tokens[call.openTokenIndex + 1];
const brace = tokens[call.closeTokenIndex + 1];
if (name && name.kind === "identifier" && brace && brace.text === "{") {
bindComponentCallees(componentAlias(name.text), collectBraceIdentifiers(tokens, call.closeTokenIndex + 1));
}
}
if (call.callee === "MipsAtomComp_Proc_") {
const functionName = findFunctionNameBefore(tokens, call.calleeTokenIndex);
let braceIndex = -1;
for (let tokenIndex = call.openTokenIndex + 1; tokenIndex < call.closeTokenIndex; tokenIndex += 1) {
if (tokens[tokenIndex].text === "{") {
braceIndex = tokenIndex;
break;
}
}
if (functionName && braceIndex >= 0) {
bindComponentCallees(componentAlias(functionName.text), collectBraceIdentifiers(tokens, braceIndex));
}
}
if (!domain) continue;
const name = tokens[call.calleeTokenIndex];
const after = tokens[call.closeTokenIndex + 1];
if (!name || !after || after.text !== "{") continue;
if (/^(?:gp0_|gp1_|gte_|mac_)/.test(name.text)) index.macros.set(name.text, domain);
}
return {
index: resolveComponentDomains(cloneIndex(index)),
declarations,
tokens,
contexts,
errors: [...lexical.errors, ...balanced.errors],
};
}
module.exports = {
createIndex,
domainFromPath,
mergeIndexes,
resolveComponentDomains,
scanSource,
};
@@ -0,0 +1,71 @@
{
"scopeName": "tape_atom.injection",
"injectionSelector": "L:source.c -comment -string, L:source.cpp -comment -string",
"patterns": [
{ "include": "#atom-declarations" },
{ "include": "#component-declarations" },
{ "include": "#annotation-arguments" },
{ "include": "#annotations" },
{ "include": "#delay-slots" },
{ "include": "#types" },
{ "include": "#attributes" }
],
"repository": {
"atom-declarations": {
"patterns": [
{
"match": "\\b(MipsAtom_)\\s*\\(\\s*([A-Za-z_][A-Za-z0-9_]*)",
"captures": {
"1": { "name": "keyword.control.duffle.atom" },
"2": { "name": "entity.name.function.duffle.atom" }
}
},
{ "match": "\\bMipsAtom_Proc_\\b", "name": "keyword.control.duffle.atom" },
{ "match": "\\b[A-Za-z_][A-Za-z0-9_]*_proc\\b", "name": "entity.name.function.duffle.atom" }
]
},
"component-declarations": {
"patterns": [
{
"match": "\\b(MipsAtomComp_)\\s*\\(\\s*(ac_[A-Za-z0-9_]*)",
"captures": {
"1": { "name": "keyword" },
"2": { "name": "entity.name.function.duffle.component" }
}
},
{ "match": "\\bMipsAtomComp_Proc_\\b", "name": "keyword" }
]
},
"annotation-arguments": {
"patterns": [
{
"match": "\\b(atom_offset)\\s*\\(\\s*([A-Za-z_][A-Za-z0-9_]*)\\s*,\\s*([A-Za-z_][A-Za-z0-9_]*)",
"captures": {
"1": { "name": "support.function.duffle.annotation" },
"2": { "name": "entity.name.label.duffle.atom" },
"3": { "name": "entity.name.label.duffle.atom" }
}
},
{ "match": "(?<=\\batom_bind\\()\\s*Binds_[A-Za-z0-9_]+", "name": "entity.name.type.duffle.bind" },
{ "match": "(?<=\\batom_phase\\()\\s*[A-Za-z_][A-Za-z0-9_]*", "name": "entity.name.tag.duffle.phase" },
{ "match": "(?<=\\batom_label\\()\\s*[A-Za-z_][A-Za-z0-9_]*", "name": "entity.name.label.duffle.atom" }
]
},
"annotations": {
"match": "\\b(atom_info|atom_bind|atom_reads|atom_writes|atom_label|atom_offset|atom_reg|atom_type|atom_ctx|atom_phase|atom_auto_reg|phase_auto_reg|atom_dbg_skip)\\b",
"name": "support.function.duffle.annotation"
},
"delay-slots": {
"match": "\\b(LdSlot_|BdSlot_|DmaSlot_|GteDelay_)\\b",
"name": "keyword.operator.duffle.delayslot"
},
"types": {
"match": "\\b(?:Binds_[A-Za-z0-9_]+|RegUse_[A-Za-z0-9_]+)\\b",
"name": "storage.type.duffle.type"
},
"attributes": {
"match": "\\b(?:FI_|I_|NI_|Relative_|Struct_|Enum_|Union_|Array_|Slice_|TypeR_|TypeV_|align_|internal|local_persist|global|RO_|LP_|gknown|expect_|cexpr_|asm|asm_words|asm_rpins|asm_clobber|O_|S_|C_|T_|tmpl|glue|r_|v_|tr_|tv_|rgcc|r_use|r_set|r_mod|r_imm|r_mem|u[1248]_|u[1248]_r|u[1248]_v|s[1248]_)\\b",
"name": "keyword"
}
}
}
Binary file not shown.
+128
View File
@@ -0,0 +1,128 @@
"use strict";
const assert = require("node:assert/strict");
const test = require("node:test");
const { classifyDocument } = require("../classifier");
const { createIndex } = require("../source-index");
function byText(result, text) {
return result.spans.filter((span) => span.text === text);
}
test("classifyDocument distinguishes declaration, annotation, phase, bind, and label roles", () => {
const source = [
"typedef Struct_(Binds_CubeTri) { U4 PrimCursor; };",
"MipsAtom_(cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4),",
"\tatom_reads(R_PrimCursor), atom_writes(R_FaceCursor)) {",
"\tbranch_le_zero(R_T0, atom_offset(cull, exit)),",
"\tatom_label(exit)",
"};",
].join("\n");
const result = classifyDocument(source, "C:/x/code/hello_camera/hello_camera.atom.c", createIndex());
assert.equal(byText(result, "MipsAtom_")[0].type, "tapeAtomKeyword");
assert.deepEqual(byText(result, "cube_g4_face")[0].modifiers, ["declaration"]);
assert.equal(byText(result, "atom_bind")[0].type, "tapeAnnotation");
assert.equal(byText(result, "Binds_CubeTri").at(-1).type, "tapeBindType");
assert.equal(byText(result, "cube_g4")[0].type, "tapePhase");
assert.deepEqual(byText(result, "cube_g4")[0].modifiers, ["declaration"]);
assert.equal(byText(result, "cull")[0].type, "tapeLabel");
assert.equal(byText(result, "exit").every((span) => span.type === "tapeLabel"), true);
});
test("classifyDocument applies read and write modifiers to GPRs", () => {
const source = "atom_info(atom_reads(R_PrimCursor), atom_writes(R_FaceCursor))";
const result = classifyDocument(source, "C:/x/code/test.atom.c", createIndex());
assert.deepEqual(byText(result, "R_PrimCursor")[0].modifiers, ["tapeRead"]);
assert.deepEqual(byText(result, "R_FaceCursor")[0].modifiers, ["tapeWrite"]);
});
test("classifyDocument separates CPU, GTE, GPU, and component domains", () => {
const workspace = createIndex();
workspace.macros.set("load_word", "cpu");
workspace.macros.set("gte_cmdw_rtpt", "gte");
workspace.macros.set("gp1_word_DisplayOn", "gpu");
workspace.macros.set("mac_yield", "control");
workspace.componentAliases.add("mac_yield");
const source = "load_word(R_T0, R_T1, 0), gte_cmdw_rtpt, gp1_word_DisplayOn(), mac_yield(), C2_MAC0, gte_cr_OFX_Code";
const result = classifyDocument(source, "C:/x/code/test.c", workspace);
assert.equal(byText(result, "load_word")[0].type, "tapeCpuInstruction");
assert.equal(byText(result, "gte_cmdw_rtpt")[0].type, "tapeGteInstruction");
assert.equal(byText(result, "gp1_word_DisplayOn")[0].type, "tapeGpuInstruction");
assert.equal(byText(result, "mac_yield")[0].type, "tapeControlFlow");
assert.equal(byText(result, "C2_MAC0")[0].type, "tapeCop2Register");
assert.equal(byText(result, "gte_cr_OFX_Code")[0].type, "tapeCop2Register");
});
test("component invocations keep the domain resolved from their emitted instructions", () => {
const workspace = createIndex();
workspace.macros.set("mac_load_word_imm", "cpu");
workspace.macros.set("mac_gcmd_push", "gpu");
workspace.macros.set("mac_gte_store_f3", "gte");
workspace.macros.set("mac_load_v3s4", "cpu");
const source = "mac_load_word_imm(dst, imm), mac_gcmd_push(cmd), mac_gte_store_f3(cursor), mac_load_v3s4()";
const result = classifyDocument(source, "C:/x/code/hello_camera/hello_camera.atom.c", workspace);
assert.equal(byText(result, "mac_load_word_imm")[0].type, "tapeCpuInstruction");
assert.equal(byText(result, "mac_gcmd_push")[0].type, "tapeGpuInstruction");
assert.equal(byText(result, "mac_gte_store_f3")[0].type, "tapeGteInstruction");
assert.equal(byText(result, "mac_load_v3s4")[0].type, "tapeCpuInstruction");
});
test("utility macros without a hardware domain use the standard macro token", () => {
const workspace = createIndex();
workspace.macros.set("load_word", "cpu");
workspace.macros.set("assert", "utility");
workspace.macros.set("stringify", "utility");
workspace.macros.set("u4_hi", "utility");
const source = "load_word(R_T0, R_T1, 0), assert(ok), stringify(name), u4_hi(imm)";
const result = classifyDocument(source, "C:/x/code/hello_camera/hello_camera.c", workspace);
assert.equal(byText(result, "load_word")[0].type, "tapeCpuInstruction");
assert.equal(byText(result, "assert")[0].type, "macro");
assert.equal(byText(result, "stringify")[0].type, "macro");
assert.equal(byText(result, "u4_hi")[0].type, "macro");
});
test("document-local declarations override an empty workspace index", () => {
const source = [
"MipsAtomComp_(ac_new_component) { nop };",
"MipsAtomComp_Proc_(ab, { nop })",
"mac_new_component(),",
].join("\n");
const result = classifyDocument(source, "C:/x/code/duffle/math.atom.c", createIndex());
assert.equal(byText(result, "MipsAtomComp_")[0].type, "keyword");
assert.equal(byText(result, "MipsAtomComp_Proc_")[0].type, "keyword");
assert.equal(byText(result, "ac_new_component")[0].type, "tapeComponentName");
assert.equal(byText(result, "mac_new_component")[0].type, "tapeComponentInstruction");
});
test("delay slot markers share the tapeDelaySlot token", () => {
const source = "LdSlot_ nop, BdSlot_ nop, DmaSlot_ nop2, GteDelay_ nop";
const result = classifyDocument(source, "C:/x/code/duffle/gte.atom.c", createIndex());
assert.equal(byText(result, "LdSlot_")[0].type, "tapeDelaySlot");
assert.equal(byText(result, "BdSlot_")[0].type, "tapeDelaySlot");
assert.equal(byText(result, "DmaSlot_")[0].type, "tapeDelaySlot");
assert.equal(byText(result, "GteDelay_")[0].type, "tapeDelaySlot");
});
test("classifier returns ordered non-overlapping spans and partial malformed output", () => {
const source = "atom_reads(R_A /* broken";
const result = classifyDocument(source, "C:/x/code/test.atom.c", createIndex());
assert.equal(result.errors.some((error) => error.kind === "unterminated-block-comment"), true);
for (let spanIndex = 1; spanIndex < result.spans.length; spanIndex += 1) {
const previous = result.spans[spanIndex - 1];
const current = result.spans[spanIndex];
assert.equal(previous.start + previous.length <= current.start, true);
}
});
+88
View File
@@ -0,0 +1,88 @@
"use strict";
const assert = require("node:assert/strict");
const fs = require("node:fs");
const path = require("node:path");
const test = require("node:test");
const { TOKEN_MODIFIERS, TOKEN_TYPES } = require("../classifier");
const ROOT = path.resolve(__dirname, "..");
function readJson(filePath) {
const raw = fs.readFileSync(filePath, "utf8");
const stripped = raw.replace(/\/\/.*$/gm, "").replace(/,\s*([}\]])/g, "$1");
return JSON.parse(stripped);
}
function collectScopeNames(value, output = new Set()) {
if (Array.isArray(value)) {
for (const entry of value) collectScopeNames(entry, output);
return output;
}
if (!value || typeof value !== "object") return output;
if (typeof value.name === "string") output.add(value.name);
for (const child of Object.values(value)) collectScopeNames(child, output);
return output;
}
test("package semantic legend matches classifier exports", () => {
const packageJson = readJson(path.join(ROOT, "package.json"));
const contributedTypes = packageJson.contributes.semanticTokenTypes.map((entry) => entry.id);
const contributedModifiers = packageJson.contributes.semanticTokenModifiers.map((entry) => entry.id);
assert.equal(packageJson.version, "0.3.0");
assert.deepEqual(contributedTypes, TOKEN_TYPES);
assert.deepEqual(contributedModifiers, TOKEN_MODIFIERS.filter((name) => name !== "declaration"));
});
test("package includes runtime files only and acknowledges local-only metadata", () => {
const packageJson = readJson(path.join(ROOT, "package.json"));
assert.deepEqual(packageJson.files, [
"classifier.js",
"extension.js",
"lexer.js",
"source-index.js",
"syntaxes/tape_atom.tmLanguage.json",
]);
assert.equal(packageJson.scripts.package.includes("--allow-missing-repository"), true);
assert.equal(packageJson.scripts.package.includes("--skip-license"), true);
});
test("every semantic token has a scope mapping; DSL-specific tokens also have grammar scopes", () => {
const packageJson = readJson(path.join(ROOT, "package.json"));
const grammar = readJson(path.join(ROOT, "syntaxes", "tape_atom.tmLanguage.json"));
const mappings = packageJson.contributes.semanticTokenScopes[0].scopes;
const grammarScopes = collectScopeNames(grammar);
const grammarRequired = new Set([
"tapeAtomKeyword", "tapeAtomName", "tapeComponentName",
"tapeAnnotation", "tapeBindType", "tapePhase", "tapeLabel",
"tapeDelaySlot", "tapeDuffleType", "keyword",
]);
for (const tokenType of TOKEN_TYPES) {
assert.equal(Array.isArray(mappings[tokenType]), true, `missing scope mapping: ${tokenType}`);
if (grammarRequired.has(tokenType)) {
assert.equal(mappings[tokenType].some((scope) => grammarScopes.has(scope)), true, `grammar does not emit: ${tokenType}`);
}
}
});
test("TextMate offset labels stay scoped to atom_offset calls", () => {
const grammar = readJson(path.join(ROOT, "syntaxes", "tape_atom.tmLanguage.json"));
const serialized = JSON.stringify(grammar);
const offsetRule = grammar.repository["annotation-arguments"].patterns
.find((rule) => rule.match.includes("atom_offset"));
assert.equal(serialized.includes("(?<=,)"), false);
assert.equal(offsetRule.captures[1].name, "support.function.duffle.annotation");
assert.equal(offsetRule.captures[2].name, "entity.name.label.duffle.atom");
assert.equal(offsetRule.captures[3].name, "entity.name.label.duffle.atom");
});
test("workspace enables semantic highlighting", () => {
const settings = readJson(path.resolve(ROOT, "..", "settings.json"));
assert.equal(settings["editor.semanticHighlighting.enabled"], true);
});
+77
View File
@@ -0,0 +1,77 @@
"use strict";
const assert = require("node:assert/strict");
const test = require("node:test");
const { buildCallContexts, lex, nearestCall } = require("../lexer");
test("lex skips comments, strings, and character literals", () => {
const source = [
"MipsAtom_(visible)",
"// MipsAtom_(line_comment)",
"const char *s = \"atom_reads(R_Hidden)\";",
"char c = '\\''; /* gte_cmdw_hidden */",
"atom_reads(R_Visible)",
].join("\n");
const result = lex(source);
const identifiers = result.tokens
.filter((token) => token.kind === "identifier")
.map((token) => token.text);
assert.deepEqual(result.errors, []);
assert.equal(identifiers.includes("visible"), true);
assert.equal(identifiers.includes("R_Visible"), true);
assert.equal(identifiers.includes("line_comment"), false);
assert.equal(identifiers.includes("R_Hidden"), false);
assert.equal(identifiers.includes("gte_cmdw_hidden"), false);
});
test("lex reports unterminated block comments without returning comment tokens", () => {
const result = lex("R_Visible /* atom_reads(R_Hidden)");
assert.equal(result.tokens.some((token) => token.text === "R_Visible"), true);
assert.equal(result.tokens.some((token) => token.text === "R_Hidden"), false);
assert.deepEqual(result.errors.map((error) => error.kind), ["unterminated-block-comment"]);
});
test("line comments stop at CRLF boundaries", () => {
const result = lex("// atom_reads(R_Hidden)\r\natom_reads(R_Visible)\r\n");
const identifiers = result.tokens
.filter((token) => token.kind === "identifier")
.map((token) => token.text);
assert.equal(identifiers.includes("R_Hidden"), false);
assert.equal(identifiers.includes("R_Visible"), true);
});
test("balanced contexts retain multiline nesting and argument indexes", () => {
const source = [
"atom_info(",
"\tatom_phase(cube_g4),",
"\tatom_reads(R_A, nested(R_B, R_C)),",
"\tatom_writes(R_D)",
")",
].join("\n");
const lexical = lex(source);
const balanced = buildCallContexts(lexical.tokens);
const byText = new Map();
lexical.tokens.forEach((token, index) => {
if (token.kind === "identifier") byText.set(token.text, index);
});
assert.equal(nearestCall(balanced.contexts, byText.get("cube_g4")).callee, "atom_phase");
assert.equal(nearestCall(balanced.contexts, byText.get("R_A")).callee, "atom_reads");
assert.equal(nearestCall(balanced.contexts, byText.get("R_A")).argIndex, 0);
assert.equal(nearestCall(balanced.contexts, byText.get("R_C")).callee, "nested");
assert.equal(nearestCall(balanced.contexts, byText.get("R_D")).callee, "atom_writes");
assert.deepEqual(balanced.errors, []);
});
test("balanced contexts report unmatched parentheses", () => {
const lexical = lex("atom_reads(R_A");
const balanced = buildCallContexts(lexical.tokens);
assert.deepEqual(balanced.errors.map((error) => error.kind), ["unmatched-open-paren"]);
});
+134
View File
@@ -0,0 +1,134 @@
"use strict";
const assert = require("node:assert/strict");
const test = require("node:test");
const {
createIndex,
domainFromPath,
mergeIndexes,
scanSource,
} = require("../source-index");
test("scanSource discovers current atom and component forms", () => {
const source = [
"MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4), atom_reads(R_PrimCursor)) { mac_yield() };",
"MipsAtomComp_(ac_load_pair) { load_word(R_T0, R_T1, 0) };",
"internal MipsAtom* normalize_proc(AtomArena_R aa) MipsAtom_Proc_(aa, { mac_yield() })",
"FI_ void ac_store_pair(MipsAtomBuilder_R ab) atom_dbg_skip MipsAtomComp_Proc_(ab, { store_word(R_T0, R_T1, 0) })",
].join("\n");
const result = scanSource(source, "C:/projects/Pikuma/ps1/code/duffle/mips.atom.c");
assert.equal(result.index.atoms.has("cube_g4_face"), true);
assert.equal(result.index.atoms.has("normalize"), true);
assert.equal(result.index.components.has("ac_load_pair"), true);
assert.equal(result.index.components.has("ac_store_pair"), true);
assert.equal(result.index.componentAliases.has("mac_load_pair"), true);
assert.equal(result.index.componentAliases.has("mac_store_pair"), true);
assert.equal(result.index.macros.get("mac_store_pair"), "component");
assert.equal(result.index.componentCallees.get("mac_store_pair").includes("store_word"), true);
assert.equal(result.index.phases.has("cube_g4"), true);
assert.equal(result.index.registers.get("R_PrimCursor"), "gpr");
assert.deepEqual(result.errors, []);
});
test("scanSource discovers binds, labels, registers, typedefs, and macro domains", () => {
const source = [
"typedef Struct_(Binds_CubeTri) { U4 PrimCursor; };",
"typedef Enum_(U4, PadStatus) { PadStatus_Ok };",
"typedef U4 const MipsCode;",
"enum { R_PrimCursor = R_T7 atom_reg, C2_Custom = 12, gte_cr_Custom = 13 };",
"#define load_word(rt, base, off) enc_i(rt, base, off)",
"atom_bind(Binds_CubeTri)",
"atom_label(exit)",
"atom_offset(entry, exit)",
].join("\n");
const result = scanSource(source, "C:/projects/Pikuma/ps1/code/duffle/mips.h");
assert.equal(result.index.bindTypes.has("Binds_CubeTri"), true);
assert.equal(result.index.types.has("PadStatus"), true);
assert.equal(result.index.types.has("MipsCode"), true);
assert.equal(result.index.registers.get("R_PrimCursor"), "gpr");
assert.equal(result.index.registers.get("C2_Custom"), "cop2");
assert.equal(result.index.registers.get("gte_cr_Custom"), "cop2");
assert.equal(result.index.macros.get("load_word"), "cpu");
assert.equal(result.index.labels.has("entry"), true);
assert.equal(result.index.labels.has("exit"), true);
});
test("domainFromPath uses the declaration file rather than parent directory names", () => {
assert.equal(domainFromPath("C:/x/code/hello_gte/hello_gte.atom.c"), null);
assert.equal(domainFromPath("C:/x/code/duffle/mips.h"), "cpu");
assert.equal(domainFromPath("C:/x/code/duffle/gte.h"), "gte");
assert.equal(domainFromPath("C:/x/code/duffle/gp.h"), "gpu");
});
test("component aliases inherit the domain of the instructions they emit", () => {
const headers = mergeIndexes(
scanSource("#define load_word(a,b,c) 1\n#define store_word(a,b,c) 1\n#define shift_aright_var(a,b,c) 1\n#define jump_reg(rd) 1\n", "C:/x/code/duffle/mips.h").index,
scanSource("#define gte_sw(rt, base, off) 1\n", "C:/x/code/duffle/gte.h").index
);
const math = scanSource(
[
"MipsAtomComp_(ac_load_v3s4) { load_word(R_T0, R_T1, 0) };",
"#define mac_load_p3s4 mac_load_v3s4",
].join("\n"),
"C:/x/code/duffle/math.atom.c"
);
const shift = scanSource(
"MipsAtomComp_(ac_shift_aright_var_v3_self) { shift_aright_var(R_T0, R_T0, R_T1) };",
"C:/x/code/duffle/gte.atom.c"
);
const gte = scanSource(
"MipsAtomComp_(ac_gte_store_f3) { gte_sw(C2_SXY0, R_T0, 0) };",
"C:/x/code/duffle/gte.atom.c"
);
const yieldAtom = scanSource(
"MipsAtomComp_(ac_yield) { load_word(R_AtomJmp, R_TapePtr, 0), jump_reg(R_AtomJmp), nop };",
"C:/x/code/duffle/lottes_tape.h"
);
const merged = mergeIndexes(headers, math.index, shift.index, gte.index, yieldAtom.index);
assert.equal(merged.macros.get("mac_load_v3s4"), "cpu");
assert.equal(merged.macros.get("mac_load_p3s4"), "cpu");
assert.equal(merged.macros.get("mac_shift_aright_var_v3_self"), "cpu");
assert.equal(merged.macros.get("mac_gte_store_f3"), "gte");
assert.equal(merged.macros.get("mac_yield"), "control");
});
test("scanSource tags utility header defines as utility, not a hardware domain", () => {
const source = [
"#define assert(cond) ((void)(cond))",
"#define stringify(name) #name",
"#define u4_hi(imm) ((imm) >> 16)",
].join("\n");
const result = scanSource(source, "C:/projects/Pikuma/ps1/code/duffle/dsl.h");
assert.equal(result.index.macros.get("assert"), "utility");
assert.equal(result.index.macros.get("stringify"), "utility");
assert.equal(result.index.macros.get("u4_hi"), "utility");
});
test("mergeIndexes prefers a hardware domain over a later utility define", () => {
const left = createIndex();
left.macros.set("sub_s", "utility");
const right = createIndex();
right.macros.set("sub_s", "cpu");
assert.equal(mergeIndexes(left, right).macros.get("sub_s"), "cpu");
assert.equal(mergeIndexes(right, left).macros.get("sub_s"), "cpu");
});
test("mergeIndexes preserves domain-specific aliases", () => {
const left = createIndex();
left.macros.set("load_word", "cpu");
const right = createIndex();
right.componentAliases.add("mac_gte_store");
right.macros.set("mac_gte_store", "gte");
const merged = mergeIndexes(left, right);
assert.equal(merged.macros.get("load_word"), "cpu");
assert.equal(merged.macros.get("mac_gte_store"), "gte");
});
+14 -21
View File
@@ -1,24 +1,17 @@
This is free and unencumbered software released into the public domain. Copyright (C) 2026 Edward R. Gonzalez
Anyone is free to copy, modify, publish, use, compile, sell, or This software is provided 'as-is', without any express or implied
distribute this software, either in source code form or as a compiled warranty. In no event will the authors be held liable for any damages
binary, for any purpose, commercial or non-commercial, and by any arising from the use of this software.
means.
In jurisdictions that recognize copyright laws, the author or authors Permission is granted to anyone to use this software for any purpose,
of this software dedicate any and all copyright interest in the including commercial applications, and to alter it and redistribute it
software to the public domain. We make this dedication for the benefit freely, subject to the following restrictions:
of the public at large and to the detriment of our heirs and
successors. We intend this dedication to be an overt act of
relinquishment in perpetuity of all present and future rights to this
software under copyright law.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, 1. The origin of this software must not be misrepresented; you must not
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF claim that you wrote the original software. If you use this software
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. in a product, an acknowledgment in the product documentation would be
IN NO EVENT SHALL THE AUTHORS BE LIABLE FOR ANY CLAIM, DAMAGES OR appreciated but is not required.
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, 2. Altered source versions must be plainly marked as such, and must not be
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR misrepresented as being the original software.
OTHER DEALINGS IN THE SOFTWARE. 3. This notice may not be removed or altered from any source distribution.
For more information, please refer to <https://unlicense.org>
+1
View File
@@ -1,6 +1,7 @@
#ifdef INTELLISENSE_DIRECTIVES #ifdef INTELLISENSE_DIRECTIVES
# pragma once # pragma once
#endif #endif
enum { enum {
bios_init_pad_2 = 0x12, bios_init_pad_2 = 0x12,
bios_start_pad_2 = 0x13, bios_start_pad_2 = 0x13,
+2 -2
View File
@@ -70,8 +70,8 @@
/* ---------------------------------------------------------------------------- /* ----------------------------------------------------------------------------
* atom_reg (per-enum opt-in marker for the DWARF register-alias registry) * atom_reg (per-enum opt-in marker for the DWARF register-alias registry)
* *
* The bare `atom_reg` token adjacent to an enum entry in mips.h / lottes_tape.h flags that alias as debug-visible for scan_source's register_alias_registry. * Bare `atom_reg` token adjacent to an enum entry that alias as debug-visible for scan_source's register_alias_registry.
* The C preprocessor strips it to a comment so no runtime symbol is created; the Lua scanner reads the bare token. * Lua scanner reads the bare token.
* ----------------------------------------------------------------------------*/ * ----------------------------------------------------------------------------*/
#define atom_reg /* atom_reg: opt the preceding enum entry into the DWARF registry */ #define atom_reg /* atom_reg: opt the preceding enum entry into the DWARF registry */
+16 -13
View File
@@ -3,7 +3,7 @@
# include "assert.h" # include "assert.h"
#endif #endif
#define offset_of(type, member) cast(U8,__builtin_offsetof(type,member)) #define offset_of(type, member) cast(U8,__builtin_offsetof(type,member)) // Compiler builtin version of O_
#define static_assert _Static_assert #define static_assert _Static_assert
#define typeof __typeof__ #define typeof __typeof__
#define typeof_ptr(ptr) typeof((ptr)[0]) #define typeof_ptr(ptr) typeof((ptr)[0])
@@ -97,6 +97,7 @@
#define Array_expand(type,len) type Array_sym(type, len)[len]; typedef PtrSet_(Array_sym(type, len)) #define Array_expand(type,len) type Array_sym(type, len)[len]; typedef PtrSet_(Array_sym(type, len))
#define Array_(type,len) Array_expand(type,len) #define Array_(type,len) Array_expand(type,len)
#define Bit_(id,b) id = (1 << b), tmpl(id,pos) = b #define Bit_(id,b) id = (1 << b), tmpl(id,pos) = b
#define Bitmask_(b) (1u << b)
#define Enum_(underlying_type, symbol) underlying_type TSet_(symbol); enum symbol #define Enum_(underlying_type, symbol) underlying_type TSet_(symbol); enum symbol
#define Proc_(symbol) symbol #define Proc_(symbol) symbol
#define Relative_(symbol) // Does nothing but annotate that a symbol is associated with another. #define Relative_(symbol) // Does nothing but annotate that a symbol is associated with another.
@@ -139,16 +140,16 @@ enum { false = 0, true = 1, true_overflow, };
typedef void Proc_(VoidFn) (void); typedef void Proc_(VoidFn) (void);
#define kilo(n) (C_(U4, n) << 10) #define Kilo_(n) (C_(U4, n) << 10)
#define mega(n) (C_(U4, n) << 20) #define Mega_(n) (C_(U4, n) << 20)
#define giga(n) (C_(U4, n) << 30) #define Giga_(n) (C_(U4, n) << 30)
#define tera(n) (C_(U4, n) << 40) #define Tera_(n) (C_(U4, n) << 40)
#define null C_(U4, 0) #define null C_(U4, 0)
#define nullptr C_(void*, 0) #define nullptr C_(void*, 0)
#define O_(type, field) C_(U4, & C_(type*,0)->field) #define O_(type, field) C_(U4, & C_(type*,0)->field)
#define OA_(type, member, idx) C_(U4, & C_(type*,0)->member[idx]) #define OA_(type, aexpr) C_(U4, & C_(type*,0) aexpr)
#define OT_(field) O_(typeof_ptr(& field), filed)) #define OT_(field) O_(typeof_ptr(& field), field))
#define S_(data) C_(U4, sizeof(data)) #define S_(data) C_(U4, sizeof(data))
#define sop_1(op,a,b) C_(U1, s1_(a) op s1_(b)) #define sop_1(op,a,b) C_(U1, s1_(a) op s1_(b))
@@ -185,7 +186,7 @@ def_signed_ops(le, <=)
#define alignas _Alignas #define alignas _Alignas
#define alignof _Alignof #define alignof _Alignof
#define byte_pad(amount, ...) B1 glue(_PAD_, __VA_ARGS__) [amount] #define byte_pad(amount, ...) B1 glue(_PAD_, __VA_ARGS__) [amount]
#define pcast(type, data) (C_(type*, & (data)) [0]) #define C_ptr(type, data) (C_(type*, & (data)) [0])
#define dbg_args(...) __VA_ARGS__ #define dbg_args(...) __VA_ARGS__
@@ -200,6 +201,8 @@ def_signed_ops(le, <=)
#define defer_info(type,expr, ...) for(type info= {__VA_ARGS__}; info.once!=1;++info.once,(expr)) // Defer with tracked state #define defer_info(type,expr, ...) for(type info= {__VA_ARGS__}; info.once!=1;++info.once,(expr)) // Defer with tracked state
#define do_while(cond) for (U8 once=0; once!=1 || (cond); ++once) #define do_while(cond) for (U8 once=0; once!=1 || (cond); ++once)
#define Jmp_nZero_(cond,label) if (cond) goto label;
#pragma endregion Control Flow & Iteration #pragma endregion Control Flow & Iteration
#define span_iter(type, iter, m_begin, op, m_end) ( \ #define span_iter(type, iter, m_begin, op, m_end) ( \
@@ -216,16 +219,16 @@ def_signed_ops(le, <=)
typedef Span_(S4); typedef Span_(S4);
typedef Span_(U4); typedef Span_(U4);
#if 0
#pragma region Debug #pragma region Debug
#define debug_trap() __builtin_debugtrap() #define debug_trap() __builtin_trap()
#if BUILD_DEBUG #if BUILD_DEBUG
IA_ void assert(U8 cond) { if(cond){return;} else{debug_trap(); ms_exit_process(1);} } #define assert(cond) if(cond == false){debug_trap();}
#else #else
#define assert(cond) # ifndef assert
# include <assert.h>
# endif
#endif #endif
#pragma endregion Debug #pragma endregion Debug
#endif
#define GCC_OPTIMIZATION_DISABLE _Pragma("GCC push_options") _Pragma("GCC optimize(\"O0\")") #define GCC_OPTIMIZATION_DISABLE _Pragma("GCC push_options") _Pragma("GCC optimize(\"O0\")")
#define GCC_OPTIMIZATION_ENABLE _Pragma("GCC pop_options") #define GCC_OPTIMIZATION_ENABLE _Pragma("GCC pop_options")
+200 -89
View File
@@ -17,7 +17,7 @@
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h // source: C:\projects\Pikuma\ps1\code\duffle\bios.h
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h // source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c // source: C:\projects\Pikuma\ps1\code\duffle\pad.c
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c // source: C:\projects\Pikuma\ps1\code\duffle\math.atom.h
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c // source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c // source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c // source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
@@ -35,15 +35,11 @@
* These do NOT yield. They are expanded inline inside Tape Atoms. * These do NOT yield. They are expanded inline inside Tape Atoms.
* ---------------------------------------------------------------------------*/ * ---------------------------------------------------------------------------*/
// The 'Yield' sequence for Tape Atoms (mac_yield). // The 'Yield' sequence for Tape Atoms (mac_yield).
// - mac_yield() is the safe default for atom-endings: 4 words, BD-slot of jr is mandatory nop.
// - mac_yield_load() + mac_yield_tail():
// - unconditional branch: mac_yield_load fills the branch's BD-slot (replaces a nop);
// - mac_yield_tail runs at the branch target (does NOT re-load R_AtomJmp).
#define mac_yield(...) \ #define mac_yield(...) \
load_word(R_AtomJmp, R_TapePtr, 0) \ load_word(R_AtomJmp, R_TapePtr, 0) \
, add_ui_self( R_TapePtr, S_(MipsCode)) \ , add_ui_self( R_TapePtr, S_(MipsCode)) \
, jump_reg( R_AtomJmp) \ , jump_reg( R_AtomJmp) \
, nop , BdSlot_ nop
WORD_COUNT(mac_yield, 4) WORD_COUNT(mac_yield, 4)
/* atom_dbg_skip */ /* atom_dbg_skip */
@@ -55,13 +51,24 @@ WORD_COUNT(mac_yield_load, 1)
#define mac_yield_tail(...) \ #define mac_yield_tail(...) \
add_ui_self(R_TapePtr, S_(MipsCode)) \ add_ui_self(R_TapePtr, S_(MipsCode)) \
, jump_reg( R_AtomJmp) \ , jump_reg( R_AtomJmp) \
, nop , BdSlot_ nop
WORD_COUNT(mac_yield_tail, 3) WORD_COUNT(mac_yield_tail, 3)
/* atom_dbg_skip */
#define mac_load_half_v3(tx, ty, tz, base, offset) \
load_half(tx, base, offset + OA_(U2,[0])) \
, load_half(ty, base, offset + OA_(U2,[1])) \
, load_half(tz, base, offset + OA_(U2,[2]))
WORD_COUNT(mac_load_half_v3, 3)
#define mac_load_v3s2(transfer, base, offset) \
mac_load_half_v3(transfer.x, transfer.y, transfer.z, base, offset)
WORD_COUNT(mac_load_v3s2, 3)
/* atom_dbg_skip */ /* atom_dbg_skip */
#define mac_load_v2s2(rs_x, rs_y, r_base, offset) \ #define mac_load_v2s2(rs_x, rs_y, r_base, offset) \
load_half( rs_x, r_base, offset + O_(V3_S2,x)) \ load_half(rs_x, r_base, offset + O_(V3_S2,x)) \
, load_half( rs_y, r_base, offset + O_(V3_S2,y)) , load_half(rs_y, r_base, offset + O_(V3_S2,y))
WORD_COUNT(mac_load_v2s2, 2) WORD_COUNT(mac_load_v2s2, 2)
/* atom_dbg_skip */ /* atom_dbg_skip */
@@ -71,26 +78,75 @@ WORD_COUNT(mac_load_v2s2, 2)
WORD_COUNT(mac_store_v2s2, 2) WORD_COUNT(mac_store_v2s2, 2)
/* atom_dbg_skip */ /* atom_dbg_skip */
#define mac_load_v3s4(rs_x, rs_y, rs_z, r_base, offset) \ #define mac_load_word_v3(tx, ty, tz, base, offset) \
load_word( rs_x, r_base, offset + O_(V3_S4,x)) \ load_word(tx, base, offset + OA_(U4,[0])) \
, load_word( rs_y, r_base, offset + O_(V3_S4,y)) \ , load_word(ty, base, offset + OA_(U4,[1])) \
, load_word( rs_z, r_base, offset + O_(V3_S4,z)) , load_word(tz, base, offset + OA_(U4,[2]))
WORD_COUNT(mac_load_word_v3, 3)
#define mac_load_v3s4(transfer, base, offset) \
mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
WORD_COUNT(mac_load_v3s4, 3) WORD_COUNT(mac_load_v3s4, 3)
/* atom_dbg_skip */ #define mac_load_p3s4(transfer, base, offset) \
#define mac_store_v3s4(rt_x, rt_y, rt_z, base, offset) \ mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
store_word(rt_x, base, offset + O_(V3_S4,x)) \ WORD_COUNT(mac_load_p3s4, 3)
, store_word(rt_y, base, offset + O_(V3_S4,y)) \
, store_word(rt_z, base, offset + O_(V3_S4,z))
WORD_COUNT(mac_store_v3s4, 3)
/* atom_dbg_skip */ /* atom_dbg_skip */
#define mac_sub_v3s4(rds_x, rds_y, rds_z, rt_x, rt_y, rt_z) \ #define mac_store_half_v3(tx, ty, tz, base, offset) \
sub_s(rds_x, rds_x, rt_x) \ store_half(tx, base, offset + OA_(U2,[0])) \
, sub_s(rds_y, rds_y, rt_y) \ , store_half(ty, base, offset + OA_(U2,[1])) \
, sub_s(rds_z, rds_z, rt_z) , store_half(tz, base, offset + OA_(U2,[2]))
WORD_COUNT(mac_store_half_v3, 3)
#define mac_store_v3s2(transfer, base, offset) \
mac_store_half_v3(transfer.x, transfer.y, transfer.z, base, offset)
WORD_COUNT(mac_store_v3s2, 3)
/* atom_dbg_skip */
#define mac_store_word_v3(tx, ty, tz, base, offset) \
store_word(tx, base, offset + OA_(U4,[0])) \
, store_word(ty, base, offset + OA_(U4,[1])) \
, store_word(tz, base, offset + OA_(U4,[2]))
WORD_COUNT(mac_store_word_v3, 3)
#define mac_store_v3s4(transfer, base, offset) \
mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
WORD_COUNT(mac_store_v3s4, 3)
#define mac_store_p3s4(transfer, base, offset) \
mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
WORD_COUNT(mac_store_p3s4, 3)
/* atom_dbg_skip */
#define mac_add_si_v3s4(rt_x, rt_y, rt_z, base, offset) \
add_si(rt_x, base, O_(V3_S4,x)) \
, add_si(rt_y, base, O_(V3_S4,y)) \
, add_si(rt_z, base, O_(V3_S4,z))
WORD_COUNT(mac_add_si_v3s4, 3)
/* atom_dbg_skip */
#define mac_sub_s_v3(dx, dy, dz, sx, sy, sz, tx, ty, tz) \
sub_s(dx, sx, tx) \
, sub_s(dy, sy, ty) \
, sub_s(dz, sz, tz)
WORD_COUNT(mac_sub_s_v3, 3)
#define mac_sub_v3s4(d, s, t) \
mac_sub_s_v3(d.x, d.y, d.z, s.x, s.y, s.z, t.x, t.y, t.z)
WORD_COUNT(mac_sub_v3s4, 3) WORD_COUNT(mac_sub_v3s4, 3)
/* atom_dbg_skip */
#define mac_sub_s_v3_self(ds_x, ds_y, ds_z, tx, ty, tz) \
sub_s(ds_x, ds_x, tx) \
, sub_s(ds_y, ds_y, ty) \
, sub_s(ds_z, ds_z, tz)
WORD_COUNT(mac_sub_s_v3_self, 3)
#define mac_sub_v3s4_self(ds, t) \
mac_sub_s_v3_self(ds.x, ds.y, ds.z, t.x, t.y, t.z)
WORD_COUNT(mac_sub_v3s4_self, 3)
/* atom_dbg_skip */ /* atom_dbg_skip */
#define mac_store_rects2(rt_x, rt_y, rt_width, rt_height, base, offset) \ #define mac_store_rects2(rt_x, rt_y, rt_width, rt_height, base, offset) \
store_half(rt_x, base, offset + O_(Rect_S2,x)) \ store_half(rt_x, base, offset + O_(Rect_S2,x)) \
@@ -99,6 +155,39 @@ WORD_COUNT(mac_sub_v3s4, 3)
, store_half(rt_height, base, offset + O_(Rect_S2,height)) , store_half(rt_height, base, offset + O_(Rect_S2,height))
WORD_COUNT(mac_store_rects2, 4) WORD_COUNT(mac_store_rects2, 4)
/* atom_dbg_skip */
#define mac_load_word_imm(dst, imm) \
load_upper_i(dst, u4_hi(imm)) \
, or_i_self( dst, u4_lo(imm))
WORD_COUNT(mac_load_word_imm, 2)
#define mac_shift_aright_v3_self(dt_x, dt_y, dt_z, shift_amount) \
shift_aright(dt_x, dt_x, shift_amount) \
, shift_aright(dt_y, dt_y, shift_amount) \
, shift_aright(dt_z, dt_z, shift_amount)
WORD_COUNT(mac_shift_aright_v3_self, 3)
#define mac_shift_aright_v3s4_self(dt, shift) \
mac_shift_aright_v3_self(dt.x, dt.y, dt.z, shift)
WORD_COUNT(mac_shift_aright_v3s4_self, 3)
#define mac_shift_aright_var_v3(rd_v0, rd_v1, rd_v2, rs_v0, rs_v1, rs_v2, r_shift) \
shift_aright_var(rd_v0, rs_v0, r_shift) \
, shift_aright_var(rd_v1, rs_v1, r_shift) \
, shift_aright_var(rd_v2, rs_v2, r_shift)
WORD_COUNT(mac_shift_aright_var_v3, 3)
/* atom_dbg_skip */
#define mac_shift_aright_var_v3_self(rds_v0, rds_v1, rds_v2, r_shift) \
shift_aright_var(rds_v0, rds_v0, r_shift) \
, shift_aright_var(rds_v1, rds_v1, r_shift) \
, shift_aright_var(rds_v2, rds_v2, r_shift)
WORD_COUNT(mac_shift_aright_var_v3_self, 3)
#define mac_shift_aright_var_v3s4_self(ds, shift) \
mac_shift_aright_var_v3_self(ds.x, ds.y, ds.z, shift)
WORD_COUNT(mac_shift_aright_var_v3s4_self, 3)
/* atom_dbg_skip */ /* atom_dbg_skip */
#define mac_load_tri_indices(r_face_cusor, r_i0, r_i1, r_i2) \ #define mac_load_tri_indices(r_face_cusor, r_i0, r_i1, r_i2) \
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)) \ load_half_u(r_i0, r_face_cusor, 0 * S_(S2)) \
@@ -106,6 +195,30 @@ WORD_COUNT(mac_store_rects2, 4)
, load_half_u(r_i2, r_face_cusor, 2 * S_(S2)) , load_half_u(r_i2, r_face_cusor, 2 * S_(S2))
WORD_COUNT(mac_load_tri_indices, 3) WORD_COUNT(mac_load_tri_indices, 3)
#define mac_gte_mv_to_cr_diag_v3s4(v) \
gte_mv_to_ctrl_r(v.y, gte_cr_RT13) \
, gte_mv_to_ctrl_r(v.z, gte_cr_RT22) \
, gte_mv_to_ctrl_r(v.x, gte_cr_RT11)
WORD_COUNT(mac_gte_mv_to_cr_diag_v3s4, 3)
#define mac_gte_ld_ir123_v3s4(v) \
gte_mv_to_data_r(v.x, C2_IR1) \
, gte_mv_to_data_r(v.y, C2_IR2) \
, gte_mv_to_data_r(v.z, C2_IR3)
WORD_COUNT(mac_gte_ld_ir123_v3s4, 3)
/* atom_dbg_skip */
#define mac_gte_op_cross_v3s4(a, b) \
mac_gte_mv_to_cr_diag_v3s4(a) \
GteDelay_ /* RT diagonal: D1 = a.x, D2 = a.y, D3 = a.z */ \
, mac_gte_ld_ir123_v3s4(b) \
GteDelay_ /* IR: second operand (b.xyz) */ \
, gte_cmdw_cross /* OP: MAC1/2/3 = a × b (S12.20) */ \
, mac_gte_mv_from_mac123_v3s4(a) \
GteDelay_ /* Read MAC1/2/3 → a.xyz (overwrites source-A's load targets) */ \
, mac_shift_aright_v3s4_self(a, 12) /* Right-shift MAC by 12 (S12.20 → S12.0 OuterProduct12) */
WORD_COUNT(mac_gte_op_cross_v3s4, 16)
/* atom_dbg_skip */ /* atom_dbg_skip */
#define mac_gte_store_f3(r_primitive_cursor) \ #define mac_gte_store_f3(r_primitive_cursor) \
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)) \ gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)) \
@@ -119,19 +232,19 @@ WORD_COUNT(mac_gte_store_f3, 3)
, add_u_self(R_AT, r_vert_base) \ , add_u_self(R_AT, r_vert_base) \
, load_word(R_V0, R_AT, O_(V3_S2,x)) \ , load_word(R_V0, R_AT, O_(V3_S2,x)) \
, load_word(R_V1, R_AT, O_(V3_S2,z)) \ , load_word(R_V1, R_AT, O_(V3_S2,z)) \
, gte_mv_to_data_r(R_V0, C2_VXY0) \ , LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY0) \
, gte_mv_to_data_r(R_V1, C2_VZ0) \ , gte_mv_to_data_r(R_V1, C2_VZ0) \
, shift_lleft(R_AT, r_v1, v3s2_byteoff) \ , shift_lleft(R_AT, r_v1, v3s2_byteoff) \
, add_u_self(R_AT, r_vert_base) \ , add_u_self(R_AT, r_vert_base) \
, load_word(R_V0, R_AT, O_(V3_S2,x)) \ , load_word(R_V0, R_AT, O_(V3_S2,x)) \
, load_word(R_V1, R_AT, O_(V3_S2,z)) \ , load_word(R_V1, R_AT, O_(V3_S2,z)) \
, gte_mv_to_data_r(R_V0, C2_VXY1) \ , LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY1) \
, gte_mv_to_data_r(R_V1, C2_VZ1) \ , gte_mv_to_data_r(R_V1, C2_VZ1) \
, shift_lleft(R_AT, r_v2, v3s2_byteoff) \ , shift_lleft(R_AT, r_v2, v3s2_byteoff) \
, add_u_self(R_AT, r_vert_base) \ , add_u_self(R_AT, r_vert_base) \
, load_word(R_V0, R_AT, O_(V3_S2,x)) \ , load_word(R_V0, R_AT, O_(V3_S2,x)) \
, load_word(R_V1, R_AT, O_(V3_S2,z)) \ , load_word(R_V1, R_AT, O_(V3_S2,z)) \
, gte_mv_to_data_r(R_V0, C2_VXY2) \ , LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY2) \
, gte_mv_to_data_r(R_V1, C2_VZ2) , gte_mv_to_data_r(R_V1, C2_VZ2)
WORD_COUNT(mac_gte_load_tri_verts, 18) WORD_COUNT(mac_gte_load_tri_verts, 18)
@@ -149,23 +262,28 @@ WORD_COUNT(mac_gte_store_g4_p3, 1)
/* atom_dbg_skip */ /* atom_dbg_skip */
#define mac_gte_sqr_v3(r_sx, r_sy, r_sz, r_sq_x, r_sq_y, r_sq_z) \ #define mac_gte_sqr_v3(r_sx, r_sy, r_sz, r_sq_x, r_sq_y, r_sq_z) \
gte_mv_to_data_r(r_sx, C2_IR1) \ mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop) \
, gte_mv_to_data_r(r_sy, C2_IR2) \
, gte_mv_to_data_r(r_sz, C2_IR3) \
, nop \
, gte_cmdw_sqr \
, gte_mv_from_data_r(r_sq_x, C2_MAC1) \ , gte_mv_from_data_r(r_sq_x, C2_MAC1) \
, gte_mv_from_data_r(r_sq_y, C2_MAC2) \ , gte_mv_from_data_r(r_sq_y, C2_MAC2) \
, gte_mv_from_data_r(r_sq_z, C2_MAC3) , gte_mv_from_data_r(r_sq_z, C2_MAC3)
WORD_COUNT(mac_gte_sqr_v3, 8) WORD_COUNT(mac_gte_sqr_v3, 8)
/* atom_dbg_skip */
#define mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, delay_slot) \
gte_mv_to_data_r(r_sx, C2_IR1) \
, gte_mv_to_data_r(r_sy, C2_IR2) \
, gte_mv_to_data_r(r_sz, C2_IR3) \
, delay_slot \
, gte_cmdw_sqr
WORD_COUNT(mac_gte_sqr_v3s4, 5)
/* atom_dbg_skip */ /* atom_dbg_skip */
#define mac_gte_gpf_scale(r_sx, r_sy, r_sz, r_recip_est, r_shift, r_dx, r_dy, r_dz) \ #define mac_gte_gpf_scale(r_sx, r_sy, r_sz, r_recip_est, r_shift, r_dx, r_dy, r_dz) \
gte_mv_to_data_r(r_recip_est, C2_IR0) \ gte_mv_to_data_r(r_recip_est, C2_IR0) \
, gte_mv_to_data_r(r_sx, C2_IR1) \ , gte_mv_to_data_r(r_sx, C2_IR1) \
, gte_mv_to_data_r(r_sy, C2_IR2) \ , gte_mv_to_data_r(r_sy, C2_IR2) \
, gte_mv_to_data_r(r_sz, C2_IR3) \ , gte_mv_to_data_r(r_sz, C2_IR3) \
, nop2 /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */ \ , GteDelay_ nop2 /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */ \
, gte_cmdw_gpf \ , gte_cmdw_gpf \
, gte_mv_from_data_r(r_dx, C2_MAC1) \ , gte_mv_from_data_r(r_dx, C2_MAC1) \
, gte_mv_from_data_r(r_dy, C2_MAC2) \ , gte_mv_from_data_r(r_dy, C2_MAC2) \
@@ -173,60 +291,52 @@ WORD_COUNT(mac_gte_sqr_v3, 8)
, shift_aright_var(r_dx, r_dx, r_shift) \ , shift_aright_var(r_dx, r_dx, r_shift) \
, shift_aright_var(r_dy, r_dy, r_shift) \ , shift_aright_var(r_dy, r_dy, r_shift) \
, shift_aright_var(r_dz, r_dz, r_shift) , shift_aright_var(r_dz, r_dz, r_shift)
WORD_COUNT(mac_gte_gpf_scale, 13) WORD_COUNT(mac_gte_gpf_scale, 12)
#define mac_apply_matrix_lv(r_mtx, r_vec, r_out, r_t0, r_t1, r_t2) \ #define mac_trans_mt3s3s4(r_mtx, r_off, r_t0, r_t1, r_t2) \
load_word(r_t0, r_mtx, 0) \ load_word( r_t0, r_off, O_(V3_S4,x)) \
, nop \ , load_word( r_t1, r_off, O_(V3_S4,y)) \
, gte_mv_to_ctrl_r(r_t0, gte_cr_RT11_Code) \ , load_word( r_t2, r_off, O_(V3_S4,z)) \
, load_word(r_t0, r_mtx, 4) \ , store_word(r_t0, r_mtx, O_(MT3_S2S4,t[0])) \
, nop \
, gte_mv_to_ctrl_r(r_t0, gte_cr_RT12_Code) \
, load_word(r_t0, r_mtx, 8) \
, nop \
, gte_mv_to_ctrl_r(r_t0, gte_cr_RT13_Code) \
, load_word(r_t0, r_mtx, 12) \
, nop \
, gte_mv_to_ctrl_r(r_t0, gte_cr_RT21_Code) \
, load_half_u(r_t0, r_mtx, 16) \
, nop \
, gte_mv_to_ctrl_r(r_t0, gte_cr_RT22_Code) \
, nop2 /* Load PACKED pos into V0 (libgte SVECTOR layout).
* r_vec points to atom-0-staged packed data ((pos.y << 16) | pos.x at +0, pos.z at +4).
* LWC2 base register MUST be the pointer r_vec, NOT the loaded value r_t0. */ \
, load_word(r_t0, r_vec, 0) \
, nop \
, gte_lw(C2_VXY0, r_vec, 0) \
, load_word(r_t0, r_vec, 4) \
, nop \
, gte_lw(C2_VZ0, r_vec, 4) /* RTPS: cv=3 (no translation), sf=1 (no shift, integer), v=0 (V0 input),
* mx=0 (rotation matrix). MAC = RT row · V0 + 0. RTPS also writes
* SXY0/1/2 + SZ0..SZ3 (perspective division); ignored. */ \
, gte_cmdw_rtps_sf1 /* Read MAC1/2/3 → out. */ \
, gte_mv_from_data_r(r_t0, C2_MAC1) \
, gte_mv_from_data_r(r_t1, C2_MAC2) \
, gte_mv_from_data_r(r_t2, C2_MAC3) \
, nop \
, store_word(r_t0, r_out, 0) \
, store_word(r_t1, r_out, 4) \
, store_word(r_t2, r_out, 8)
WORD_COUNT(mac_apply_matrix_lv, 31)
#define mac_trans_matrix(r_mtx, r_off, r_t1) \
load_word(r_t1, r_off, O_(V3_S4,x)) \
, nop \
, store_word(r_t1, r_mtx, O_(MT3_S2S4,t[0])) \
, load_word(r_t1, r_off, O_(V3_S4,y)) \
, nop \
, store_word(r_t1, r_mtx, O_(MT3_S2S4,t[1])) \ , store_word(r_t1, r_mtx, O_(MT3_S2S4,t[1])) \
, load_word(r_t1, r_off, O_(V3_S4,z)) \ , store_word(r_t2, r_mtx, O_(MT3_S2S4,t[2]))
, nop \ WORD_COUNT(mac_trans_mt3s3s4, 6)
, store_word(r_t1, r_mtx, O_(MT3_S2S4,t[2]))
WORD_COUNT(mac_trans_matrix, 9)
/* atom_dbg_skip */
#define mac_lzcr_round_even_half_shift(r_shift, r_mag_sq, r_mag_sq_copy) \
and_i(r_shift, r_shift, gte_lzcr_even_mask) \
, or_u(r_mag_sq_copy, r_mag_sq, 0) \
, li_s( r_mag_sq, 31) \
, sub_s( r_mag_sq, r_mag_sq, r_shift) \
, shift_aright(r_mag_sq, r_mag_sq, 1)
WORD_COUNT(mac_lzcr_round_even_half_shift, 5)
#define mac_gte_general_purpose_interopolation(to_ir0, to_ir1, to_ir2, to_ir3, fr_mac1, fr_mac2, fr_mac3, nop_slot1, nop_slot2) \
gte_mv_to_data_r(to_ir0, C2_IR0) \
, gte_mv_to_data_r(to_ir1, C2_IR1) /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */ \
, gte_mv_to_data_r(to_ir2, C2_IR2) \
, gte_mv_to_data_r(to_ir3, C2_IR3) /* IR3 = src.z (reloaded) */ \
, GteDelay_ nop_slot1 \
, GteDelay_ nop_slot2 \
, gte_cmdw_gpf \
, gte_mv_from_data_r(fr_mac1, C2_MAC1) \
, gte_mv_from_data_r(fr_mac2, C2_MAC2) \
, gte_mv_from_data_r(fr_mac3, C2_MAC3)
WORD_COUNT(mac_gte_general_purpose_interopolation, 10)
#define mac_gte_mv_from_data_r_mac123(fr_mac1, fr_mac2, fr_mac3) \
gte_mv_from_data_r(fr_mac1, C2_MAC1) \
, gte_mv_from_data_r(fr_mac2, C2_MAC2) \
, gte_mv_from_data_r(fr_mac3, C2_MAC3)
WORD_COUNT(mac_gte_mv_from_data_r_mac123, 3)
#define mac_gte_mv_from_mac123_v3s4(v) \
mac_gte_mv_from_data_r_mac123(v.x, v.y, v.z)
WORD_COUNT(mac_gte_mv_from_mac123_v3s4, 3)
/* atom_dbg_skip */
#define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \ #define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \
load_upper_i(reg_transfer, u4_hi(cmd)) \ mac_load_word_imm(reg_transfer, cmd) \
, or_i_self( reg_transfer, u4_lo(cmd)) /* load_upper_i(reg_transfer, cmd >> 16), // or_i_self( reg_transfer, cmd & 0xFFFF), */ \
, store_word( reg_transfer, reg_base, port) , store_word( reg_transfer, reg_base, port)
WORD_COUNT(mac_gcmd_push, 3) WORD_COUNT(mac_gcmd_push, 3)
@@ -249,6 +359,7 @@ WORD_COUNT(mac_pack_color_word, 3)
mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b)
WORD_COUNT(mac_format_f3_color, 3) WORD_COUNT(mac_format_f3_color, 3)
/* atom_dbg_skip */
#define mac_format_g4_color(r_prim_cursor, r0, g0, b0, r1, g1, b1, r2, g2, b2, r3, g3, b3) \ #define mac_format_g4_color(r_prim_cursor, r0, g0, b0, r1, g1, b1, r2, g2, b2, r3, g3, b3) \
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0) \ mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0) \
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1) \ , mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1) \
@@ -270,16 +381,16 @@ WORD_COUNT(mac_format_g4_color, 12)
WORD_COUNT(mac_insert_ot_tag, 11) WORD_COUNT(mac_insert_ot_tag, 11)
/* atom_dbg_skip */ /* atom_dbg_skip */
#define mac_pad_set_centered_axes(r_state, r_scratch) \ #define mac_pad_set_centered_axes(state, scratch) \
load_upper_i(r_scratch, (PadAxis_Centered_Word >> 16) & 0xFFFF) \ load_upper_i(scratch, (PadAxis_Centered >> 16) & 0xFFFF) \
, or_i_self( r_scratch, PadAxis_Centered_Word & 0xFFFF) \ , or_i_self( scratch, PadAxis_Centered & 0xFFFF) /* mac_load_word_imm(scratch, PadAxis_Centered), */ \
, store_word( r_scratch, r_state, O_(PadState,axes)) , store_word( scratch, state, O_(PadState,axes))
WORD_COUNT(mac_pad_set_centered_axes, 3) WORD_COUNT(mac_pad_set_centered_axes, 3)
/* atom_dbg_skip */ /* atom_dbg_skip */
#define mac_pad_set_id_byte(r_state, r_id, id_value) \ #define mac_pad_set_id_byte(state, r_id, id_value) \
add_ui( r_id, R_0, id_value) \ add_ui( r_id, R_0, id_value) \
, store_byte(r_id, r_state, O_(PadState,id)) , store_byte(r_id, state, O_(PadState,id))
WORD_COUNT(mac_pad_set_id_byte, 2) WORD_COUNT(mac_pad_set_id_byte, 2)
/* atom_dbg_skip */ /* atom_dbg_skip */
@@ -291,6 +402,6 @@ WORD_COUNT(mac_pad_set_status, 2)
/* atom_dbg_skip */ /* atom_dbg_skip */
#define mac_pad_store_inverted_buttons(r_buttons, r_pad_state) \ #define mac_pad_store_inverted_buttons(r_buttons, r_pad_state) \
nor_u( r_buttons, r_buttons, R_0) \ nor_u( r_buttons, r_buttons, R_0) \
, store_half( r_buttons, r_pad_state, O_(PadState, buttons)) , store_half(r_buttons, r_pad_state, O_(PadState,buttons))
WORD_COUNT(mac_pad_store_inverted_buttons, 2) WORD_COUNT(mac_pad_store_inverted_buttons, 2)
+10 -2
View File
@@ -14,7 +14,7 @@
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h // source: C:\projects\Pikuma\ps1\code\duffle\bios.h
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h // source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c // source: C:\projects\Pikuma\ps1\code\duffle\pad.c
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c // source: C:\projects\Pikuma\ps1\code\duffle\math.atom.h
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c // source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c // source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c // source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
@@ -25,7 +25,15 @@
#pragma region duffle #pragma region duffle
// --- atom: normalize_v3s4 (66 words) --- // --- atom: example_atom_proc (10 words) ---
#define _atom_offset_example_atom_proc_skip 2
enum {
atom_offset_example_atom_proc_skip = _atom_offset_example_atom_proc_skip,
};
// --- atom: build_normalize_v3s4 (67 words) ---
#define _atom_offset_aligned_done_srav_path 3 #define _atom_offset_aligned_done_srav_path 3
#define _atom_offset_srav_path_aligned_done 4 #define _atom_offset_srav_path_aligned_done 4
+9 -10
View File
@@ -9,36 +9,34 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(gp_atom_c);
#pragma region MACs (Mips Atom Components) #pragma region MACs (Mips Atom Components)
FI_ Slice_MipsCode ac_gcmd_push(AtomBuilder_R ab, U4 cmd, U4 reg_transfer, U4 reg_base, U2 port) FI_ Slice_MipsCode ac_gcmd_push(AtomBuilder_R ab, U4 cmd, U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_gcmd_push, ab, { atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_upper_i(reg_transfer, u4_hi(cmd)), mac_load_word_imm(reg_transfer, cmd),
or_i_self( reg_transfer, u4_lo(cmd)),
// load_upper_i(reg_transfer, cmd >> 16),
// or_i_self( reg_transfer, cmd & 0xFFFF),
store_word( reg_transfer, reg_base, port), store_word( reg_transfer, reg_base, port),
}) })
FI_ Slice_MipsCode ac_store_rgb8(AtomBuilder_R ab, U1 rr, U1 rg, U1 rb, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rgb8, ab, { FI_ Slice_MipsCode ac_store_rgb8(AtomBuilder_R ab, U1 rr, U1 rg, U1 rb, U4 base, U4 offset)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
store_byte(rr, base, offset + O_(RGB8,r)), store_byte(rr, base, offset + O_(RGB8,r)),
store_byte(rg, base, offset + O_(RGB8,g)), store_byte(rg, base, offset + O_(RGB8,g)),
store_byte(rb, base, offset + O_(RGB8,b)), store_byte(rb, base, offset + O_(RGB8,b)),
}) })
FI_ Slice_MipsCode ac_pack_color_word(AtomBuilder_R ab, U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b) FI_ Slice_MipsCode ac_pack_color_word(AtomBuilder_R ab, U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b)
atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, ab, { atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_upper_i(R_AT, (cmd) << 8 | (b)), load_upper_i(R_AT, (cmd) << 8 | (b)),
or_i_self( R_AT, ((g) << 8) | (r)), or_i_self( R_AT, ((g) << 8) | (r)),
store_word( R_AT, r_base, (off)), store_word( R_AT, r_base, (off)),
}) })
FI_ Slice_MipsCode ac_format_f3_color(AtomBuilder_R ab, U4 r_base, U1 r, U1 g, U1 b) FI_ Slice_MipsCode ac_format_f3_color(AtomBuilder_R ab, U4 r_base, U1 r, U1 g, U1 b)
atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, ab, { mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) }) atom_dbg_skip MipsAtomComp_Proc_(ab, { mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
FI_ Slice_MipsCode ac_format_g4_color(AtomBuilder_R ab, U4 r_prim_cursor, FI_ Slice_MipsCode ac_format_g4_color(AtomBuilder_R ab, U4 r_prim_cursor,
U1 r0, U1 g0, U1 b0, U1 r0, U1 g0, U1 b0,
U1 r1, U1 g1, U1 b1, U1 r1, U1 g1, U1 b1,
U1 r2, U1 g2, U1 b2, U1 r2, U1 g2, U1 b2,
U1 r3, U1 g3, U1 b3) U1 r3, U1 g3, U1 b3)
MipsAtomComp_Proc_(ac_format_g4_color, ab, { atom_dbg_skip MipsAtomComp_Proc_(ab, {
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0), mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0),
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1), mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1),
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c2), 0, r2,g2,b2), mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c2), 0, r2,g2,b2),
@@ -46,7 +44,8 @@ MipsAtomComp_Proc_(ac_format_g4_color, ab, {
}) })
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list. */ /* Words: 11; Correctly inserts a primitive into the Ordering Table linked list. */
I_ Slice_MipsCode ac_insert_ot_tag(AtomBuilder_R ab, U4 r_ot_base, U4 r_prim_cursor, U4 poly_size) MipsAtomComp_Proc_(ac_insert_ot_tag, ab, { // TODO(Ed): Expose R_T1 as a r_t0, r_V0 as r_t2
I_ Slice_MipsCode ac_insert_ot_tag(AtomBuilder_R ab, Reg r_ot_base, Reg r_prim_cursor, U2 poly_size) MipsAtomComp_Proc_(ab, {
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1) shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ] add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
+58 -61
View File
@@ -21,7 +21,7 @@
* 4. Semantic encoders gp0_word_poly_f3(r,g,b) * 4. Semantic encoders gp0_word_poly_f3(r,g,b)
* 3. Composite encoders enc_color_word(cmd, r, g, b) * 3. Composite encoders enc_color_word(cmd, r, g, b)
* 2. Per-field encoders enc_gp0_color_r(r), enc_gp0_color_g(g), ... * 2. Per-field encoders enc_gp0_color_r(r), enc_gp0_color_g(g), ...
* 1. Bitfield layout consts gp0_color_red_shift = 0, gp0_color_red_mask = 0xFF * 1. Bitfield layout consts gp0_color_red_shift = 0, gp0_color_red_width = 8
* 0. Opcode IDs gp0_cmd_poly_f3 = 0x20 * 0. Opcode IDs gp0_cmd_poly_f3 = 0x20
* *
* Vendor mnemonics (gte_mtc2, gte_mfc2, etc.) are NOT in this header. * Vendor mnemonics (gte_mtc2, gte_mfc2, etc.) are NOT in this header.
@@ -68,13 +68,14 @@ enum {
#define gp0_send(word) (HW_GP0[0] = (word)) #define gp0_send(word) (HW_GP0[0] = (word))
#define gp1_send(word) (HW_GP1[0] = (word)) #define gp1_send(word) (HW_GP1[0] = (word))
#define DmaSlot_ // Annotate an instruction as filling a CPU <-> Command DMA delay slot/s
/* ============================================================================ /* ============================================================================
* GP0 command byte constants + Layer 1 (GPU bitfield shifts) * GP0 command byte constants + Layer 1 (GPU bitfield shifts)
* ============================================================================ * ============================================================================
* 8-bit GP0 opcodes (the upper byte of a primitive's first word). These are the BYTE only. * 8-bit GP0 opcodes (the upper byte of a primitive's first word). These are the BYTE only.
* NO macro body past this point uses a raw shift or raw mask. * NO macro body past this point uses a raw shift or raw mask.
* Mirrors the OPCODE_SHIFT / RS_SHIFT / REG_MASK convention from mips.h. * Mirrors the OPCODE_SHIFT / RS_SHIFT convention from mips.h.
* ============================================================================ */ * ============================================================================ */
enum { enum {
gp0_cmd_Nop = 0x00, gp0_cmd_Nop = 0x00,
@@ -116,21 +117,20 @@ enum {
gp0_cmd_SetDrawOffset = 0xE5, gp0_cmd_SetDrawOffset = 0xE5,
gp0_cmd_SetMaskBit = 0xE6, gp0_cmd_SetMaskBit = 0xE6,
/* bitfield shifts / widths / masks ---- /* bitfield shifts / widths ----
* Generic GP0/GP1 command byte (upper 8 bits of every word sent to either port). */ * Generic GP0/GP1 command byte (upper 8 bits of every word sent to either port). */
gp0_cmd_shift = 24, gp0_cmd_shift = 24,
gp0_cmd_width = 8, gp0_cmd_width = 8,
gp0_cmd_mask = 0xFF,
/* Color word layout (lives in Poly_F3.color, Poly_G4.c0..c3, etc.): /* Color word layout (lives in Poly_F3.color, Poly_G4.c0..c3, etc.):
* bits 31..24 = command byte * bits 31..24 = command byte
* bits 23..16 = BLUE * bits 23..16 = BLUE
* bits 15..08 = GREEN * bits 15..08 = GREEN
* bits 07..00 = RED (PSX GPU is BGR, NOT RGB) */ * bits 07..00 = RED (PSX GPU is BGR, NOT RGB) */
gp0_color_cmd_shift = 24, gp0_color_cmd_width = 8, gp0_color_cmd_mask = 0xFF, gp0_color_cmd_shift = 24, gp0_color_cmd_width = 8,
gp0_color_blue_shift = 16, gp0_color_blue_width = 8, gp0_color_blue_mask = 0xFF, gp0_color_blue_shift = 16, gp0_color_blue_width = 8,
gp0_color_green_shift = 8, gp0_color_green_width = 8, gp0_color_green_mask = 0xFF, gp0_color_green_shift = 8, gp0_color_green_width = 8,
gp0_color_red_shift = 0, gp0_color_red_width = 8, gp0_color_red_mask = 0xFF, gp0_color_red_shift = 0, gp0_color_red_width = 8,
}; };
/* ============================================================================ /* ============================================================================
@@ -143,12 +143,12 @@ enum {
* ============================================================================ */ * ============================================================================ */
/* ---- Layer 1.5: per-field encoders ---- */ /* ---- Layer 1.5: per-field encoders ---- */
#define enc_gp0_cmd(cmd) (((cmd) & gp0_cmd_mask) << gp0_cmd_shift) #define enc_gp0_cmd(cmd) ((cmd) << gp0_cmd_shift)
#define enc_gp0_color_cmd(cmd) (((cmd) & gp0_color_cmd_mask) << gp0_color_cmd_shift) #define enc_gp0_color_cmd(cmd) ((cmd) << gp0_color_cmd_shift)
#define enc_gp0_color_r(r) (((r) & gp0_color_red_mask) << gp0_color_red_shift) #define enc_gp0_color_r(r) ((r) << gp0_color_red_shift)
#define enc_gp0_color_g(g) (((g) & gp0_color_green_mask) << gp0_color_green_shift) #define enc_gp0_color_g(g) ((g) << gp0_color_green_shift)
#define enc_gp0_color_b(b) (((b) & gp0_color_blue_mask) << gp0_color_blue_shift) #define enc_gp0_color_b(b) ((b) << gp0_color_blue_shift)
/* ---- Layer 2: composite encoders ---- */ /* ---- Layer 2: composite encoders ---- */
#define enc_color_word(cmd, r, g, b) (enc_gp0_color_cmd(cmd) | enc_gp0_color_r(r) | enc_gp0_color_g(g) | enc_gp0_color_b(b)) #define enc_color_word(cmd, r, g, b) (enc_gp0_color_cmd(cmd) | enc_gp0_color_r(r) | enc_gp0_color_g(g) | enc_gp0_color_b(b))
@@ -211,38 +211,38 @@ enum {
gp1_disp_Color24 = 0x1, gp1_disp_Color24 = 0x1,
gp1_disp_VInterlace = 0x1, gp1_disp_VInterlace = 0x1,
/* ---- Layer 1: GP1 display-mode + range + draw-area shifts/masks ---- */ /* ---- Layer 1: GP1 display-mode + range + draw-area shifts/widths ---- */
gp1_disp_hres_shift = 0, gp1_disp_hres_width = 2, gp1_disp_hres_mask = 0x3, gp1_disp_hres_shift = 0, gp1_disp_hres_width = 2,
gp1_disp_vres_shift = 2, gp1_disp_vres_width = 1, gp1_disp_vres_mask = 0x1, gp1_disp_vres_shift = 2, gp1_disp_vres_width = 1,
gp1_disp_color_shift = 4, gp1_disp_color_width = 1, gp1_disp_color_mask = 0x1, gp1_disp_color_shift = 4, gp1_disp_color_width = 1,
gp1_disp_interlace_shift = 5, gp1_disp_interlace_width = 1, gp1_disp_interlace_mask = 0x1, gp1_disp_interlace_shift = 5, gp1_disp_interlace_width = 1,
/* GP1 horizontal display range: bits 0..11 = X2, bits 12..23 = X1 */ /* GP1 horizontal display range: bits 0..11 = X2, bits 12..23 = X1 */
gp1_hrange_x1_shift = 12, gp1_hrange_x1_width = 12, gp1_hrange_x1_mask = 0xFFF, gp1_hrange_x1_shift = 12, gp1_hrange_x1_width = 12,
gp1_hrange_x2_shift = 0, gp1_hrange_x2_width = 12, gp1_hrange_x2_mask = 0xFFF, gp1_hrange_x2_shift = 0, gp1_hrange_x2_width = 12,
/* GP1 vertical display range: bits 0..9 = Y2, bits 10..19 = Y1 */ /* GP1 vertical display range: bits 0..9 = Y2, bits 10..19 = Y1 */
gp1_vrange_y1_shift = 10, gp1_vrange_y1_width = 10, gp1_vrange_y1_mask = 0x3FF, gp1_vrange_y1_shift = 10, gp1_vrange_y1_width = 10,
gp1_vrange_y2_shift = 0, gp1_vrange_y2_width = 10, gp1_vrange_y2_mask = 0x3FF, gp1_vrange_y2_shift = 0, gp1_vrange_y2_width = 10,
/* GP1 draw area (top-left or bottom-right): bits 0..9 = X, bits 10..19 = Y /* GP1 draw area (top-left or bottom-right): bits 0..9 = X, bits 10..19 = Y
* (10-bit signed — caller pre-signs and masks with the named mask) */ * (10-bit signed — caller pre-signs) */
gp1_draw_x_shift = 0, gp1_draw_x_width = 10, gp1_draw_x_mask = 0x3FF, gp1_draw_x_shift = 0, gp1_draw_x_width = 10,
gp1_draw_y_shift = 10, gp1_draw_y_width = 10, gp1_draw_y_mask = 0x3FF, gp1_draw_y_shift = 10, gp1_draw_y_width = 10,
}; };
/* ---- Layer 1.5: GP1 per-field encoders ---- */ /* ---- Layer 1.5: GP1 per-field encoders ---- */
#define enc_gp1_disp_hres(h) (((h) & gp1_disp_hres_mask) << gp1_disp_hres_shift) #define enc_gp1_disp_hres(h) ((h) << gp1_disp_hres_shift)
#define enc_gp1_disp_vres(v) (((v) & gp1_disp_vres_mask) << gp1_disp_vres_shift) #define enc_gp1_disp_vres(v) ((v) << gp1_disp_vres_shift)
#define enc_gp1_disp_color(c) (((c) & gp1_disp_color_mask) << gp1_disp_color_shift) #define enc_gp1_disp_color(c) ((c) << gp1_disp_color_shift)
#define enc_gp1_disp_interlace(i) (((i) & gp1_disp_interlace_mask) << gp1_disp_interlace_shift) #define enc_gp1_disp_interlace(i) ((i) << gp1_disp_interlace_shift)
#define enc_gp1_hrange_x1(x1) (((x1) & gp1_hrange_x1_mask) << gp1_hrange_x1_shift) #define enc_gp1_hrange_x1(x1) ((x1) << gp1_hrange_x1_shift)
#define enc_gp1_hrange_x2(x2) (((x2) & gp1_hrange_x2_mask) << gp1_hrange_x2_shift) #define enc_gp1_hrange_x2(x2) ((x2) << gp1_hrange_x2_shift)
#define enc_gp1_vrange_y1(y1) (((y1) & gp1_vrange_y1_mask) << gp1_vrange_y1_shift) #define enc_gp1_vrange_y1(y1) ((y1) << gp1_vrange_y1_shift)
#define enc_gp1_vrange_y2(y2) (((y2) & gp1_vrange_y2_mask) << gp1_vrange_y2_shift) #define enc_gp1_vrange_y2(y2) ((y2) << gp1_vrange_y2_shift)
#define enc_gp1_draw_x(x) (((x) & gp1_draw_x_mask) << gp1_draw_x_shift) #define enc_gp1_draw_x(x) ((x) << gp1_draw_x_shift)
#define enc_gp1_draw_y(y) (((y) & gp1_draw_y_mask) << gp1_draw_y_shift) #define enc_gp1_draw_y(y) ((y) << gp1_draw_y_shift)
/* ---- Layer 2: GP1 composite encoders ---- */ /* ---- Layer 2: GP1 composite encoders ---- */
#define enc_gp1_disp_mode_word(h, v, c, i) (enc_gp0_cmd(gp1_cmd_DisplayMode) | enc_gp1_disp_hres(h) | enc_gp1_disp_vres(v) | enc_gp1_disp_color(c) | enc_gp1_disp_interlace(i)) #define enc_gp1_disp_mode_word(h, v, c, i) (enc_gp0_cmd(gp1_cmd_DisplayMode) | enc_gp1_disp_hres(h) | enc_gp1_disp_vres(v) | enc_gp1_disp_color(c) | enc_gp1_disp_interlace(i))
@@ -419,14 +419,11 @@ typedef Struct_(PolyTag) {
}; };
}; };
/* DSL cast convention: every cast uses `C_()`, every pointer qualifier is `R_` (restrict) or `V_` (volatile).
* No raw C-style casts. RHS values are assumed to be `U4` — caller passes a `U4` directly. */
#define set_len(tag,v) (C_(PolyTag_R,tag)->len = u4_(v)) #define set_len(tag,v) (C_(PolyTag_R,tag)->len = u4_(v))
#define set_addr(tag,v) (C_(PolyTag_R,tag)->addr = u4_(v)) #define set_addr(tag,v) (C_(PolyTag_R,tag)->addr = u4_(v))
/* `set_code` is no longer in the new PolyTag design — the code byte lives in the primitive body /* `set_code` is no longer in the new PolyTag design
* (e.g. `((Poly_F3*)(p))->code`), not in the tag. * (e.g. `((Poly_F3*)(p))->code`), not in the tag.
* Use the typed primitive structs (Poly_F3, Poly_G4, etc.) and the `set_poly_*` setters, * Use the typed primitive structs (Poly_F3, Poly_G4, etc.) and the `set_poly_*` setters, which set both the tag's length and the code. */
* which set both the tag's length and the code. */
#define get_len(tag) C_(U4,C_(PolyTag_R,tag)->len) #define get_len(tag) C_(U4,C_(PolyTag_R,tag)->len)
#define get_addr(tag) C_(U4,C_(PolyTag_R,tag)->addr) #define get_addr(tag) C_(U4,C_(PolyTag_R,tag)->addr)
@@ -555,14 +552,14 @@ typedef Struct_(Poly_GT4) {
* bits 12..31 = reserved (zero) * bits 12..31 = reserved (zero)
* ============================================================================ */ * ============================================================================ */
enum { enum {
/* ---- Layer 1: TPage bitfield shifts / widths / masks ---- */ /* ---- Layer 1: TPage bitfield shifts / widths ---- */
gp0_tpage_x_shift = 0, gp0_tpage_x_width = 4, gp0_tpage_x_mask = 0xF, gp0_tpage_x_shift = 0, gp0_tpage_x_width = 4,
gp0_tpage_y_shift = 4, gp0_tpage_y_width = 1, gp0_tpage_y_mask = 0x1, gp0_tpage_y_shift = 4, gp0_tpage_y_width = 1,
gp0_tpage_semi_trans_shift = 5, gp0_tpage_semi_trans_width = 2, gp0_tpage_semi_trans_mask = 0x3, gp0_tpage_semi_trans_shift = 5, gp0_tpage_semi_trans_width = 2,
gp0_tpage_color_depth_shift = 7, gp0_tpage_color_depth_width = 2, gp0_tpage_color_depth_mask = 0x3, gp0_tpage_color_depth_shift = 7, gp0_tpage_color_depth_width = 2,
gp0_tpage_dither_shift = 9, gp0_tpage_dither_width = 1, gp0_tpage_dither_mask = 0x1, gp0_tpage_dither_shift = 9, gp0_tpage_dither_width = 1,
gp0_tpage_draw_to_disp_shift = 10, gp0_tpage_draw_to_disp_width = 1, gp0_tpage_draw_to_disp_mask = 0x1, gp0_tpage_draw_to_disp_shift = 10, gp0_tpage_draw_to_disp_width = 1,
gp0_tpage_tex_disable_shift = 11, gp0_tpage_tex_disable_width = 1, gp0_tpage_tex_disable_mask = 0x1, gp0_tpage_tex_disable_shift = 11, gp0_tpage_tex_disable_width = 1,
/* TPage color-depth payload values (NOT bit positions — these go in /* TPage color-depth payload values (NOT bit positions — these go in
* the 2-bit field at gp0_tpage_color_depth_shift). */ * the 2-bit field at gp0_tpage_color_depth_shift). */
@@ -573,7 +570,7 @@ enum {
/* Default TPage value libpsyx's SetDefDrawEnv writes (matches the `li v1, 10; sh v1, 20(v0)` sequence at C11_only.elf:0x8001273C). */ /* Default TPage value libpsyx's SetDefDrawEnv writes (matches the `li v1, 10; sh v1, 20(v0)` sequence at C11_only.elf:0x8001273C). */
gp0_tpage_default = 10, gp0_tpage_default = 10,
/* TPage semi-transparency mode payload values (NOT bit positions). */ /* TPage semi-transparency mode payload values. */
gp0_tpage_semi_trans_none = 0x0, gp0_tpage_semi_trans_none = 0x0,
gp0_tpage_semi_trans_alpha = 0x1, gp0_tpage_semi_trans_alpha = 0x1,
gp0_tpage_semi_trans_add = 0x2, gp0_tpage_semi_trans_add = 0x2,
@@ -581,13 +578,13 @@ enum {
}; };
/* ---- Layer 1.5: TPage per-field encoders. Mirrors enc_gte_sf/mx/v in gte.h. ---- */ /* ---- Layer 1.5: TPage per-field encoders. Mirrors enc_gte_sf/mx/v in gte.h. ---- */
#define enc_gp0_tpage_x(x) (((x) & gp0_tpage_x_mask) << gp0_tpage_x_shift) #define enc_gp0_tpage_x(x) ((x) << gp0_tpage_x_shift)
#define enc_gp0_tpage_y(y) (((y) & gp0_tpage_y_mask) << gp0_tpage_y_shift) #define enc_gp0_tpage_y(y) ((y) << gp0_tpage_y_shift)
#define enc_gp0_tpage_semi_trans(s) (((s) & gp0_tpage_semi_trans_mask) << gp0_tpage_semi_trans_shift) #define enc_gp0_tpage_semi_trans(s) ((s) << gp0_tpage_semi_trans_shift)
#define enc_gp0_tpage_color_depth(c) (((c) & gp0_tpage_color_depth_mask) << gp0_tpage_color_depth_shift) #define enc_gp0_tpage_color_depth(c) ((c) << gp0_tpage_color_depth_shift)
#define enc_gp0_tpage_dither(d) (((d) & gp0_tpage_dither_mask) << gp0_tpage_dither_shift) #define enc_gp0_tpage_dither(d) ((d) << gp0_tpage_dither_shift)
#define enc_gp0_tpage_draw_to_disp(d) (((d) & gp0_tpage_draw_to_disp_mask) << gp0_tpage_draw_to_disp_shift) #define enc_gp0_tpage_draw_to_disp(d) ((d) << gp0_tpage_draw_to_disp_shift)
#define enc_gp0_tpage_tex_disable(t) (((t) & gp0_tpage_tex_disable_mask) << gp0_tpage_tex_disable_shift) #define enc_gp0_tpage_tex_disable(t) ((t) << gp0_tpage_tex_disable_shift)
/* ---- Layer 2: TPage composite encoder. Mirrors enc_gte_cmdw in gte.h ---- */ /* ---- Layer 2: TPage composite encoder. Mirrors enc_gte_cmdw in gte.h ---- */
#define enc_gp0_tpage_word(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \ #define enc_gp0_tpage_word(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
@@ -617,17 +614,17 @@ typedef Struct_(TexturePage) { U4 raw; };
* bits 24..31 = command byte — 0x20 (4bpp load) or 0x25 (8bpp load) * bits 24..31 = command byte — 0x20 (4bpp load) or 0x25 (8bpp load)
* ============================================================================ */ * ============================================================================ */
enum { enum {
/* ---- Layer 1: CLUT bitfield shifts / widths / masks ---- */ /* ---- Layer 1: CLUT bitfield shifts / widths ---- */
gp0_clut_y_shift = 0, gp0_clut_y_width = 6, gp0_clut_y_mask = 0x3F, gp0_clut_y_shift = 0, gp0_clut_y_width = 6,
gp0_clut_x_shift = 6, gp0_clut_x_width = 9, gp0_clut_x_mask = 0x1FF, gp0_clut_x_shift = 6, gp0_clut_x_width = 9,
/* CLUT-load cmd-byte variants — the upper byte of the GP0 word. */ /* CLUT-load cmd-byte variants — the upper byte of the GP0 word. */
gp0_clut_cmd_Load4bpp = 0x20, gp0_clut_cmd_Load4bpp = 0x20,
gp0_clut_cmd_Load8bpp = 0x25, gp0_clut_cmd_Load8bpp = 0x25,
}; };
/* ---- Layer 1.5: CLUT per-field encoders ---- */ /* ---- Layer 1.5: CLUT per-field encoders ---- */
#define enc_gp0_clut_x(x) (((x) & gp0_clut_x_mask) << gp0_clut_x_shift) #define enc_gp0_clut_x(x) ((x) << gp0_clut_x_shift)
#define enc_gp0_clut_y(y) (((y) & gp0_clut_y_mask) << gp0_clut_y_shift) #define enc_gp0_clut_y(y) ((y) << gp0_clut_y_shift)
/* ---- Layer 2: CLUT composite encoder ---- */ /* ---- Layer 2: CLUT composite encoder ---- */
#define enc_gp0_clut_word(cmd, x, y) (enc_gp0_cmd(cmd) | enc_gp0_clut_x(x) | enc_gp0_clut_y(y)) #define enc_gp0_clut_word(cmd, x, y) (enc_gp0_cmd(cmd) | enc_gp0_clut_x(x) | enc_gp0_clut_y(y))
+238 -204
View File
@@ -11,25 +11,62 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(gte_atom_c);
#pragma region MACs (Mips Atom Components) #pragma region MACs (Mips Atom Components)
/* Words: 3; Loads 3 S2 indices from the face array */ /* Words: 3; Loads 3 S2 indices from the face array */
FI_ Slice_MipsCode ac_load_tri_indices(AtomBuilder_R ab, U4 r_face_cusor, U4 r_i0, U4 r_i1, U4 r_i2) atom_dbg_skip MipsAtomComp_Proc_(ac_load_tri_indices, ab, { FI_ Slice_MipsCode ac_load_tri_indices(AtomBuilder_R ab, U4 r_face_cusor, U4 r_i0, U4 r_i1, U4 r_i2)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)), load_half_u(r_i0, r_face_cusor, 0 * S_(S2)),
load_half_u(r_i1, r_face_cusor, 1 * S_(S2)), load_half_u(r_i1, r_face_cusor, 1 * S_(S2)),
load_half_u(r_i2, r_face_cusor, 2 * S_(S2)), load_half_u(r_i2, r_face_cusor, 2 * S_(S2)),
}) })
FI_ Slice_MipsCode ac_gte_mv_to_cr_diag_v3s4(AtomBuilder_R ab, Reg_(V3_S4) v) MipsAtomComp_Proc_(ab, {
gte_mv_to_ctrl_r(v.y, gte_cr_RT13),
gte_mv_to_ctrl_r(v.z, gte_cr_RT22),
gte_mv_to_ctrl_r(v.x, gte_cr_RT11),
})
FI_ Slice_MipsCode ac_gte_ld_ir123_v3s4(AtomBuilder_R ab, Reg_(V3_S4) v) MipsAtomComp_Proc_(ab, {
gte_mv_to_data_r(v.x, C2_IR1),
gte_mv_to_data_r(v.y, C2_IR2),
gte_mv_to_data_r(v.z, C2_IR3),
})
/* ─── GTE OP cross product (a × b → a) ───
* Sets up RT diagonal from a.xyz, IR1/2/3 from b.xyz, fires OP,
* reads MAC1/2/3, shifts right 12 (S12.20 → S12.0 OuterProduct12), writes back to a.xyz.
* Composes the three sub-primitives (RT-load, IR-load, OP, MAC-read, shift)
* into one component for use by atoms that need the cross product inline.
*
* Output gpr (a) aliases source-A gpr; MAC read clobbers source-A's load targets,
* but by that point the RT load is complete and source A is dead.
* Pipeline: clobbers IR1..3, MAC1..3, RT11..33.
*
* The CPU→COP2 transfer chains (3 ctc2, 3 mtc2) require a 2-slot retirement gap,
* and the MFC2→GPR chain (3 mfc2) requires a 1-slot retirement gap, before the GPR can be read.
* The hazard nops are inlined below — same convention as ac_gte_gpf_scale — so any atom body inlining this component inherits them.
*
* Words: 18 (3 ctc2 + 2 nop + 3 mtc2 + 2 nop + 1 op + 3 mfc2 + 1 nop + 3 sra).
*/
FI_ Slice_MipsCode ac_gte_op_cross_v3s4(AtomBuilder_R ab, Reg_(V3_S4) a, Reg_(V3_S4) b) atom_dbg_skip MipsAtomComp_Proc_(ab, {
mac_gte_mv_to_cr_diag_v3s4(a), GteDelay_ /* RT diagonal: D1 = a.x, D2 = a.y, D3 = a.z */
mac_gte_ld_ir123_v3s4(b), GteDelay_ /* IR: second operand (b.xyz) */
gte_cmdw_cross, /* OP: MAC1/2/3 = a × b (S12.20) */
mac_gte_mv_from_mac123_v3s4(a), GteDelay_ /* Read MAC1/2/3 → a.xyz (overwrites source-A's load targets) */
mac_shift_aright_v3s4_self(a, 12), /* Right-shift MAC by 12 (S12.20 → S12.0 OuterProduct12) */
})
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3. /* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */ * PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
FI_ Slice_MipsCode ac_gte_store_f3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_f3, ab, { FI_ Slice_MipsCode ac_gte_store_f3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)), gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)),
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_F3,p1)), gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_F3,p1)),
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_F3,p2)), gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_F3,p2)),
}) })
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */ /* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_load_tri_verts, ab, { I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ab, {
shift_lleft(R_AT, r_v0, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0), shift_lleft(R_AT, r_v0, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
shift_lleft(R_AT, r_v1, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1), shift_lleft(R_AT, r_v1, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
shift_lleft(R_AT, r_v2, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2), shift_lleft(R_AT, r_v2, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
}) })
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the /* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
@@ -37,7 +74,7 @@ I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, U4 r_vert_base, U4 r_v
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). * PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3 * MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
* (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */ * (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
FI_ Slice_MipsCode ac_gte_store_g4_p012(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p012, ab, { FI_ Slice_MipsCode ac_gte_store_g4_p012(AtomBuilder_R ab, Reg r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)), gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)),
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)), gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)),
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2)), gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2)),
@@ -47,33 +84,41 @@ FI_ Slice_MipsCode ac_gte_store_g4_p012(AtomBuilder_R ab, U4 r_primitive_cursor)
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2; * PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2;
* SXY0 still holds v0.screen from the earlier RTPT. * SXY0 still holds v0.screen from the earlier RTPT.
*/ */
FI_ Slice_MipsCode ac_gte_store_g4_p3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p3, ab, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) }) FI_ Slice_MipsCode ac_gte_store_g4_p3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) })
/* ─── STAGE 1 of normalize: SQR + mfc2 MAC1/2/3 ─── /* ─── STAGE 1 of normalize: SQR + mfc2 MAC1/2/3 ───
* Emits squared magnitude per component (in MAC1/2/3) into caller-provided scratch regs. * Emits squared magnitude per component (in MAC1/2/3) into caller-provided scratch regs. */
* Stage 2 of normalize consumes these directly. FI_ Slice_MipsCode ac_gte_sqr_v3(AtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_sq_x, U4 r_sq_y, U4 r_sq_z) atom_dbg_skip MipsAtomComp_Proc_(ab, {
* Words: 8. Clobbers: IR1/2/3, MAC1/2/3. Uses gte_cmdw_sqr (sf=0, lm=1). */ mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop),
FI_ Slice_MipsCode ac_gte_sqr_v3(AtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_sq_x, U4 r_sq_y, U4 r_sq_z) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_sqr_v3, ab, {
gte_mv_to_data_r(r_sx, C2_IR1),
gte_mv_to_data_r(r_sy, C2_IR2),
gte_mv_to_data_r(r_sz, C2_IR3),
nop, gte_cmdw_sqr,
gte_mv_from_data_r(r_sq_x, C2_MAC1), gte_mv_from_data_r(r_sq_x, C2_MAC1),
gte_mv_from_data_r(r_sq_y, C2_MAC2), gte_mv_from_data_r(r_sq_y, C2_MAC2),
gte_mv_from_data_r(r_sq_z, C2_MAC3), gte_mv_from_data_r(r_sq_z, C2_MAC3),
}) })
/* ─── SQR FIRE — mtc2 3 GPRs into IR1/IR2/IR3, then fire SQR. ─── */
FI_ Slice_MipsCode ac_gte_sqr_v3s4(AtomBuilder_R ab, Reg r_sx, Reg r_sy, Reg r_sz, MipsCode delay_slot)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
gte_mv_to_data_r(r_sx, C2_IR1),
gte_mv_to_data_r(r_sy, C2_IR2),
gte_mv_to_data_r(r_sz, C2_IR3),
delay_slot, gte_cmdw_sqr,
})
/* ─── STAGE 4 of normalize: mtc2 IR0..3 + GPF + mfc2 MAC + srav finalize ─── /* ─── STAGE 4 of normalize: mtc2 IR0..3 + GPF + mfc2 MAC + srav finalize ───
* Reusable standalone — given an IR0 = 1/|v| estimate (typically from a sqrtbl lookup) and a shift count * Reusable standalone — given an IR0 = 1/|v| estimate (typically from a sqrtbl lookup) and a shift count
* (typically (31 - LZCR)/2), multiplies IR0*IR[i] via GPF and shifts right to produce the normalized output. * (typically (31 - LZCR)/2), multiplies IR0*IR[i] via GPF and shifts right to produce the normalized output.
* Used standalone for "scale vector by scalar". * Used standalone for "scale vector by scalar".
* Words: 11. Clobbers: IR0..3, MAC1..3. Uses gte_cmdw_gpf (sf=0, lm=0). */ * Words: 11. Clobbers: IR0..3, MAC1..3. Uses gte_cmdw_gpf (sf=0, lm=0). */
FI_ Slice_MipsCode ac_gte_gpf_scale(AtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_recip_est, U4 r_shift, U4 r_dx, U4 r_dy, U4 r_dz) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_gpf_scale, ab, { FI_ Slice_MipsCode ac_gte_gpf_scale(AtomBuilder_R ab,
U4 r_sx, U4 r_sy, U4 r_sz,
U4 r_recip_est, U4 r_shift,
U4 r_dx, U4 r_dy, U4 r_dz)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
gte_mv_to_data_r(r_recip_est, C2_IR0), gte_mv_to_data_r(r_recip_est, C2_IR0),
gte_mv_to_data_r(r_sx, C2_IR1), gte_mv_to_data_r(r_sx, C2_IR1),
gte_mv_to_data_r(r_sy, C2_IR2), gte_mv_to_data_r(r_sy, C2_IR2),
gte_mv_to_data_r(r_sz, C2_IR3), gte_mv_to_data_r(r_sz, C2_IR3),
nop2, /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */ GteDelay_ nop2, /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */
gte_cmdw_gpf, gte_cmdw_gpf,
gte_mv_from_data_r(r_dx, C2_MAC1), gte_mv_from_data_r(r_dx, C2_MAC1),
gte_mv_from_data_r(r_dy, C2_MAC2), gte_mv_from_data_r(r_dy, C2_MAC2),
@@ -83,94 +128,82 @@ FI_ Slice_MipsCode ac_gte_gpf_scale(AtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz,
shift_aright_var(r_dz, r_dz, r_shift), shift_aright_var(r_dz, r_dz, r_shift),
}) })
/* ─── APPLY MATRIX LV (libgte ApplyMatrixLV port) ───
* Atom component — auto-generates mac_apply_matrix_lv Mac composer macro.
* Uses GTE RTPS (cv=1, sf=1, v=0) with lwc2-loaded V0/VZ0 inputs.
* Per PSX-SPX `geometrytransformationenginegte.md` lines 416-418:
* IR1 = MAC1 = (TRX*1000h + RT11*VX0 + RT12*VY0 + RT13*VZ0) SAR (sf*12)
* IR2 = MAC2 = (TRY*1000h + RT21*VX0 + RT22*VY0 + RT23*VZ0) SAR (sf*12)
* IR3 = MAC3 = (TRZ*1000h + RT31*VX0 + RT32*VY0 + RT33*VZ0) SAR (sf*12)
* RTPS uses the FULL row of the rotation matrix (not just diagonal like MVMVA with mx=0).
* libgte's `gte_ApplyMatrix` calls `gte_rtv0()` = RTPS cv=1 v=0 mx=0.
* Per `gte.h` line 405 the body sets cv=3 (BK, zero-initialized) so no TR contribution.
*
* Operands:
* r_mtx : MT3_S2S4* (matrix pointer)
* r_vec : U4 (pointer to PACKED V0 data — (pos.y << 16) | pos.x at +0, pos.z at +4)
* r_out : V3_S4* (output pointer; MAC1/2/3 stored here)
* r_t0/1/2 : 3 GPR codes for matrix load + intermediate state
* Words: ~26. Clobbers: r_t0, r_t1, r_t2 (C2 $0..$4, VXY0/VZ0, MAC1/2/3, SXY0/1/2). */
FI_ Slice_MipsCode ac_apply_matrix_lv(AtomBuilder_R ab
, U4 r_mtx, U4 r_vec, U4 r_out
, U4 r_t0, U4 r_t1, U4 r_t2
) MipsAtomComp_Proc_(ac_apply_matrix_lv, ab, {
/* Load MATRIX rows into GTE RT11..RT33 (libgte convention: ctc2 to C2 $0..$4 in order).
* load_half_u zero-extends the last word so RT33 = m[2][2] and TRX = 0. */
load_word(r_t0, r_mtx, 0), nop,
gte_mv_to_ctrl_r(r_t0, gte_cr_RT11_Code),
load_word(r_t0, r_mtx, 4), nop,
gte_mv_to_ctrl_r(r_t0, gte_cr_RT12_Code),
load_word(r_t0, r_mtx, 8), nop,
gte_mv_to_ctrl_r(r_t0, gte_cr_RT13_Code),
load_word(r_t0, r_mtx, 12), nop,
gte_mv_to_ctrl_r(r_t0, gte_cr_RT21_Code),
load_half_u(r_t0, r_mtx, 16), nop,
gte_mv_to_ctrl_r(r_t0, gte_cr_RT22_Code),
nop2,
/* Load PACKED pos into V0 (libgte SVECTOR layout).
* r_vec points to atom-0-staged packed data ((pos.y << 16) | pos.x at +0, pos.z at +4).
* LWC2 base register MUST be the pointer r_vec, NOT the loaded value r_t0. */
load_word(r_t0, r_vec, 0), nop,
gte_lw(C2_VXY0, r_vec, 0),
load_word(r_t0, r_vec, 4), nop,
gte_lw(C2_VZ0, r_vec, 4),
/* RTPS: cv=3 (no translation), sf=1 (no shift, integer), v=0 (V0 input),
* mx=0 (rotation matrix). MAC = RT row · V0 + 0. RTPS also writes
* SXY0/1/2 + SZ0..SZ3 (perspective division); ignored. */
gte_cmdw_rtps_sf1,
/* Read MAC1/2/3 → out. */
gte_mv_from_data_r(r_t0, C2_MAC1),
gte_mv_from_data_r(r_t1, C2_MAC2),
gte_mv_from_data_r(r_t2, C2_MAC3),
nop,
store_word(r_t0, r_out, 0),
store_word(r_t1, r_out, 4),
store_word(r_t2, r_out, 8),
})
/* ─── TRANS MATRIX (libgte TransMatrix port) ─── /* ─── TRANS MATRIX (libgte TransMatrix port) ───
* Atom component — auto-generates mac_trans_matrix Mac composer macro. * Atom component — auto-generates mac_trans_matrix Mac composer macro.
* m->t = v (struct copy; libgte's TransMatrix at 0x8001a540 is just 3 store_words, no GTE, no add). * m->t = v (struct copy; libgte's TransMatrix at 0x8001a540 is just 3 store_words, no GTE, no add).
* Uses 1 GPR (r_t1 = off value) per axis; per-axis load-delay-slot pattern. * Uses 1 GPR (r_t1 = off value) per axis; per-axis load-delay-slot pattern.
* Words: 9. Clobbers: r_t1. */ * Words: 9. Clobbers: r_t1. */
FI_ Slice_MipsCode ac_trans_matrix(AtomBuilder_R ab FI_ Slice_MipsCode ac_trans_mt3s3s4(AtomBuilder_R ab
, U4 r_mtx, U4 r_off , U4 r_mtx, U4 r_off
, U4 r_t1 , U4 r_t0, U4 r_t1, U4 r_t2
) MipsAtomComp_Proc_(ac_trans_matrix, ab, { ) MipsAtomComp_Proc_(ab, {
load_word(r_t1, r_off, O_(V3_S4,x)), load_word( r_t0, r_off, O_(V3_S4,x)),
nop, load_word( r_t1, r_off, O_(V3_S4,y)),
store_word(r_t1, r_mtx, O_(MT3_S2S4,t[0])), load_word( r_t2, r_off, O_(V3_S4,z)),
store_word(r_t0, r_mtx, O_(MT3_S2S4,t[0])),
load_word(r_t1, r_off, O_(V3_S4,y)),
nop,
store_word(r_t1, r_mtx, O_(MT3_S2S4,t[1])), store_word(r_t1, r_mtx, O_(MT3_S2S4,t[1])),
store_word(r_t2, r_mtx, O_(MT3_S2S4,t[2])),
load_word(r_t1, r_off, O_(V3_S4,z)),
nop,
store_word(r_t1, r_mtx, O_(MT3_S2S4,t[2])),
}) })
/* ─── LZCR ROUND EVEN + HALF-SHIFT ───
* Takes the raw LZCR leading-zero/ones count (from mfc2 C2_LZCR, range 1..32 per PSX-SPX cop2r31) and the |v|² sum (in r_mag_sq from the MAC1+MAC2+MAC3 add).
* Produces:
* r_shift ← LZCR rounded down to even (clear bit 0)
* r_mag_sq_copy ← |v|² sum (moved out of r_mag_sq before it's overwritten)
* r_mag_sq ← (31 - even_LZCR) / 2 = the final srav/GPF shift amount
*
* Rounding to even ensures (31 - LZCR) is always odd, so the >> 1 division is consistent — no 0.5 loss.
* The caller branches on LZCR < 24 to decide left-shift vs right-shift of r_mag_sq_copy, then saves the shift count.
*
* Note: C2_LZCR (cop2r31) is a fixed read-only C2 data register — the caller must read it via mfc2 from C2_LZCR;
* there is no register choice at the hardware level. Only the GPR that holds the result is caller-determined. */
FI_ Slice_MipsCode ac_lzcr_round_even_half_shift(AtomBuilder_R ab,
U4 r_shift,
U4 r_mag_sq,
U4 r_mag_sq_copy)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
and_i(r_shift, r_shift, gte_lzcr_even_mask),
or_u(r_mag_sq_copy, r_mag_sq, 0),
li_s( r_mag_sq, 31),
sub_s( r_mag_sq, r_mag_sq, r_shift),
shift_aright(r_mag_sq, r_mag_sq, 1),
})
FI_ Slice_MipsCode ac_gte_general_purpose_interopolation(AtomBuilder_R ab
, Reg to_ir0, Reg to_ir1, Reg to_ir2, Reg to_ir3
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3
, MipsCode nop_slot1, MipsCode nop_slot2)
MipsAtomComp_Proc_(ab, {
gte_mv_to_data_r(to_ir0, C2_IR0),
gte_mv_to_data_r(to_ir1, C2_IR1), /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */
gte_mv_to_data_r(to_ir2, C2_IR2),
gte_mv_to_data_r(to_ir3, C2_IR3), /* IR3 = src.z (reloaded) */
GteDelay_ nop_slot1,
GteDelay_ nop_slot2,
gte_cmdw_gpf,
gte_mv_from_data_r(fr_mac1, C2_MAC1),
gte_mv_from_data_r(fr_mac2, C2_MAC2),
gte_mv_from_data_r(fr_mac3, C2_MAC3),
})
FI_ Slice_MipsCode ac_gte_mv_from_data_r_mac123(AtomBuilder_R ab
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3)
MipsAtomComp_Proc_(ab, {
gte_mv_from_data_r(fr_mac1, C2_MAC1),
gte_mv_from_data_r(fr_mac2, C2_MAC2),
gte_mv_from_data_r(fr_mac3, C2_MAC3),
})
FI_ Slice_MipsCode ac_gte_mv_from_mac123_v3s4(AtomBuilder_R ab, Reg_(V3_S4) v) MipsAtomComp_ProcMap_(ab, mac_gte_mv_from_data_r_mac123(v.x, v.y, v.z))
#pragma endregion MACs (Mips Atom Components) #pragma endregion MACs (Mips Atom Components)
#pragma region Atom Procs #pragma region Atom Procs
/* ─── Local copy of PSYQ's sqrtbl (1/sqrt lookup table for VectorNormal). ─── /* ─── Local copy of PSYQ's sqrtbl (1/sqrt lookup table for VectorNormal). ───
* Source: PSYQ 4.7 libgte sqrtbl at 0x800185B4 in hello_camera.elf. * Source: PSYQ 4.7 libgte sqrtbl at 0x800185B4 in hello_camera.elf.
* objdump -s --start-address=0x800185B4 --stop-address=0x800185F4 hello_camera.elf * objdump -s --start-address=0x800185B4 --stop-address=0x800185F4 hello_camera.elf → 192 entries × 16-bit signed, in 1.12 fixed-point (max value 0x1000 = 1.0).
* → 192 entries × 16-bit signed, in 1.12 fixed-point (max value 0x1000 = 1.0).
* *
* Data is identical to the libgte original (byte-for-byte verified). * Data is identical to the libgte original (byte-for-byte verified).
* *
@@ -205,8 +238,9 @@ FI_ Slice_MipsCode ac_trans_matrix(AtomBuilder_R ab
* and the load upper_halves of the table bracket the input range. * and the load upper_halves of the table bracket the input range.
* The later 64 entries (octaves 2-3) are the `srav` branch when the magnitude's top bit is well above bit 24. * The later 64 entries (octaves 2-3) are the `srav` branch when the magnitude's top bit is well above bit 24.
* *
* 192-entry table is reproduced verbatim from libgte (verified against libpsn00b/psxgte/vector.s:100-123 — 24 rows × 8 halfwords, last entry 0x0804). */ * Reproduced verbatim from libgte (verified against libpsn00b/psxgte/vector.s:100-123 — 24 rows × 8 halfwords, last entry 0x0804).
internal RO_ S2 gte_normalize_sqr_tbl[192] align_(2) = { * */
internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
0x1000, 0x0fe0, 0x0fc1, 0x0fa3, 0x0f85, 0x0f68, 0x0f4c, 0x0f30, 0x1000, 0x0fe0, 0x0fc1, 0x0fa3, 0x0f85, 0x0f68, 0x0f4c, 0x0f30,
0x0f15, 0x0efb, 0x0ee1, 0x0ec7, 0x0eae, 0x0e96, 0x0e7e, 0x0e66, 0x0f15, 0x0efb, 0x0ee1, 0x0ec7, 0x0eae, 0x0e96, 0x0e7e, 0x0e66,
0x0e4f, 0x0e38, 0x0e22, 0x0e0c, 0x0df7, 0x0de2, 0x0dcd, 0x0db9, 0x0e4f, 0x0e38, 0x0e22, 0x0e0c, 0x0df7, 0x0de2, 0x0dcd, 0x0db9,
@@ -233,130 +267,120 @@ internal RO_ S2 gte_normalize_sqr_tbl[192] align_(2) = {
0x0820, 0x081c, 0x0818, 0x0814, 0x0810, 0x080c, 0x0808, 0x0804, 0x0820, 0x081c, 0x0818, 0x0814, 0x0810, 0x080c, 0x0808, 0x0804,
}; };
typedef Struct_(Binds_NormalizeV3S4) {
U2 src_offset; /* offset of src V3_S4 within the BIOS scratchpad */
U2 dst_offset; /* offset of dst V3_S4 within the BIOS scratchpad */
};
typedef Struct_(RegUse_build_normalize_v3s4) {
Reg scratch; /* scratchpad base; loaded via load_word_imm below. */
Reg src_ptr;
Reg dst_ptr;
Reg recip_est; /* |v|² sum + shift-input + sqrtbl[index] */
Reg norm; Reg shift;
Reg src_x;
union { Reg mac1_scratch, dst_offset; } t3;
union { Reg mac2_scratch; } t4;
union { Reg btarget, shift_count, lookup_addr, src_z, src_offset; } t5;
};
/* ─── Full normalize (all 4 stages inline) ─── /* ─── Full normalize (all 4 stages inline) ───
* Generic 4-stage GTE normalize (SQR → sum+LZCR → align+sqrtbl → GPF+srav). * Generic 4-stage GTE normalize (SQR → sum+LZCR → align+sqrtbl → GPF+srav). */
* internal MipsAtom* build_normalize_v3s4(AtomArena_R aa, RegUse_build_normalize_v3s4 r)
* Parameterized by caller-provided scratch base + src/dst offsets. MipsAtom_Proc_(aa, {
* The caller passes r_src_offset and r_dst_offset as compile-time constants /* Load scratch base via immediate (always Scratchpad_Loc = 0x1F800000 — the BIOS
* (typically derived from O_ macros in the caller's struct schema, e.g., `O_(CallerBundleScratch, fwd)`). * scratchpad, aliased by every consumer's ResolveLookAtScratch struct). */
* mac_load_word_imm(r.scratch, Scratchpad_Loc),
* This design lets any caller (with a scratch base + struct schema) use `normalize_v3s4_proc` /* Tape pop: src_offset, dst_offset = 4 bytes (packed into 1 U4: low16=src, high16=dst).
* without putting magic offsets in the C-side bundle helper — the offsets come from O_ macros at the call site. * Loads back-to-back fill each other's load-delay slots; the subsequent add_u
* * (2 cycles after the matching load) sees a valid value. */
* Body uses 9 GPRs (r_src_ptr..r_branch_tmp): load_half(r.t5.src_offset, R_TapePtr, O_(Binds_NormalizeV3S4, src_offset)),
* r_src_ptr, r_dst_ptr : src/dst pointers (computed from r_scratch + caller offsets) load_half(r.t3.dst_offset, R_TapePtr, O_(Binds_NormalizeV3S4, dst_offset)),
* r_tmp : src.x PRESERVED across stages 1-2 (NOT clobbered by mfc2 MAC2) → fed to IR1 in stage 4 LdSlot_ add_u(r.src_ptr, r.scratch, r.t5.src_offset),
* r_mac1_scratch : MAC1 result scratch (also holds aligned |v|² in stage 3) LdSlot_ add_u(r.dst_ptr, r.scratch, r.t3.dst_offset),
* r_mac2_scratch : MAC2 result scratch → result.x after stage 4 sra LdSlot_ add_ui_self(R_TapePtr, S_(Binds_NormalizeV3S4)),
* r_recip_est : src.y PRESERVED across stages 1-2 → fed to IR2 in stage 4 → result.y
* r_lzcr : |v|² sum (stage 2) → shift count (stage 3) → 1/|v| (stage 4 IR0)
* r_shift : shift count SAVED in stage 3 → consumed by stage 4 srav
* r_branch_tmp : src.z PRESERVED across stages 1-2 → fed to IR3 in stage 4 → result.z (also sqrtbl base addr)
*
* Atom_labels are srav_path / aligned_done
* (NOT namespaced — they're internal to this proc;
* the metaprogram's per-atom-name enum emission handles any collision across different atoms/files that share the same labels).
*
* Pool cost: 11 GPRs (well within the 9-10 caller-trash GPR budget when r_scratch is a wave-context carrier).
*
* Direct port of PSYQ libgte msc02.rel.text VectorNormal disassembly (0x800160a0..0x8001615c).
* Words: ~59 (matches libgte 0x800160a0..0x8001615c at +/- 0-2 words for BD-slot reshuffling).
* Sqrtbl: hardcoded to 0x800185B4 (libgte msc02.rel.data). Note: swapped to local.
* Pipeline: clobbers IR0..3, MAC1..3, LZCS, LZCR.
*/
/* MipsAtom_Proc_ wrapper: declares the static MipsCode[] body, then calls atombuilder_unroll(ab, ...) to copy the encoded instructions into the caller's MipsAtomBuilder arena. */
internal MipsAtom* normalize_v3s4_proc(AtomArena_R aa, U4 r_scratch /* GPR code: scratch base carrier (e.g., R_T4 = R_ResolveScratch) */
, U4 r_src_offset, U4 r_dst_offset /* GPR codes: PARAMETERIZED offsets (caller passes O_ macros) */
, U4 r_src_ptr, U4 r_dst_ptr, U4 r_tmp /* GPR codes: 3 scratch regs (src/dst computed + tmp) */
, U4 r_mac1_scratch, U4 r_mac2_scratch /* GPR codes: 2 more: MAC1/MAC2 scratch */
, U4 r_recip_est /* GPR code: |v|² sum + shift-input + sqrtbl[index] */
, U4 r_lzcr, U4 r_shift /* GPR codes: lzcr + final srav amount */
, U4 r_branch_tmp /* GPR code: scratch (shift count, branch target, lookup addr) */
)
MipsAtom_Proc_(normalize_v3s4, aa, {
add_si(r_src_ptr, r_scratch, r_src_offset), /* r_src_ptr = &src */
add_si(r_dst_ptr, r_scratch, r_dst_offset), /* r_dst_ptr = &dst */
nop,
/* Load src.x/y/z from r_src_ptr (caller-determined address) into r_tmp/r_recip_est/r_branch_tmp. /* Load src.x/y/z from r_src_ptr (caller-determined address) into r_tmp/r_recip_est/r_branch_tmp.
* r_tmp holds src.x throughout stages 1-2 — r_mac2_scratch is clobbered to MAC2 in stage 1.5 (line below). */ * r.rt1_src_x holds src.x throughout stages 1-2 — r_mac2_scratch is clobbered to MAC2 in stage 1.5 (line below).
load_word(r_tmp, r_src_ptr, O_(V3_S4,x)), * t5.src_offset/dst_offset are dead by here; t5 is reused for src.z in the mac_load_word_v3 below. */
load_word(r_recip_est, r_src_ptr, O_(V3_S4,y)), mac_load_word_v3(r.src_x, r.recip_est, r.t5.src_z, r.src_ptr, 0),
load_word(r_branch_tmp, r_src_ptr, O_(V3_S4,z)),
nop, /* load-delay */
/* Stage 1: mtc2 src → IR1/2/3, SQR fires. */ /* Stage 1: mtc2 src → IR1/2/3, SQR fires. */
gte_mv_to_data_r(r_tmp, C2_IR1), LdSlot_ mac_gte_sqr_v3s4(r.src_x, r.recip_est, r.t5.src_z, LdSlot_ nop),
gte_mv_to_data_r(r_recip_est, C2_IR2),
gte_mv_to_data_r(r_branch_tmp, C2_IR3),
nop, gte_cmdw_sqr,
/* Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS. */ /* Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS. */
gte_mv_from_data_r(r_mac1_scratch, C2_MAC1), mac_gte_mv_from_data_r_mac123(r.t3.mac1_scratch, r.t4.mac2_scratch, r.norm), LdSlot_ nop,
gte_mv_from_data_r(r_mac2_scratch, C2_MAC2), add_u_self( r.norm, r.t3.mac1_scratch),
gte_mv_from_data_r(r_lzcr, C2_MAC3), add_u_self( r.norm, r.t4.mac2_scratch),
nop, gte_mv_to_data_r( r.norm, C2_LZCS), GteDelay_ nop2,
add_u(r_lzcr, r_lzcr, r_mac2_scratch), gte_mv_from_data_r(r.shift, C2_LZCR), GteDelay_ nop,
add_u(r_lzcr, r_lzcr, r_mac1_scratch),
gte_mv_to_data_r(r_lzcr, C2_LZCS),
nop2,
gte_mv_from_data_r(r_shift, C2_LZCR),
nop,
/* Stage 3: compute srav amount (r_lzcr) + align |v|² to bit 24. /* Stage 3: round LZCR to even, compute half-shift, align |v|² to bit 24.
* IMPORTANT: the sllv/srav below writes the aligned |v|² to r_mac1_scratch (NOT r_lzcr), * r_norm holds |v|² sum; r_shift holds the LZCR count from mfc2.
* so r_lzcr retains the shift count all the way to the start of stage 4. * After the component: r_shift = even(LZCR), r_norm = half-shift, r_mac1_scratch = |v|². */
*/ mac_lzcr_round_even_half_shift(r.shift, r.norm, r.t3.mac1_scratch),
and_i( r_shift, r_shift, -2),
or_u(r_mac1_scratch, r_lzcr, 0), /* FIX B: save sum before clobbering r_lzcr with shift count */
li_s( r_lzcr, 31),
sub_s( r_lzcr, r_lzcr, r_shift),
shift_aright(r_lzcr, r_lzcr, 1),
/* r_branch_tmp = LZCR - 24 (overwrites r_branch_tmp; src.z no longer needed after SQR) */ /* r_branch_tmp = LZCR - 24 (overwrites r_branch_tmp; src.z no longer needed after SQR) */
add_si( r_branch_tmp, r_shift, -24), add_si( r.t5.btarget, r.shift, -24),
branch_lt_zero(r_branch_tmp, atom_offset(aligned_done, srav_path)), nop, /* FIX A: bltz → srav_path (LZCR<24 path) */ branch_lt_zero(r.t5.btarget, atom_offset(aligned_done, srav_path)), BdSlot_ nop, /* bltz → srav_path (LZCR < 24 path) */
jump_rel(atom_offset(srav_path, aligned_done)), /* FIX A: b → aligned_done (LZCR>=24 path) */ jump_rel(atom_offset(srav_path, aligned_done)), /* b → aligned_done (LZCR >= 24 path) */
shift_lleft_var(r_mac1_scratch, r_mac1_scratch, r_branch_tmp), /* FIX B: src=sum (r_mac1_scratch), dst=same */ BdSlot_ shift_lleft_var(r.t3.mac1_scratch, r.t3.mac1_scratch, r.t5.btarget), /* src=sum (r_mac1_scratch), dst=same */
atom_label(srav_path) atom_label(srav_path)
li_s( r_branch_tmp, 24), li_s( r.t5.shift_count, 24),
sub_s( r_branch_tmp, r_branch_tmp, r_shift), sub_s(r.t5.shift_count, r.t5.shift_count, r.shift),
shift_aright_var(r_mac1_scratch, r_mac1_scratch, r_branch_tmp), /* FIX B: src=sum (r_mac1_scratch), dst=same */ shift_aright_var(r.t3.mac1_scratch, r.t3.mac1_scratch, r.t5.shift_count), /* src=sum (r_mac1_scratch), dst=same */
atom_label(aligned_done) atom_label(aligned_done)
/* Save the shift count to r_shift before the next 5 instructions overwrite r_lzcr // Save the shift count to r_shift before the next 5 instructions overwrite r_norm (the sqrtbl lookup loads 1/|v| into r_norm, which becomes IR0 in stage 4).
* (the sqrtbl lookup loads 1/|v| into r_lzcr, which becomes IR0 in stage 4). */ or_u(r.shift, r.norm, 0), /* r_shift ← shift count (preserved through stage 4) */
or_u(r_shift, r_lzcr, 0), /* r_shift ← shift count (preserved through stage 4) */
/* r_mac1_scratch holds |v|² aligned (top bit at bit 7). */ /* r_mac1_scratch holds |v|² aligned (top bit at bit 7). */
add_si( r_mac1_scratch, r_mac1_scratch, -64), add_si( r.t3.mac1_scratch, r.t3.mac1_scratch, -64),
shift_lleft(r_mac1_scratch, r_mac1_scratch, 1), shift_lleft(r.t3.mac1_scratch, r.t3.mac1_scratch, 1),
load_upper_i(r_branch_tmp, u4_hi(& gte_normalize_sqr_tbl)), mac_load_word_imm(r.t5.lookup_addr, & gte_normalize_sqr_tbl), add_u_self(r.t5.lookup_addr, r.t3.mac1_scratch),
or_i_self( r_branch_tmp, u4_lo(& gte_normalize_sqr_tbl)), load_half(r.norm, r.t5.lookup_addr, 0), /* r_norm = sqrtbl[aligned-64] = 1/|v| (IR0 in stage 4) */
add_u(r_branch_tmp, r_branch_tmp, r_mac1_scratch),
load_half(r_lzcr, r_branch_tmp, 0), nop, /* r_lzcr = sqrtbl[aligned-64] = 1/|v| (IR0 in stage 4) */
/* FIX bug C: r_branch_tmp held the sqrtbl base+index, NOT src.z. Reload src.z from scratch now that r_branch_tmp is free. */ /* r_branch_tmp held the sqrtbl base+index, NOT src.z. Reload src.z from scratch now that r_branch_tmp is free. */
load_word(r_branch_tmp, r_src_ptr, O_(V3_S4,z)), nop, /* r_branch_tmp = src.z (for IR3 in stage 4) */ LdSlot_ load_word(r.t5.src_z, r.src_ptr, O_(V3_S4,z)), /* r_branch_tmp = src.z (for IR3 in stage 4) */
/* Stage 4: GPF + srav finalize (r_shift = shift count, r_lzcr = 1/|v|). */
gte_mv_to_data_r(r_lzcr, C2_IR0),
gte_mv_to_data_r(r_tmp, C2_IR1), /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */
gte_mv_to_data_r(r_recip_est, C2_IR2),
gte_mv_to_data_r(r_branch_tmp, C2_IR3), /* IR3 = src.z (reloaded) */
nop2, gte_cmdw_gpf,
gte_mv_from_data_r(r_mac2_scratch, C2_MAC1),
gte_mv_from_data_r(r_recip_est, C2_MAC2),
gte_mv_from_data_r(r_branch_tmp, C2_MAC3),
shift_aright_var(r_mac2_scratch, r_mac2_scratch, r_shift), /* sra by r_shift = (31-LZCR)/2 (saved before sqrtbl lookup) */
shift_aright_var(r_recip_est, r_recip_est, r_shift),
shift_aright_var(r_branch_tmp, r_branch_tmp, r_shift),
/* Stage 4: GPF + srav finalize (r_shift = shift count, r_norm = 1/|v|). */
LdSlot_ mac_gte_general_purpose_interopolation(
r.norm,
r.src_x, /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */
r.recip_est,
r.t5.src_z, /* IR3 = src.z (reloaded) */
r.t4.mac2_scratch, r.recip_est, r.t5.src_z,
GteDelay_ nop,
GteDelay_ nop
),
/* sra by r_shift = (31-LZCR)/2 (saved before sqrtbl lookup) */
mac_shift_aright_var_v3_self(r.t4.mac2_scratch, r.recip_est, r.t5.src_z, r.shift),
/* Store result.x/y/z to r_dst_ptr (caller-determined dst address). */ /* Store result.x/y/z to r_dst_ptr (caller-determined dst address). */
store_word(r_mac2_scratch, r_dst_ptr, O_(V3_S4,x)), mac_store_word_v3(r.t4.mac2_scratch, r.recip_est, r.t5.src_z, r.dst_ptr, 0),
store_word(r_recip_est, r_dst_ptr, O_(V3_S4,y)),
store_word(r_branch_tmp, r_dst_ptr, O_(V3_S4,z)),
mac_yield() mac_yield()
}) })
/* ─── GTE OP cross product (a × b → out) ───
* Generalized V3_S4 cross product via GTE OP (OuterProduct12 libpsyx convention).
* The >> 12 shift converts S12.20 → S12.0 OuterProduct12. */
typedef Struct_(Binds_gte_cross_v3s4) { V3_S4* src_a; V3_S4* src_b; V3_S4* out; };
typedef Struct_(RegUse_gte_cross_v3s4) {
Reg_(V3_S4) a;
Reg_(V3_S4) b;
union { Reg out, t0; } x;
union { Reg src_a, t1, rt11; } y;
union { Reg src_b, t2, rt22; } z;
};
internal MipsAtom* gte_cross_v3s4(AtomArena_R aa, RegUse_gte_cross_v3s4 r)
atom_info(atom_bind(Binds_gte_cross_v3s4)) MipsAtom_Proc_(aa, {
load_word(r.y.src_a, R_TapePtr, O_(Binds_gte_cross_v3s4,src_a)),
load_word(r.z.src_b, R_TapePtr, O_(Binds_gte_cross_v3s4,src_b)),
load_word(r.x.out, R_TapePtr, O_(Binds_gte_cross_v3s4,out)),
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_gte_cross_v3s4)),
mac_load_v3s4(r.a, r.y.src_a, 0), LdSlot_
mac_load_v3s4(r.b, r.z.src_b, 0), LdSlot_
mac_gte_op_cross_v3s4(r.a, r.b), /* RT diagonal + IR + OP + MAC read + shift */
mac_store_v3s4(r.a, r.x.out, 0),
mac_yield()
})
#pragma endregion Atom Procs #pragma endregion Atom Procs
#pragma region Baked Atoms #pragma region Baked Atoms
@@ -372,12 +396,22 @@ internal MipsAtom_(set_gte_mt3s2s4) atom_info(
load_word(R_T3, R_TapePtr, O_(Binds_SetGteMT3S2S4,transform)), load_word(R_T3, R_TapePtr, O_(Binds_SetGteMT3S2S4,transform)),
add_ui_self( R_TapePtr, S_(Binds_SetGteMT3S2S4)), add_ui_self( R_TapePtr, S_(Binds_SetGteMT3S2S4)),
/* Load 3x3 Rotation + 3x1 Translation from R_T3 into GTE CONTROL Regs (ctc2) */ /* Load 3x3 Rotation + 3x1 Translation from R_T3 into GTE CONTROL Regs (ctc2) */
load_word(R_T0, R_T3, 0), load_word(R_T1, R_T3, 4), load_word(R_T0, R_T3, 0),
gte_mv_to_ctrl_r(R_T0, gte_cr_RT11), gte_mv_to_ctrl_r(R_T1, gte_cr_RT12), load_word(R_T1, R_T3, 4),
load_word(R_T0, R_T3, 8), load_word(R_T1, R_T3, 12), load_word(R_T2, R_T3, 16), gte_mv_to_ctrl_r(R_T0, gte_cr_RT11),
gte_mv_to_ctrl_r(R_T0, gte_cr_RT13), gte_mv_to_ctrl_r(R_T1, gte_cr_RT21), gte_mv_to_ctrl_r(R_T2, gte_cr_RT22), gte_mv_to_ctrl_r(R_T1, gte_cr_RT12),
load_word(R_T0, R_T3, 20), load_word(R_T1, R_T3, 24), load_word(R_T2, R_T3, 28), load_word(R_T0, R_T3, 8),
gte_mv_to_ctrl_r(R_T0, gte_cr_TRX), gte_mv_to_ctrl_r(R_T1, gte_cr_TRY), gte_mv_to_ctrl_r(R_T2, gte_cr_TRZ), load_word(R_T1, R_T3, 12),
load_word(R_T2, R_T3, 16),
gte_mv_to_ctrl_r(R_T0, gte_cr_RT13),
gte_mv_to_ctrl_r(R_T1, gte_cr_RT21),
gte_mv_to_ctrl_r(R_T2, gte_cr_RT22),
load_word(R_T0, R_T3, 20),
load_word(R_T1, R_T3, 24),
load_word(R_T2, R_T3, 28),
gte_mv_to_ctrl_r(R_T0, gte_cr_TRX),
gte_mv_to_ctrl_r(R_T1, gte_cr_TRY),
gte_mv_to_ctrl_r(R_T2, gte_cr_TRZ),
mac_yield() mac_yield()
}; };
+64 -81
View File
@@ -16,9 +16,6 @@
* gte_mv_to_data_r (gte + mv + to + data + register) * gte_mv_to_data_r (gte + mv + to + data + register)
* gte_lw_v0_xy(base) (gte + lw + v0 + xy) * gte_lw_v0_xy(base) (gte + lw + v0 + xy)
* load_upper_i (load-upper + immediate, unique verb) * load_upper_i (load-upper + immediate, unique verb)
*
* Vendor mnemonics (gte_mtc2, gte_mfc2, gte_lwc2, gte_swc2, etc.) are NOT in this header.
* They are in the opt-in `gte_vendor_sym.h` for users who prefer the textbook MIPS assembly mnemonics.
* ============================================================================ */ * ============================================================================ */
#ifdef INTELLISENSE_DIRECTIVES #ifdef INTELLISENSE_DIRECTIVES
@@ -33,7 +30,7 @@
* gte.h — Geometry Transformation Engine (COP2) for the PS1 * gte.h — Geometry Transformation Engine (COP2) for the PS1
* ============================================================================ * ============================================================================
* *
* Hand-rolled DSL for emitting GTE/MIPS instruction words as raw `.word` constants from C. * Hand-rolled DSL for emitting GTE/MIPS instruction words from C.
* No GCC inline-assembly string syntax in the code body. * No GCC inline-assembly string syntax in the code body.
* *
* STYLE NOTES * STYLE NOTES
@@ -101,20 +98,20 @@ enum {
/* Semantic Aliases for GTE Data Registers */ /* Semantic Aliases for GTE Data Registers */
enum { enum {
gte_in_v0_xy = C2_VXY0, /* Input Vector 0 (X, Y) */ C2_InV0_XY = C2_VXY0, /* Input Vector 0 (X, Y) */
gte_in_v0_z = C2_VZ0, /* Input Vector 0 (Z) */ C2_InV0_Z = C2_VZ0, /* Input Vector 0 (Z) */
gte_in_v1_xy = C2_VXY1, /* Input Vector 1 (X, Y) */ C2_InV1_XY = C2_VXY1, /* Input Vector 1 (X, Y) */
gte_in_v1_z = C2_VZ1, /* Input Vector 1 (Z) */ C2_InV1_Z = C2_VZ1, /* Input Vector 1 (Z) */
gte_in_v2_xy = C2_VXY2, /* Input Vector 2 (X, Y) */ C2_InV2_XY = C2_VXY2, /* Input Vector 2 (X, Y) */
gte_in_v2_z = C2_VZ2, /* Input Vector 2 (Z) */ C2_InV2_Z = C2_VZ2, /* Input Vector 2 (Z) */
gte_in_rgb = C2_RGB, /* Input Color (R, G, B, MipsCode) */ C2_In_RGB = C2_RGB, /* Input Color (R, G, B, MipsCode) */
gte_out_scr_xy0 = C2_SXY0, /* Output Screen Coord 0 (X, Y) */ C2_OutSrc_XY0 = C2_SXY0, /* Output Screen Coord 0 (X, Y) */
gte_out_scr_xy1 = C2_SXY1, /* Output Screen Coord 1 (X, Y) */ C2_OutSrc_XY1 = C2_SXY1, /* Output Screen Coord 1 (X, Y) */
gte_out_scr_xy2 = C2_SXY2, /* Output Screen Coord 2 (X, Y) */ C2_OutSrc_XY2 = C2_SXY2, /* Output Screen Coord 2 (X, Y) */
gte_out_depth = C2_OTZ, /* Output Ordering Table Z (Depth) */ C2_OutDepth = C2_OTZ, /* Output Ordering Table Z (Depth) */
gte_math_accum0 = C2_MAC0, /* Math Accumulator 0 */ C2_MathAccu0 = C2_MAC0, /* Math Accumulator 0 */
gte_math_accum1 = C2_MAC1, /* Math Accumulator 1 */ C2_MathAccu1 = C2_MAC1, /* Math Accumulator 1 */
gte_math_accum2 = C2_MAC2, /* Math Accumulator 2 */ C2_MathAccu2 = C2_MAC2, /* Math Accumulator 2 */
}; };
/* --- GTE Command Semantics (The Bitfield Meanings) --- /* --- GTE Command Semantics (The Bitfield Meanings) ---
@@ -177,42 +174,36 @@ enum {
* Mirrors the OPCODE_SHIFT / RS_SHIFT convention used in mips.h. * Mirrors the OPCODE_SHIFT / RS_SHIFT convention used in mips.h.
*/ */
gte_shift_sf = 19, gte_width_sf = 1, gte_mask_sf = 0x1, gte_shift_sf = 19, gte_width_sf = 1,
gte_shift_mx = 17, gte_width_mx = 2, gte_mask_mx = 0x3, gte_shift_mx = 17, gte_width_mx = 2,
gte_shift_v = 15, gte_width_v = 2, gte_mask_v = 0x3, gte_shift_v = 15, gte_width_v = 2,
gte_shift_cv = 13, gte_width_cv = 2, gte_mask_cv = 0x3, gte_shift_cv = 13, gte_width_cv = 2,
gte_shift_lm = 10, gte_width_lm = 1, gte_mask_lm = 0x1, gte_shift_lm = 10, gte_width_lm = 1,
gte_shift_cmd = 0, gte_width_cmd = 6, gte_mask_cmd = 0x3F, gte_shift_cmd = 0, gte_width_cmd = 6,
/* Fake command number (bits 24-20) — IGNORED by the GTE hardware per PSX-SPX `geometrytransformationenginegte.md` line 48. /* Fake command number (bits 24-20) — IGNORED by the GTE hardware per PSX-SPX `geometrytransformationenginegte.md` line 48.
* libgte's compiler emits non-zero values in this field as a disassembly signature. */ * libgte's compiler emits non-zero values in this field as a disassembly signature. */
gte_shift_fake_cmd = 20, gte_shift_fake_cmd = 20,
gte_width_fake_cmd = 5, gte_width_fake_cmd = 5,
gte_mask_fake_cmd = 0x1F,
}; };
/* --- GTE Control Register Aliases (Pitfall 1) --- /* --- GTE Control Register Aliases (Pitfall 1) ---
* Three pairs of aliases map to the SAME C2 control-register slot on real silicon: * Three pairs of aliases map to the C2 control-register slot:
* C2[24] = gte_cr_RBK (background R) | gte_cr_OFX (screen offset X) * C2[24] = gte_cr_RBK (background R) | gte_cr_OFX (screen offset X)
* C2[25] = gte_cr_GBK (background G) | gte_cr_OFY (screen offset Y) * C2[25] = gte_cr_GBK (background G) | gte_cr_OFY (screen offset Y)
* C2[26] = gte_cr_BBK (background B) | gte_cr_H (projection plane distance H) * C2[26] = gte_cr_BBK (background B) | gte_cr_H (projection plane distance H)
* Cross-alias writes inside one atom body, or across the wave-context boundary, * Cross-alias writes inside one atom body, or across the wave-context boundary, silently clobber each other.
* silently clobber each other. The metaprogram's check_gte_cr_alias_writes * The metaprogram's check_gte_cr_alias_writes (CHECK_RULES row) warns about each pair per source.
* (CHECK_RULES row) warns about each pair per source. See * See psx-spx docs/gte_reference.md §"Control-register alias table" for the silicon rationale and the libgte outer-product convention.
* docs/gte_reference.md §"Control-register alias table" for the silicon
* rationale and the libgte outer-product convention.
*/ */
/* --- RT-matrix packed-slot convention (Pitfall 4) --- /* --- RT-matrix packed-slot convention (Pitfall 4) ---
* The silicon packs two 16-bit RT elements per 32-bit C2 slot: * The silicon packs two 16-bit RT elements per 32-bit C2 slot:
* C2[2] = (RT22 << 16) | RT13 (gte_cr_RT13 writes the low half, gte_cr_RT22 writes the high half) * C2[2] = (RT22 << 16) | RT13 (gte_cr_RT13 writes the low half, gte_cr_RT22 writes the high half)
* C2[4] = (RT33 << 16) | RT22 (gte_cr_RT22 writes the low half — clobbers prior RT22 value if RT13 was also written) * C2[4] = (RT33 << 16) | RT22 (gte_cr_RT22 writes the low half — clobbers prior RT22 value if RT13 was also written)
* OP and MVMVA read D1/D2/D3 from these packed slots. The libgte outer-product * OP and MVMVA read D1/D2/D3 from these packed slots.
* convention (see ac_apply_matrix_lv at gte.atom.c:108-122) writes C2[2] then * The libgte outer-product convention (see ac_apply_matrix_lv at gte.atom.c:108-122) writes C2[2] then C2[4] in sequence;
* C2[4] in sequence; the SECOND write's low half is RT22, not RT13. An agent * the SECOND write's low half is RT22, not RT13.
* who writes gte_cr_RT13 then gte_cr_RT22 to the SAME source GPR clobbers the
* RT13 value. See docs/gte_reference.md §"RT-matrix packed-slot convention"
* for the canonical write pattern.
*/ */
/* --- GTE Control Register Indices (for ctc2/cfc2) --- /* --- GTE Control Register Indices (for ctc2/cfc2) ---
@@ -301,8 +292,7 @@ enum { _C2_TX_SUBS_ = 0
// #define gte_mv_from_data_r(rt, rd) enc_gte_tx(cop_mf, (rt), (rd)) /* Move GTE Control Register (rd) to GPR (rt) */ // #define gte_mv_from_data_r(rt, rd) enc_gte_tx(cop_mf, (rt), (rd)) /* Move GTE Control Register (rd) to GPR (rt) */
/* GTE Data vs Control Register Transfers /* GTE Data vs Control Register Transfers
* * Each macro emits a single instruction for one of MFC2/CFC2/MTC2/CTC2.
* Each macro emits a single .word constant for one of MFC2/CFC2/MTC2/CTC2.
* *
* `rd` is the C2 register index in the file the sub-opcode names: * `rd` is the C2 register index in the file the sub-opcode names:
* gte_mv_from_data_r / gte_mv_to_data_r → C2 data register file * gte_mv_from_data_r / gte_mv_to_data_r → C2 data register file
@@ -317,14 +307,14 @@ enum { _C2_TX_SUBS_ = 0
#define gte_mv_from_ctrl_r(rt, rd) enc_gte_tx(sub_cfc2, (rt), (rd)) /* Copy From ctrl reg */ #define gte_mv_from_ctrl_r(rt, rd) enc_gte_tx(sub_cfc2, (rt), (rd)) /* Copy From ctrl reg */
#define gte_mv_to_data_r(rt, rd) enc_gte_tx(sub_mtc2, (rt), (rd)) /* Move To data reg */ #define gte_mv_to_data_r(rt, rd) enc_gte_tx(sub_mtc2, (rt), (rd)) /* Move To data reg */
#define gte_mv_to_ctrl_r(rt, rd) enc_gte_tx(sub_ctc2, (rt), (rd)) /* Copy To ctrl reg */ #define gte_mv_to_ctrl_r(rt, rd) enc_gte_tx(sub_ctc2, (rt), (rd)) /* Copy To ctrl reg */
#define GteDelay_ // Annotate an instruction as filling a CPU <-> GTE DMA delay slot/s
/* COP2 Data Load (lwc2): `lwc2 rt, off(rs)` /* COP2 Data Load (lwc2): `lwc2 rt, off(rs)`
* Layout: [op_lwc2:6][rs:5][rt:5][imm:16] * Layout: [op_lwc2:6][rs:5][rt:5][imm:16]
* - rs: GPR base address * - rs: GPR base address
* - rt: COP2 data register index (0..31) * - rt: COP2 data register index (0..31)
* - imm: signed 16-bit offset * - imm: signed 16-bit offset
* NOTE: When `rs` is a runtime register, the encoding cannot be pre-baked * NOTE: When `rs` is a runtime register, the encoding cannot be pre-baked into a .word — use the string-style `gte_load_v0` macro below instead. */
* into a .word — use the string-style `gte_load_v0` macro below instead. */
#define enc_gte_lw(rt, base, off) enc_i(op_lwc2, (base), (rt), (off)) #define enc_gte_lw(rt, base, off) enc_i(op_lwc2, (base), (rt), (off))
/* Store Word */ /* Store Word */
#define enc_gte_sw(rt, base, off) enc_i(op_swc2, (base), (rt), (off)) #define enc_gte_sw(rt, base, off) enc_i(op_swc2, (base), (rt), (off))
@@ -333,8 +323,7 @@ enum { _C2_TX_SUBS_ = 0
* `swc2` is redundant when we're already inside the `gte_` namespace. * `swc2` is redundant when we're already inside the `gte_` namespace.
* gte_lw rt, base, off → lwc2 rt, off(base) * gte_lw rt, base, off → lwc2 rt, off(base)
* gte_sw rt, base, off → swc2 rt, off(base) * gte_sw rt, base, off → swc2 rt, off(base)
* For the typical user-facing vector-level load (xy + z as two instructions), * For the typical user-facing vector-level load (xy + z as two instructions), use the higher-level `gte_load_vN` macros below. */
* use the higher-level `gte_load_vN` macros below. */
#define gte_lw(rt, base, off) enc_gte_lw(rt, base, off) #define gte_lw(rt, base, off) enc_gte_lw(rt, base, off)
#define gte_sw(rt, base, off) enc_gte_sw(rt, base, off) #define gte_sw(rt, base, off) enc_gte_sw(rt, base, off)
@@ -351,13 +340,13 @@ enum { _C2_TX_SUBS_ = 0
#define gte_cmd_base (enc_op(op_cop2) | (1 << 25)) #define gte_cmd_base (enc_op(op_cop2) | (1 << 25))
/* Per-field encoders. Each one does (value & mask) << shift on its own. */ /* Per-field encoders. Each one does (value & mask) << shift on its own. */
#define enc_gte_sf(sf) (((sf) & gte_mask_sf ) << gte_shift_sf ) #define enc_gte_sf(sf) ((sf) << gte_shift_sf )
#define enc_gte_mx(mx) (((mx) & gte_mask_mx ) << gte_shift_mx ) #define enc_gte_mx(mx) ((mx) << gte_shift_mx )
#define enc_gte_v(v) (((v) & gte_mask_v ) << gte_shift_v ) #define enc_gte_v(v) ((v) << gte_shift_v )
#define enc_gte_cv(cv) (((cv) & gte_mask_cv ) << gte_shift_cv ) #define enc_gte_cv(cv) ((cv) << gte_shift_cv )
#define enc_gte_lm(lm) (((lm) & gte_mask_lm ) << gte_shift_lm ) #define enc_gte_lm(lm) ((lm) << gte_shift_lm )
#define enc_gte_cmd(cmd) (((cmd) & gte_mask_cmd ) << gte_shift_cmd ) #define enc_gte_cmd(cmd) ((cmd) << gte_shift_cmd )
#define enc_gte_fake_cmd(x) (((x) & gte_mask_fake_cmd) << gte_shift_fake_cmd) #define enc_gte_fake_cmd(x) ((x) << gte_shift_fake_cmd)
/* Composite: all six GTE fields + the COP2/CO base. */ /* Composite: all six GTE fields + the COP2/CO base. */
#define enc_gte_cmdw(sf, mx, v, cv, lm, cmd) ( \ #define enc_gte_cmdw(sf, mx, v, cv, lm, cmd) ( \
@@ -409,10 +398,11 @@ enum { _C2_TX_SUBS_ = 0
#define gte_cmdw_rtpt (gte_cmd_base | enc_gte_cmd(gte_cmd_rtpt ) | gte_cmdw_psyq_compat) #define gte_cmdw_rtpt (gte_cmd_base | enc_gte_cmd(gte_cmd_rtpt ) | gte_cmdw_psyq_compat)
#define gte_cmdw_nclip (gte_cmd_base | enc_gte_cmd(gte_cmd_nclip)) #define gte_cmdw_nclip (gte_cmd_base | enc_gte_cmd(gte_cmd_nclip))
#define gte_cmdw_op (gte_cmd_base | enc_gte_cmd(gte_cmd_op )) #define gte_cmdw_op (gte_cmd_base | enc_gte_cmd(gte_cmd_op ))
#define gte_cmdw_outer_product gte_cmdw_op /* "outer product" -- NOCASH/Sdk terminology */ #define gte_cmdw_outer_product gte_cmdw_op /* "outer product" -- PSY-Q terminology */
#define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology. #define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology. */
* RGA(Lengyel): the GTE OP is a 3D signed-16-bit D x IR cross, not a generic RGA exterior product. #define gte_cmdw_cross gte_cmdw_op /* "cross product" -- geometric-algebra terminology.
* The wedge alias is the 3D complement interpretation of the same 3 scalars (MAC1..MAC3). */ * RGA(Lengyel): The GTE OP is a 3D signed-16-bit D x IR cross, not a generic RGA exterior product.
* The wedge alias is a 3D complement interpretation of the same 3 scalars (MAC1..MAC3). */
#define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva)) #define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva))
/* MVMVA with sf=0 (no shift, full-integer), cv=3 (no translation), v=3 (IR vector input). /* MVMVA with sf=0 (no shift, full-integer), cv=3 (no translation), v=3 (IR vector input).
@@ -440,29 +430,17 @@ enum { _C2_TX_SUBS_ = 0
#define gte_cmdw_mvmva_pass1_c11 (gte_cmd_base | enc_gte_v(2) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva)) #define gte_cmdw_mvmva_pass1_c11 (gte_cmd_base | enc_gte_v(2) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
#define gte_cmdw_mvmva_no_tr gte_cmdw_mvmva_ir #define gte_cmdw_mvmva_no_tr gte_cmdw_mvmva_ir
/* MVMVA pass 2 — EXACT C11 ApplyMatrixLV command. /* MVMVA pass 2 — C11 ApplyMatrixLV command.
* Command word: 0x4A49E012. * Decoded: op_cop2 | CO | fake_cmd=4 | sf=1 (>>12) | mx=0 (RT matrix) | v=3 (IR) | cv=3 (no translation) | lm=0 | cmd=MVMVA.
* bits 31-26: 010010 = COP2 * Reads (RT row · IR) >> 12 into MAC1/2/3. Per-field composition (no opaque literal)
* bit 25: 1 (CO set) * keeps the bit layout visible at the call site + matches the libgte C-side byte-exact. */
* bits 24-20: 01001 = 9 (fake_cmd) #define gte_cmdw_mvmva_c11_pass2 (gte_cmd_base | enc_gte_fake_cmd(4) | enc_gte_sf(1) | enc_gte_v(3) | enc_gte_mx(0) | enc_gte_cv(3) | enc_gte_cmd(gte_cmd_mvmva))
* bit 19: 1 (sf=1)
* bits 18-17: 00 (mx=0, RT matrix)
* bits 16-15: 11 (v=3, IR)
* bits 14-13: 11 (cv=3, no translation)
* bits 5-0: 010010 = MVMVA
* sf=1, mx=0, v=3, cv=3. Pass 2 reads RT matrix, IR input, >>12. */
#define gte_cmdw_mvmva_c11_pass2_exact 0x4A49E012
/* MVMVA pass 1 — C11's exact command: 0x4A41E012.
* bit 25: 1, sf=0, mx=0, v=3, cv=3. Pass 1 reads RT matrix, IR input, no shift. */
#define gte_cmdw_mvmva_c11_pass1_exact 0x4A41E012
/* MVMVA: sf=1 (>>12), mx=0 (RT matrix), v=0 (V0), cv=3 (no TR). */ /* MVMVA: sf=1 (>>12), mx=0 (RT matrix), v=0 (V0), cv=3 (no TR). */
#define gte_cmdw_mvmva_sf1_mx0_v0_cv3 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_v(0) | enc_gte_mx(0) | enc_gte_cmd(gte_cmd_mvmva)) #define gte_cmdw_mvmva_sf1_mx0_v0_cv3 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_v(0) | enc_gte_mx(0) | enc_gte_cmd(gte_cmd_mvmva))
/* RTPS with sf=1 (12-bit shift, no translation): matches the output of libgte's /* RTPS with sf=1 (12-bit shift, no translation): matches the output of libgte's ApplyMatrixLV when the GTE pipeline expects R*pos >> 12.
* ApplyMatrixLV when the GTE pipeline expects R*pos >> 12. The shift produces * The shift produces values like (-270, 710, 1713) which match the C11 reference path. */
* values like (-270, 710, 1713) which match the C11 reference path. */
#define gte_cmdw_rtps_sf1 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_cmd(gte_cmd_rtps)) #define gte_cmdw_rtps_sf1 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_cmd(gte_cmd_rtps))
/* SQR / GPF cosmetic-bits compat helpers. /* SQR / GPF cosmetic-bits compat helpers.
@@ -485,16 +463,22 @@ enum { _C2_TX_SUBS_ = 0
/* GPF — General-purpose Interpolation. /* GPF — General-purpose Interpolation.
* PSX-SPX `geometrytransformationenginegte.md` §"GPF": * PSX-SPX `geometrytransformationenginegte.md` §"GPF":
* [MAC1,MAC2,MAC3] = (([IR1,IR2,IR3] * IR0) + [MAC1,MAC2,MAC3]) SAR (sf*12) * [MAC1,MAC2,MAC3] = (([IR1,IR2,IR3] * IR0) + [MAC1,MAC2,MAC3]) SAR (sf * 12)
* [IR1,IR2,IR3] = [MAC1,MAC2,MAC3] * [IR1,IR2,IR3] = [MAC1,MAC2,MAC3]
* Sourced verbatim from libgte msc02 VectorNormal disassembly at 0x8001613c: * Sourced verbatim from libgte msc02 VectorNormal disassembly at 0x8001613c:
* 0x4B90003D = gte_cmd_base | gte_cmdw_gpf_compat | enc_gte_cmd(0x3D) * 0x4B90003D = gte_cmd_base | gte_cmdw_gpf_compat | enc_gte_cmd(0x3D)
* bit 19 sf=0 * bit 19 sf = 0
* bit 10 lm=0 * bit 10 lm = 0
* bits 5-0 cmd=0x3D=GPF * bits 5-0 cmd = 0x3D = GPF
* bits 24-20 = 0x19 (libgte "nonsense SDK command number" signature) */ * bits 24-20 = 0x19 (libgte "nonsense SDK command number" signature) */
#define gte_cmdw_gpf (gte_cmd_base | enc_gte_cmd(gte_cmd_gpf) | gte_cmdw_gpf_fake_sig) #define gte_cmdw_gpf (gte_cmd_base | enc_gte_cmd(gte_cmd_gpf) | gte_cmdw_gpf_fake_sig)
/* Mask to round LZCR (leading-zero/ones count, range 1..32 per PSX-SPX cop2r31) down to even.
* The normalize_v3s4 half-shift logic computes (31 - LZCR) >> 1; clearing bit 0 ensures the subtraction result is always odd, so the >> 1 division is consistent (no 0.5 loss). */
enum {
gte_lzcr_even_mask = 0xFFFE, /* all bits except bit 0 */
};
#define gte_cmdw_rotate_translate_perspective_single gte_cmdw_rtps #define gte_cmdw_rotate_translate_perspective_single gte_cmdw_rtps
#define gte_cmdw_rotate_translate_perspective_triple gte_cmdw_rtpt #define gte_cmdw_rotate_translate_perspective_triple gte_cmdw_rtpt
/* RGA(Lengyel): RTPS/RTPT consume the matrix expansion of a rigid transformation (rotation matrix + translation vector) loaded into the RT/TR control registers. /* RGA(Lengyel): RTPS/RTPT consume the matrix expansion of a rigid transformation (rotation matrix + translation vector) loaded into the RT/TR control registers.
@@ -597,8 +581,8 @@ enum {
/* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — prelude to gte_cmd_rtpt. /* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — prelude to gte_cmd_rtpt.
* *
* Loads all three GTE input vectors (6 words) from three separate pointers, one per GTE vector register, * Loads all three GTE input vectors (6 words) from three separate pointers, one per GTE vector register, each loaded from its own base GPR.
* each loaded from its own base GPR. Caller must bind each `pN` to `bN` via a register variable. * Caller must bind each `pN` to `bN` via a register variable.
* register V3_S2* p0 rgcc(R_T4) = verts[0].ptr; // → __asm__("$12") * register V3_S2* p0 rgcc(R_T4) = verts[0].ptr; // → __asm__("$12")
* register V3_S2* p1 rgcc(R_T5) = verts[1].ptr; // → __asm__("$13") * register V3_S2* p1 rgcc(R_T5) = verts[1].ptr; // → __asm__("$13")
* register V3_S2* p2 rgcc(R_T6) = verts[2].ptr; // → __asm__("$14") * register V3_S2* p2 rgcc(R_T6) = verts[2].ptr; // → __asm__("$14")
@@ -692,8 +676,7 @@ enum {
* Loads the 3x3 rotation matrix at `r0` into the GTE's rotation-matrix control registers (RT11..RT22, indices 0..4) via ctc2. * Loads the 3x3 rotation matrix at `r0` into the GTE's rotation-matrix control registers (RT11..RT22, indices 0..4) via ctc2.
* *
* Memory layout at r0: five contiguous 32-bit words (offsets 0..16), each holding two packed 16-bit matrix elements. * Memory layout at r0: five contiguous 32-bit words (offsets 0..16), each holding two packed 16-bit matrix elements.
* The first 1.5 rows of a standard PSX SDK MATRIX struct (where each row is laid out as * The first 1.5 rows of a standard PSX SDK MATRIX struct (where each row is laid out as [RT_xx, RT_xy] | [RT_xz, pad] | ...).
* [RT_xx, RT_xy] | [RT_xz, pad] | ...).
* *
* Generated MIPS (mirrors the source macro): * Generated MIPS (mirrors the source macro):
* lw $12, 0( %0 ) ; word 0 * lw $12, 0( %0 ) ; word 0
+232 -106
View File
@@ -66,61 +66,81 @@
* */ * */
/* Register Allocation Info */ /* Register Allocation Info */
enum { enum {
R_AtomJmp = R_T8 atom_reg, /* debug-visible; tape yield handshake scratch */ R_ScratchBase = R_SP atom_reg, /* Scratchpad base address (host frame top) */
R_TapePtr = R_T9 atom_reg, /* The Instruction Stream Pointer */ R_AtomJmp = R_FP atom_reg, /* Next atom target (yield handshake scratch) */
R_TapePtr = R_RA atom_reg, /* The Instruction Stream Pointer */
/* Stringification codes for the GCC inline assembler clobber lists. */ /* Stringification codes for the GCC inline assembler clobber lists. */
#define R_AtomJmp_Code R_T8_Code #define R_ScratchBase_Code R_SP_Code
#define R_TapePtr_Code R_T9_Code #define R_AtomJmp_Code R_FP_Code
#define R_TapePtr_Code R_RA_Code
// R_InCursor = R_T4, // R_InCursor = R_T4,
// #define R_InCursor_Code R_T4_Code // #define R_InCursor_Code R_T4_Code
// Reserved Registers (Callee-saved): // Reserved Registers (Callee-saved across the host ABI transition):
// - R_T9: Holds the Tape Ptr which we need to increment // - R_SP: Holds the scratchpad base while tape code executes.
// If we hit a wall with register allocations we can clobber V0 & V1 (return values), defering as opt-in by user. // - R_FP: Holds the next atom target.
// - R_RA: Not sure?? // - R_RA: Holds the tape cursor.
// Needed by ac_yield but can be used as atom scratch: // All atom-body allocations must stay out of these.
// - R_T8: Will be used as the atom jump register. // Atom bodies may freely use R2-R25.
// All allocatable registers for mips atoms: // All allocatable registers for atom bodies (R2-R25, 24 registers):
R_TScratchVolatile = R_AT, // This one is reserved for psuedo instructions, but you can technically use it.
R_TScratch0 = R_T0, R_PsuedoVolatile = R_AT, // Assembler temporary; never allocate.
R_TScratch1 = R_T1,
R_TScratch2 = R_T2, // Atom Allocation Pool
R_TScratch3 = R_T3, R_Atom0 = R_T0,
R_TScratch4 = R_T4, R_Atom1 = R_T1,
R_TScratch5 = R_T5, R_Atom2 = R_T2,
R_TScratch6 = R_T6, R_Atom3 = R_T3,
R_TScratch7 = R_T7, R_Atom4 = R_T4,
R_TScratch8 = R_T8, R_Atom5 = R_T5,
R_TScratch10 = R_V0, // Tend to be used with gte DMAs R_Atom6 = R_T6,
R_TScratch11 = R_V1, // Tend to be used with gte DMAs R_Atom7 = R_T7,
// Note(Ed): We can technically clobber these, but don't unless we hit a bottleneck. R_Atom8 = R_T8,
// A 0-2 R_Atom9 = R_T9,
// S 0-7 R_Atom10 = R_V0, // Tend to be used with gte DMAs
R_Atom11 = R_V1, // Tend to be used with gte DMAs
R_Atom12 = R_A0,
R_Atom13 = R_A1,
R_Atom14 = R_A2,
R_Atom15 = R_A3,
R_Atom16 = R_S0,
R_Atom17 = R_S1,
R_Atom18 = R_S2,
R_Atom19 = R_S3,
R_Atom20 = R_S4,
R_Atom21 = R_S5,
R_Atom22 = R_S6,
R_Atom23 = R_S7,
}; };
typedef U2 Reg; // Register parameter used with atom or atom component procedures typedef U2 Reg; // Register parameter used with atom or atom component procedures
#define Reg_(type) tmpl(Reg,type) // Just a way to template register allocations of C-struct types.
typedef U4 const MipsCode; // Underlying type to mips asm words. typedef U4 const MipsCode; // Underlying type to mips asm words.
typedef Slice_(MipsCode); typedef Slice_(MipsCode);
typedef U4 const MipsAtom; typedef U4 const MipsAtom; // Underlying type to a mips atom defnition
typedef Slice_(MipsAtom); typedef Slice_(MipsAtom);
// Sometimes a user will define a bundle of atoms that represent a procedure of work as: // Sometimes a user will define a bundle of atoms that represent a procedure of work as:
// MipsAtom* <identifier>[...]; // MipsAtom* <identifier>[...];
// Unfortuantely if using slice_from_array it will make the slice's pointer: MipsAtom** so this enforce its defined as MipsAtom* // Unfortuantely if using slice_from_array it will make the slice's pointer: MipsAtom** so this enforce its defined as MipsAtom*
// TODO(Ed): Alternatively we can make the MipsAtom an opaque pointer to the atom... so that the blow returns 'MipsAtom'. // TODO(Ed): Alternatively we can make the MipsAtom an opaque pointer to the atom... so that the proc returns 'MipsAtom'.
#define atombundle_from_array(array) (Slice_MipsAtom){.ptr=array[0],.len=Array_len(array)} #define atombundle_from_array(array) (Slice_MipsAtom){.ptr=array[0],.len=Array_len(array)}
// Underlying type to an ptr to an array of mips asm words that must terminate with an ac_yield. // Underlying type to an ptr to an array of mips asm words that must terminate with an ac_yield.
#define MipsAtom_(sym) MipsCode sym [] align_(4) = #define MipsAtom_(sym) MipsCode sym [] align_(4) =
// Used for atoms with value-args // Used for atoms with value-args
// FI_ void ac_X(args) MipsAtomComp_Proc_(ac_X, { body }) // internal MipsAtom* X_proc(AtomArena_R aa, args) MipsAtom_Proc_(X, aa, { body })
// expands to: // expands to:
// FI_ void ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return ac_X; } // internal MipsAtom* X_proc(AtomArena_R aa, args) { MipsCode atom_comp_code[] align_(4) = { body }; return atomarena_push(aa, slice_from_array(MipsCode, atom_comp_code)); }
#define MipsAtom_Proc_(sym, aa, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; return atomarena_push(aa, slice_from_array(MipsCode, sym)); } // The atom name is derived by the Lua metaprogram from the preceding
// `MipsAtom* X_proc(...)` declaration (backward walk from the macro site,
// strips the `_proc` suffix).
#define MipsAtom_Proc_(aa, ...) { MipsCode atom_comp_code[] align_(4) = __VA_ARGS__; return atomarena_push(aa, slice_from_array(MipsCode, atom_comp_code)); }
// Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names). // Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names).
// MipsAtomComp_(ac_X) { body } // MipsAtomComp_(ac_X) { body }
@@ -129,68 +149,89 @@ typedef Slice_(MipsAtom);
#define MipsAtomComp_(sym) MipsCode sym [] align_(4) = #define MipsAtomComp_(sym) MipsCode sym [] align_(4) =
// Used for components with value-args (mandatory `ab` (atom-builder) arg). // Used for components with value-args (mandatory `ab` (atom-builder) arg).
// FI_ void ac_X(MipsAtomBuilder_R ab, args) MipsAtomComp_Proc_(ac_X, ab, { body }) // FI_ void ac_X(MipsAtomBuilder_R ab, args) MipsAtomComp_Proc_(ab, { body })
// expands to: // expands to:
// FI_ void ac_X(MipsAtomBuilder_R ab, args) { // FI_ void ac_X(MipsAtomBuilder_R ab, args) {
// MipsCode ac_X[] align_(4) = { body }; // MipsCode atom_comp_code[] align_(4) = { body };
// atombuilder_unroll(ab, slice_from_array(MipsCode, ac_X)); // atombuilder_push(ab, slice_from_array(MipsCode, atom_comp_code));
// } // }
// The body must NOT include mac_yield() (the parent atom yields). // The body must NOT include mac_yield() (the parent atom yields).
// Inline-only callers (the generated `mac_<name>` aliases) skip this arg via metaprogram filtering; // The component name is derived by the Lua metaprogram from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration (backward walk from the macro site).
// escape callers (ac_<name> invoked as a function) pass a long-lived builder. // Inline-only callers (the generated `mac_<name>` aliases) skip the `ab` arg via metaprogram filtering; escape callers (ac_<name> invoked as a function) pass a long-lived builder.
#define MipsAtomComp_Proc_(sym, ab, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; atombuilder_push(ab, slice_from_array(MipsCode, sym)); } #define MipsAtomComp_Proc_(ab, ...) { MipsCode atom_comp_code[] align_(4) = __VA_ARGS__; atombuilder_push(ab, slice_from_array(MipsCode, atom_comp_code)); }
// Used for trivial mappings from one atom component proc to the command of a more baser (meant for type-mapping)
#define MipsAtomComp_ProcMap_(ab, base_command) atom_dbg_skip MipsAtomComp_Proc_(ab, {base_command })
/* Line-table anchor: gcc only adds a file to the .debug_line file table when the contains line-numbered content. /* Line-table anchor: gcc only adds a file to the .debug_line file table when the contains line-numbered content.
Files containing only atoms and atom components. Files containing only atoms and atom components.
Place `ATOM_FILE_LINE_MARKER();` once at file scope in any `.atom.c` that defines atoms. Place `ATOM_FILE_LINE_MARKER();` once at file scope in any `.atom.c` that defines atoms.
The macro expands to a file-scope `internal U4 const` declaration keeps the file in the line table. Macro expands to a file-scope `internal U4 const` declaration keeps the file in the line table.
The constant is in `.rodata` and unreferenced; the linker may eliminate it. The constant is in `.rodata` so the linker may eliminate it. */
The two-level concat + `__LINE__` suffix makes the identifier unique per call site
(the identifier embeds the source line, so duplicates across `#include`d files don't collide). */
#define ATOM_FILE_DEBUGGER_LINE_MARKER(file_name) internal U4 const tmpl(atom_file_debugger_line_marker,file_name) = 0 #define ATOM_FILE_DEBUGGER_LINE_MARKER(file_name) internal U4 const tmpl(atom_file_debugger_line_marker,file_name) = 0
typedef Slice_MipsAtom Tape; typedef Slice_MipsAtom Tape;
/* The 'Exit' Atom */ typedef Struct_(TapeHostFrame) {
atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(R_RA), nop }; U4 s0;
U4 s1;
U4 s2;
U4 s3;
U4 s4;
U4 s5;
U4 s6;
U4 s7;
U4 fp;
U4 sp;
U4 ra;
};
// TODO(Ed): When we have a substantial workload/throughput, profile each of these to see impact at ABI boundaries. enum {
TapeHostFrame_Loc = Scratchpad_End - S_(TapeHostFrame),
TapeScratch_Len = TapeHostFrame_Loc - Scratchpad_Loc,
};
static_assert(S_(TapeHostFrame) == 11 * S_(U4));
static_assert(TapeHostFrame_Loc == 0x1F8003D4);
/* Tape Runner (Default) */ atom_dbg_skip MipsAtom_(tape_enter) {
FI_ void tape_run(Tape tape) { register U4* tape_ptr rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile( mac_load_word_imm(R_V0, u4_(TapeHostFrame_Loc)),
asm_words( store_word(R_S0, R_V0, O_(TapeHostFrame,s0)),
load_word( R_AtomJmp, R_TapePtr, 0) /* Bootstrap the first jump */ store_word(R_S1, R_V0, O_(TapeHostFrame,s1)),
, add_ui_self(R_TapePtr, S_(MipsAtom)) /* Advance tape */ store_word(R_S2, R_V0, O_(TapeHostFrame,s2)),
, call_reg( R_AtomJmp) /* jalr $t9 */ store_word(R_S3, R_V0, O_(TapeHostFrame,s3)),
, nop /* Branch delay slot */ store_word(R_S4, R_V0, O_(TapeHostFrame,s4)),
) store_word(R_S5, R_V0, O_(TapeHostFrame,s5)),
asm_rpins, r_use(tape_ptr) store_word(R_S6, R_V0, O_(TapeHostFrame,s6)),
asm_clobber: store_word(R_S7, R_V0, O_(TapeHostFrame,s7)),
rlit(R_AT), store_word(R_FP, R_V0, O_(TapeHostFrame,fp)),
rlit(R_V0), rlit(R_V1), // We clobber these for GTE ACs (that don't expose register selection, might expose them in the future...) store_word(R_SP, R_V0, O_(TapeHostFrame,sp)),
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4), store_word(R_RA, R_V0, O_(TapeHostFrame,ra)),
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), add_ui(R_TapePtr, R_A0, 0),
clb_mem_drain load_upper_i(R_ScratchBase, u4_hi(Scratchpad_Loc)),
); } load_word(R_AtomJmp, R_TapePtr, 0),
add_ui_self( R_TapePtr, S_(MipsAtom)),
jump_reg(R_AtomJmp), BdSlot_ nop,
};
/* Tape Runner (Static and Arg Clobbers) */ atom_dbg_skip MipsAtom_(tape_exit) {
FI_ void tape_run_a02_s07(Tape tape) { register U4* tape_ptr rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile( mac_load_word_imm(R_V0, u4_(TapeHostFrame_Loc)),
asm_words( load_word(R_S0, R_V0, O_(TapeHostFrame,s0)),
load_word( R_AtomJmp, R_TapePtr, 0) /* Bootstrap the first jump */ load_word(R_S1, R_V0, O_(TapeHostFrame,s1)),
, add_ui_self(R_TapePtr, S_(MipsAtom)) /* Advance tape */ load_word(R_S2, R_V0, O_(TapeHostFrame,s2)),
, call_reg( R_AtomJmp) /* jalr $t9 */ load_word(R_S3, R_V0, O_(TapeHostFrame,s3)),
, nop /* Branch delay slot */ load_word(R_S4, R_V0, O_(TapeHostFrame,s4)),
) load_word(R_S5, R_V0, O_(TapeHostFrame,s5)),
asm_rpins, r_use(tape_ptr) load_word(R_S6, R_V0, O_(TapeHostFrame,s6)),
asm_clobber: load_word(R_S7, R_V0, O_(TapeHostFrame,s7)),
rlit(R_AT), load_word(R_RA, R_V0, O_(TapeHostFrame,ra)),
rlit(R_V0), rlit(R_V1), rlit(R_A0), rlit(R_A1), rlit(R_A2), load_word(R_FP, R_V0, O_(TapeHostFrame,fp)),
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4), load_word(R_SP, R_V0, O_(TapeHostFrame,sp)),
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), jump_reg(R_RA), BdSlot_ nop,
rlit(R_S0), rlit(R_S1), rlit(R_S2), rlit(R_S3), rlit(R_S4), };
rlit(R_S5), rlit(R_S6), rlit(R_S7),
clb_mem_drain typedef void Proc_(TapeEntryFn)(MipsAtom* tape_ptr);
); }
FI_ void tape_run(Tape tape) { C_(TapeEntryFn*, tape_enter)(tape.ptr); }
// Procedural authoring of tapes: // Procedural authoring of tapes:
typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; }; typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; };
@@ -220,15 +261,11 @@ FI_ void tb_scope_run_end(TapeBuilder* tb) { tb_emit(tb,tape_exit); tape_run(tb_
* ---------------------------------------------------------------------------*/ * ---------------------------------------------------------------------------*/
// The 'Yield' sequence for Tape Atoms (mac_yield). // The 'Yield' sequence for Tape Atoms (mac_yield).
// - mac_yield() is the safe default for atom-endings: 4 words, BD-slot of jr is mandatory nop.
// - mac_yield_load() + mac_yield_tail():
// - unconditional branch: mac_yield_load fills the branch's BD-slot (replaces a nop);
// - mac_yield_tail runs at the branch target (does NOT re-load R_AtomJmp).
atom_dbg_skip MipsAtomComp_(ac_yield) { atom_dbg_skip MipsAtomComp_(ac_yield) {
load_word(R_AtomJmp, R_TapePtr, 0), load_word(R_AtomJmp, R_TapePtr, 0),
add_ui_self( R_TapePtr, S_(MipsCode)), add_ui_self( R_TapePtr, S_(MipsCode)),
jump_reg( R_AtomJmp), nop, jump_reg( R_AtomJmp), BdSlot_ nop,
}; };
atom_dbg_skip MipsAtomComp_(ac_yield_load) { atom_dbg_skip MipsAtomComp_(ac_yield_load) {
@@ -237,16 +274,13 @@ atom_dbg_skip MipsAtomComp_(ac_yield_load) {
atom_dbg_skip MipsAtomComp_(ac_yield_tail) { atom_dbg_skip MipsAtomComp_(ac_yield_tail) {
add_ui_self(R_TapePtr, S_(MipsCode)), add_ui_self(R_TapePtr, S_(MipsCode)),
jump_reg( R_AtomJmp), nop, jump_reg( R_AtomJmp), BdSlot_ nop,
}; };
#pragma endregion Macro Atom Components #pragma endregion Macro Atom Components
#pragma region Atom Builder #pragma region Atom Builder
// This helps with runtime procedural authoring of mips atoms. // This helps with runtime procedural authoring of mips atoms.
typedef Struct_(FMipsAtom512) { U4 data[512]; U4 used; };
// FArena Related
typedef Relative_(FArena) Struct_(AtomBuilder) { U4 start; U4 capacity; U4 used; }; typedef Relative_(FArena) Struct_(AtomBuilder) { U4 start; U4 capacity; U4 used; };
// Usual way to resolve an atom after the bulder is done. // Usual way to resolve an atom after the bulder is done.
@@ -267,7 +301,6 @@ FI_ void tb_emit_atombuilder(TapeBuilder_R tb, AtomBuilder_R ab) { tb_emit(tb, a
#pragma region Atom Arena #pragma region Atom Arena
// Just a dedicated FArena that is meant to mem_copy and return atom definitions made with MipsAtom_Proc_ // Just a dedicated FArena that is meant to mem_copy and return atom definitions made with MipsAtom_Proc_
typedef Relative_(FArena) Struct_(AtomArena) { U4 start; U4 capacity; U4 used; }; typedef Relative_(FArena) Struct_(AtomArena) { U4 start; U4 capacity; U4 used; };
#define atomarena_unused_start(ab) ((ab).start + (ab).used) #define atomarena_unused_start(ab) ((ab).start + (ab).used)
@@ -284,37 +317,130 @@ FI_ MipsAtom* atomarena_push(AtomArena_R aa, Slice_MipsCode code) {
return C_(MipsAtom*, dest); return C_(MipsAtom*, dest);
} }
FI_ void atomarena_reset(AtomArena_R aa) { aa->used = 0; } FI_ void atomarena_reset(AtomArena_R aa) { aa->used = 0; }
#pragma region Atom Arena #pragma endregion Atom Arena
#pragma region RegFile (Register File Allocator) #pragma region RegFile (Register File Allocator)
// A specialized allocator utilized to help the user track which registers are bound to values // A specialized allocator utilized to help the user track which registers are bound to values
// that must be preserved for the arena's bounds. // that must be preserved for the arena's bounds.
// TODO(Ed): Technically we can do this at comp-time with the metaprogram, but we may have namespace conflicts.
// Unless we follow a convention for #define <Scope_Prefix> or something per register allocation boundary.
enum { /* ABI reserves that are never handed out by alloc.
RegFileArena_Len, * R_AT is the assembler temporary (per the MIPS O32 ABI).
}; * R_K0/K1 are kernel reserves.
typedef Enum_(U4, RegFileEntry) { * R_GP stays the host global pointer.
// TODO(Ed): Define RF_Field, each field is maped by index + bit pos. * R_SP/R_FP/R_RA are tape runtime carriers between tape_enter and tape_exit. */
// the index is the upper portion of a U4 and the bit pos in the lower pos. U4 const regfile_abi_mask =
(1u << R_0) | (1u << R_AT) |
(1u << R_K0) | (1u << R_K1) |
(1u << R_GP) | (1u << R_SP) |
(1u << R_FP) | (1u << R_RA);
regfileentry_todo_, internal Reg const regfile_alloc_order[] = {
// TODO(Ed): Is there a trick we can do with the current register enums to R_V0, R_V1,
// just resolve an entry automatically when doing a pin? R_A0, R_A1, R_A2, R_A3,
R_T0, R_T1, R_T2, R_T3, R_T4, R_T5, R_T6, R_T7,
R_S0, R_S1, R_S2, R_S3, R_S4, R_S5, R_S6, R_S7,
R_T8, R_T9,
}; };
typedef Struct_(RegFile) { typedef Struct_(RegFile) {
U1 GPR[RegFileArena_Len]; A2_U2 GPR;
U1 GTE[RegFileArena_Len]; A2_U2 GTE;
U1 GP[RegFileArena_Len];
}; };
#define regfile(pin_mask) {.GPR={u4_lo(pin_mask), u4_hi(pin_mask)} }
void regfile_pin(U4 register) { FI_ void regfile_init(RegFile_R rf) {
/* pack the 32-bit ABI mask into the two U2s */
assert(false); rf->GPR[0] = u4_lo(regfile_abi_mask);
rf->GPR[1] = u4_hi(regfile_abi_mask);
rf->GTE[0] = rf->GTE[1] = 0;
} }
FI_ RegFile regfile_make(void) { RegFile rf; regfile_init(& rf); return rf; }
typedef Struct_(RegFile_RInfo) {
U2_R section;
U2 mask;
B2 occupied;
};
FI_ RegFile_RInfo regfile_rinfo(A2_U2 file, Reg r_id) {
U2 s_id = r_id >> 4;
U2_R section = & file[s_id];
U2 mask = u2_(1u << (r_id & 15));
B2 occupied = (section[0] & mask) != 0;
return (RegFile_RInfo){section, mask, occupied};
}
FI_ Reg regfile__alloc_helper(A2_U2 file, Reg r_id) {
Reg result = 0; RegFile_RInfo info = regfile_rinfo(file, r_id);
if (info.occupied == false) {
info.section[0] |= info.mask;
result = r_id;
}
return result;
}
/* regfile_alloc picks the next free GPR from regfile_alloc_order.
* The table is the first-fit allocation order: T0..T7, V0..V1, A0..A3,
* S0..S7, T8..T9. The 24 entries leave room for the tape program to use
* any of them while R0, R1, R26-R31 remain reserved. */
I_ Reg regfile_alloc(RegFile_R rf) {
Reg allocated = 0;
for index_iter(U4, r_id, R_V0, <, R_T9) {
allocated = regfile__alloc_helper(rf->GPR, r_id);
Jmp_nZero_(allocated,resolved);
}
assert(allocated != 0);
resolved: return allocated;
}
FI_ Reg regfile_pin(RegFile_R rf, Reg r_id) {
RegFile_RInfo info = regfile_rinfo(rf->GPR, r_id);
assert(info.occupied == false);
info.section[0] |= info.mask;
return r_id;
}
FI_ void regfile_pin_mask(RegFile_R rf, U4 mask) {
B4 occupied = u4_r(rf->GPR)[0] & mask;
assert(occupied == false);
u4_r(rf->GPR)[0] |= mask;
}
FI_ void regfile_free_mask(RegFile_R rf, U4 mask) {
if (regfile_abi_mask & mask) return;
u4_r(rf->GPR)[0] &= ~mask;
}
FI_ void regfile_free_reg(RegFile_R rf, Reg r_id) {
/* never free the ABI set */
if (regfile_abi_mask & (1u << r_id)) return;
RegFile_RInfo info = regfile_rinfo(rf->GPR, r_id);
info.section[0] &= ~info.mask;
}
FI_ void regfile_reset(RegFile_R rf) {
rf->GPR[0] = u4_lo(regfile_abi_mask);
rf->GPR[1] = u4_hi(regfile_abi_mask);
}
FI_ void regfile_reset_to_mask(RegFile_R rf, U4 mask) {
rf->GPR[0] = u4_lo(mask);
rf->GPR[1] = u4_hi(mask);
}
#pragma endregion RegFileArena (Register File Allocator) #pragma endregion RegFileArena (Register File Allocator)
#pragma region Mips Atom Procs #pragma region Mips Atom Procs
/* RegUse structs are a convention to organize register allocations for a mips atom procedure.
Unlike the usual enum-based declarations, they provide a namespaced scope and have view types via union declarations. */
#define RegUse_(proc_name) (tmpl(RegUse,proc_name))
typedef Struct_(RegUse_example_atom_proc) {
Reg const ro_register; // Scratch base carrier.
Reg usual_modifiable;
union { Reg view_1, view_2, view_3; } t1;
};
internal MipsAtom* example_atom_proc(AtomArena_R aa, U2 offset, RegUse_example_atom_proc r)
MipsAtom_Proc_(aa, {
add_si(r.usual_modifiable, r.ro_register, offset),
or_u(r.t1.view_1, r.ro_register, 0),
branch_lt_zero(r.t1.view_1, atom_offset(example_atom_proc, skip)), BdSlot_ nop,
li_s(r.t1.view_2, 100),
atom_label(skip)
add_si(r.t1.view_3, r.usual_modifiable, 10),
mac_yield(),
})
#pragma endregion Mips Atom Procs #pragma endregion Mips Atom Procs
-53
View File
@@ -1,53 +0,0 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "gen/macs.h"
# include "gen/offsets.h"
# include "math.h"
# include "lottes_tape.h"
#endif
ATOM_FILE_DEBUGGER_LINE_MARKER(math_atom_c);
#pragma region MACs (Mips Atom Component)
FI_ Slice_MipsCode ac_load_v2s2(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v2s2, ab, {
load_half( rs_x, r_base, offset + O_(V3_S2,x)),
load_half( rs_y, r_base, offset + O_(V3_S2,y)),
})
FI_ Slice_MipsCode ac_store_v2s2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v2s2, ab, {
store_half(rt_x, base, offset + O_(V2_S2,x)),
store_half(rt_y, base, offset + O_(V2_S2,y)),
})
FI_ Slice_MipsCode ac_load_v3s4(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 rs_z, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v3s4, ab, {
load_word( rs_x, r_base, offset + O_(V3_S4,x)),
load_word( rs_y, r_base, offset + O_(V3_S4,y)),
load_word( rs_z, r_base, offset + O_(V3_S4,z)),
})
// TODO(Ed): we could generate these mappings properly..
#define ac_load_p3s4 ac_load_v3s4
#define mac_load_p3s4 mac_load_v3s4
FI_ Slice_MipsCode ac_store_v3s4(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_z, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v3s4, ab, {
store_word(rt_x, base, offset + O_(V3_S4,x)),
store_word(rt_y, base, offset + O_(V3_S4,y)),
store_word(rt_z, base, offset + O_(V3_S4,z)),
})
// TODO(Ed): we could generate these mappings properly..
#define ac_store_p3s4 ac_store_v3s4
#define mac_store_p3s4 mac_store_v3s4
FI_ Slice_MipsCode ac_sub_v3s4(AtomBuilder_R ab, U4 rds_x, U4 rds_y, U4 rds_z, U4 rt_x, U4 rt_y, U4 rt_z) atom_dbg_skip MipsAtomComp_Proc_(ac_sub_v3s4, ab, {
sub_s(rds_x, rds_x, rt_x),
sub_s(rds_y, rds_y, rt_y),
sub_s(rds_z, rds_z, rt_z),
})
FI_ Slice_MipsCode ac_store_rects2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rects2, ab, {
store_half(rt_x, base, offset + O_(Rect_S2,x)),
store_half(rt_y, base, offset + O_(Rect_S2,y)),
store_half(rt_width, base, offset + O_(Rect_S2,width)),
store_half(rt_height, base, offset + O_(Rect_S2,height)),
})
#pragma endregion MACs (Mips Atom Component)
+96
View File
@@ -0,0 +1,96 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "gen/macs.h"
# include "gen/offsets.h"
# include "math.h"
# include "lottes_tape.h"
#endif
ATOM_FILE_DEBUGGER_LINE_MARKER(math_atom_c);
#define v3s4_R_0() ((Reg_(V3_S4)){R_0,R_0,R_0})
typedef Struct_(Reg_V3_S2) { Reg x, y, z; };
typedef Struct_(Reg_V3_S4) { Reg x, y, z; }; // Register allocation of a V3_S4
typedef Struct_(Reg_P3_S4) { Reg x, y, z; }; // Register allocation of a P3_S4
#pragma region MACs (Mips Atom Component)
FI_ Slice_MipsCode ac_load_half_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_half(tx, base, offset + OA_(U2,[0])),
load_half(ty, base, offset + OA_(U2,[1])),
load_half(tz, base, offset + OA_(U2,[2])),
})
FI_ Slice_MipsCode ac_load_v3s2(AtomBuilder_R ab, Reg_(V3_S2) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_half_v3(transfer.x, transfer.y, transfer.z, base, offset))
FI_ Slice_MipsCode ac_load_v2s2(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_half(rs_x, r_base, offset + O_(V3_S2,x)),
load_half(rs_y, r_base, offset + O_(V3_S2,y)),
})
FI_ Slice_MipsCode ac_store_v2s2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
store_half(rt_x, base, offset + O_(V2_S2,x)),
store_half(rt_y, base, offset + O_(V2_S2,y)),
})
FI_ Slice_MipsCode ac_load_word_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_word(tx, base, offset + OA_(U4,[0])),
load_word(ty, base, offset + OA_(U4,[1])),
load_word(tz, base, offset + OA_(U4,[2])),
})
FI_ Slice_MipsCode ac_load_v3s4(AtomBuilder_R ab, Reg_(V3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
FI_ Slice_MipsCode ac_load_p3s4(AtomBuilder_R ab, Reg_(P3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
FI_ Slice_MipsCode ac_store_half_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
store_half(tx, base, offset + OA_(U2,[0])),
store_half(ty, base, offset + OA_(U2,[1])),
store_half(tz, base, offset + OA_(U2,[2])),
})
FI_ Slice_MipsCode ac_store_v3s2(AtomBuilder_R ab, Reg_(V3_S2) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_store_half_v3(transfer.x, transfer.y, transfer.z, base, offset))
FI_ Slice_MipsCode ac_store_word_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
store_word(tx, base, offset + OA_(U4,[0])),
store_word(ty, base, offset + OA_(U4,[1])),
store_word(tz, base, offset + OA_(U4,[2])),
})
FI_ Slice_MipsCode ac_store_v3s4(AtomBuilder_R ab, Reg_(V3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
FI_ Slice_MipsCode ac_store_p3s4(AtomBuilder_R ab, Reg_(P3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
FI_ Slice_MipsCode ac_add_si_v3s4(AtomBuilder_R ab, Reg rt_x, Reg rt_y, Reg rt_z, Reg base, U2 offset)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
add_si(rt_x, base, O_(V3_S4,x)),
add_si(rt_y, base, O_(V3_S4,y)),
add_si(rt_z, base, O_(V3_S4,z)),
})
FI_ Slice_MipsCode ac_sub_s_v3(AtomBuilder_R ab
, Reg dx, Reg dy, Reg dz
, Reg sx, Reg sy, Reg sz
, Reg tx, Reg ty, Reg tz
) atom_dbg_skip MipsAtomComp_Proc_(ab, {
sub_s(dx, sx, tx),
sub_s(dy, sy, ty),
sub_s(dz, sz, tz),
})
FI_ Slice_MipsCode ac_sub_v3s4(AtomBuilder_R ab, Reg_(V3_S4) d, Reg_(V3_S4) s, Reg_(V3_S4) t) MipsAtomComp_ProcMap_(ab, mac_sub_s_v3(d.x, d.y, d.z, s.x, s.y, s.z, t.x, t.y, t.z))
FI_ Slice_MipsCode ac_sub_s_v3_self(AtomBuilder_R ab, Reg ds_x, Reg ds_y, Reg ds_z, Reg tx, Reg ty, Reg tz) atom_dbg_skip MipsAtomComp_Proc_(ab, {
sub_s(ds_x, ds_x, tx),
sub_s(ds_y, ds_y, ty),
sub_s(ds_z, ds_z, tz),
})
FI_ Slice_MipsCode ac_sub_v3s4_self(AtomBuilder_R ab, Reg_(V3_S4) ds, Reg_(V3_S4) t) MipsAtomComp_ProcMap_(ab, mac_sub_s_v3_self(ds.x, ds.y, ds.z, t.x, t.y, t.z))
FI_ Slice_MipsCode ac_store_rects2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
store_half(rt_x, base, offset + O_(Rect_S2,x)),
store_half(rt_y, base, offset + O_(Rect_S2,y)),
store_half(rt_width, base, offset + O_(Rect_S2,width)),
store_half(rt_height, base, offset + O_(Rect_S2,height)),
})
#pragma endregion MACs (Mips Atom Component)
+11 -5
View File
@@ -24,6 +24,7 @@ enum {
}; };
typedef Array_(U1, 2); typedef Array_(U1, 2);
typedef Array_(U2, 2);
typedef Array_(U4, 2); typedef Array_(U4, 2);
typedef Array_(S2, 2); typedef Array_(S2, 2);
typedef Array_(S2, 3); typedef Array_(S2, 3);
@@ -46,6 +47,9 @@ typedef Struct_(V4_S4) { S4 x; S4 y; S4 z; S4 w; };
// typedef Struct_(P3_S4) { S4 x; S4 y; S4 z; S4 w1; }; // RGA(Lengyel): Affine point with implicit weight one. Storage alias of V3_S4. Use P3_S4 when the value is a point. // typedef Struct_(P3_S4) { S4 x; S4 y; S4 z; S4 w1; }; // RGA(Lengyel): Affine point with implicit weight one. Storage alias of V3_S4. Use P3_S4 when the value is a point.
typedef V3_S4 P3_S4; typedef V3_S4 P3_S4;
typedef Struct_(R1_U2) { U2 p0; U2 p1; };
typedef Struct_(R1_S2) { S2 p0; S2 p1; };
typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; }; // Range-2 Signed 2-Byte (16-bit) typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; }; // Range-2 Signed 2-Byte (16-bit)
typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; }; // Range-2 Signed 4-Byte (32-bit) typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; }; // Range-2 Signed 4-Byte (32-bit)
@@ -64,6 +68,8 @@ typedef Array_(V2_S2, 2);
typedef Array_(V2_S2, 3); typedef Array_(V2_S2, 3);
typedef Array_(V2_S2, 4); typedef Array_(V2_S2, 4);
#define r1u2(p0,p1) (R1_U2){p0,p1}
enum { enum {
fp_one = (1 << 12), fp_one = (1 << 12),
}; };
@@ -106,10 +112,10 @@ FI_ void mul_a3s4(A3_S4_R out_a, A3_S4 b) {
(out_a[0])[2] *= b[2]; (out_a[0])[2] *= b[2];
} }
FI_ void add_v3s4 (V3_S4_R out_a, V3_S4 b) { add_a3s4 (pcast(A3_S4_R, out_a), pcast(A3_S4, b)); } FI_ void add_v3s4 (V3_S4_R out_a, V3_S4 b) { add_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
FI_ void add_v3s4_fp(V3_S4_R out_a, V3_S4 b) { add_a3s4_fp(pcast(A3_S4_R, out_a), pcast(A3_S4, b)); } FI_ void add_v3s4_fp(V3_S4_R out_a, V3_S4 b) { add_a3s4_fp(C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
FI_ void sub_v3s4 (V3_S4_R out_a, V3_S4 b) { sub_a3s4 (pcast(A3_S4_R, out_a), pcast(A3_S4, b)); } FI_ void sub_v3s4 (V3_S4_R out_a, V3_S4 b) { sub_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
FI_ void sub_v3s4_fp(V3_S4_R out_a, V3_S4 b) { sub_a3s4_fp(pcast(A3_S4_R, out_a), pcast(A3_S4, b)); } FI_ void sub_v3s4_fp(V3_S4_R out_a, V3_S4 b) { sub_a3s4_fp(C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
FI_ void mul_v3s4 (V3_S4_R out_a, V3_S4 b) { mul_a3s4 (pcast(A3_S4_R, out_a), pcast(A3_S4, b)); } FI_ void mul_v3s4 (V3_S4_R out_a, V3_S4 b) { mul_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
+13 -1
View File
@@ -73,7 +73,7 @@ typedef Slice_(B1);
#define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter) #define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter)
#define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = Array_decl(type,__VA_ARGS__), .len = Array_len( Array_decl(type,__VA_ARGS__)) } #define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = Array_decl(type,__VA_ARGS__), .len = Array_len( Array_decl(type,__VA_ARGS__)) }
#define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = S_(array) / S_(type) } #define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = Array_len(array) }
FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(u4_(s.ptr), s.len); } FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(u4_(s.ptr), s.len); }
#define slice_zero(s) slice_zero_(slice_to_ut(s)) #define slice_zero(s) slice_zero_(slice_to_ut(s))
@@ -131,3 +131,15 @@ FI_ U4 farena_unused_start(FArena arena) { return arena.start + arena.used; }
#define farena_push_array(arena, type, amount, ...) (tmpl(Slice,type)){ C_(type*, farena_push((arena), (amount), opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr), (amount) } #define farena_push_array(arena, type, amount, ...) (tmpl(Slice,type)){ C_(type*, farena_push((arena), (amount), opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr), (amount) }
#pragma endregion FArena #pragma endregion FArena
#pragma region BIOS Scratchpad
/* BIOS scratchpad location. 1 KB at 0x1F800000.
* TapeHostFrame occupies the final 44 bytes while tape code executes.
* Atom scratch is bounded by the TapeHostFrame_Loc declaration in lottes_tape.h. */
enum {
Scratchpad_Loc = 0x1F800000,
Scratchpad_Len = 0x400, /* 1 KB */
Scratchpad_End = Scratchpad_Loc + Scratchpad_Len, /* 0x1F800400 */
};
#define C_scratch(type) C_(type, Scratchpad_Loc)
#pragma endregion BIOS Scratchpad
+39
View File
@@ -2,11 +2,50 @@
# include "gen/macs.h" # include "gen/macs.h"
# include "gen/offsets.h" # include "gen/offsets.h"
# include "bios.h" # include "bios.h"
# include "mips.h"
# include "lottes_tape.h" # include "lottes_tape.h"
#endif #endif
ATOM_FILE_DEBUGGER_LINE_MARKER(mips_atom_c); ATOM_FILE_DEBUGGER_LINE_MARKER(mips_atom_c);
#pragma region MACs (Mips Atom Components)
FI_ Slice_MipsCode ac_load_word_imm(AtomBuilder_R ab, Reg dst, U4 imm)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_upper_i(dst, u4_hi(imm)),
or_i_self( dst, u4_lo(imm)),
})
FI_ Slice_MipsCode ac_shift_aright_v3_self(AtomBuilder_R ab, Reg dt_x, Reg dt_y, Reg dt_z, U2 shift_amount)
MipsAtomComp_Proc_( ab, {
shift_aright(dt_x, dt_x, shift_amount),
shift_aright(dt_y, dt_y, shift_amount),
shift_aright(dt_z, dt_z, shift_amount),
})
FI_ Slice_MipsCode ac_shift_aright_v3s4_self(AtomBuilder_R ab, Reg_(V3_S4) dt, U2 shift) MipsAtomComp_ProcMap_(ab, mac_shift_aright_v3_self(dt.x, dt.y, dt.z, shift))
FI_ Slice_MipsCode ac_shift_aright_var_v3(AtomBuilder_R ab
, Reg rd_v0, Reg rd_v1, Reg rd_v2
, Reg rs_v0, Reg rs_v1, Reg rs_v2
, Reg r_shift)
MipsAtomComp_Proc_(ab, {
shift_aright_var(rd_v0, rs_v0, r_shift),
shift_aright_var(rd_v1, rs_v1, r_shift),
shift_aright_var(rd_v2, rs_v2, r_shift),
})
FI_ Slice_MipsCode ac_shift_aright_var_v3_self(AtomBuilder_R ab, Reg rds_v0, Reg rds_v1, Reg rds_v2, Reg r_shift)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
shift_aright_var(rds_v0, rds_v0, r_shift),
shift_aright_var(rds_v1, rds_v1, r_shift),
shift_aright_var(rds_v2, rds_v2, r_shift),
})
FI_ Slice_MipsCode ac_shift_aright_var_v3s4_self(AtomBuilder_R ab, Reg_(V3_S4) ds, Reg shift) MipsAtomComp_ProcMap_(ab, mac_shift_aright_var_v3_self(ds.x, ds.y, ds.z, shift))
#pragma endregion MACs (Mips Atom Components)
#pragma region Baked Atoms #pragma region Baked Atoms
/* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0). /* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0).
+20 -12
View File
@@ -259,22 +259,22 @@ enum { _BitOffsets = 0
, SHAMT_SHIFT = 6 /* Shift Amount */ , SHAMT_SHIFT = 6 /* Shift Amount */
, FC_SHIFT = 0 , FC_SHIFT = 0
/* Bit Masks to prevent overflow into adjacent fields */ /* IMM_MASK is the 16-bit two's-complement truncation for the immediate field.
* It is NOT a range guard — it is load-bearing for negative branch offsets
* (the metaprogram emits raw signed offsets; the mask truncates them to the
* 16-bit representation the hardware expects). The static analysis
* `immediate_field_width` check validates ranges at build time. */
, OPCODE_MASK = 0x3F
, REG_MASK = 0x1F
, SHAMT_MASK = 0x1F /* Shift Amount */
, FC_MASK = 0x3F
, IMM_MASK = 0xFFFF , IMM_MASK = 0xFFFF
}; };
#define enc_op(op) (((op) & OPCODE_MASK) << OPCODE_SHIFT) #define enc_op(op) ((op) << OPCODE_SHIFT)
#define enc_rs(rs) (((rs) & REG_MASK) << RS_SHIFT) #define enc_rs(rs) ((rs) << RS_SHIFT)
#define enc_rt(rt) (((rt) & REG_MASK) << RT_SHIFT) #define enc_rt(rt) ((rt) << RT_SHIFT)
#define enc_rd(rd) (((rd) & REG_MASK) << RD_SHIFT) #define enc_rd(rd) ((rd) << RD_SHIFT)
#define enc_shamt(shamt) (((shamt) & SHAMT_MASK) << SHAMT_SHIFT) #define enc_shamt(shamt) ((shamt) << SHAMT_SHIFT)
#define enc_fc(fc) (((fc) & FC_MASK) << FC_SHIFT) #define enc_fc(fc) ((fc) << FC_SHIFT)
#define enc_imm(imm) (((imm) & IMM_MASK)) #define enc_imm(imm) ((imm) & IMM_MASK)
/* MIPS R-Type Instruction Format (Register-to-Register) */ /* MIPS R-Type Instruction Format (Register-to-Register) */
#define enc_r(op, rs, rt, rd, shamt, fc) (enc_op(op) | enc_rs(rs) | enc_rt(rt) | enc_rd(rd) | enc_shamt(shamt) | enc_fc(fc)) #define enc_r(op, rs, rt, rd, shamt, fc) (enc_op(op) | enc_rs(rs) | enc_rt(rt) | enc_rd(rd) | enc_shamt(shamt) | enc_fc(fc))
@@ -318,7 +318,10 @@ enum { _BitOffsets = 0
#define load_half(rt, base, off) enc_i(op_lh, (base), (rt), (off)) #define load_half(rt, base, off) enc_i(op_lh, (base), (rt), (off))
#define load_byte_u(rt, base, off) enc_i(op_lbu, (base), (rt), (off)) #define load_byte_u(rt, base, off) enc_i(op_lbu, (base), (rt), (off))
#define load_half_u(rt, base, off) enc_i(op_lhu, (base), (rt), (off)) #define load_half_u(rt, base, off) enc_i(op_lhu, (base), (rt), (off))
#define LdSlot_
#define store_word(rt, base, off) enc_i(op_sw, (base), (rt), (off)) #define store_word(rt, base, off) enc_i(op_sw, (base), (rt), (off))
#define add_ui(rt, rs, imm) enc_i(op_addiu, (rs), (rt), (imm)) #define add_ui(rt, rs, imm) enc_i(op_addiu, (rs), (rt), (imm))
#define and_i(rt, rs, imm) enc_i(op_andi, (rs), (rt), (imm)) #define and_i(rt, rs, imm) enc_i(op_andi, (rs), (rt), (imm))
// #define and_si and_i // #define and_si and_i
@@ -379,6 +382,9 @@ enum { _BitOffsets = 0
*/ */
#define jump(off) enc_i(op_j, R_0, R_0, (off)) #define jump(off) enc_i(op_j, R_0, R_0, (off))
// Annotate an instruction as filling a branch-delay slot.
#define BdSlot_
/* jump_rel off — unconditional relative jump (the within-atom-safe `jump`). /* jump_rel off — unconditional relative jump (the within-atom-safe `jump`).
* MIPS I R3000A has no "branch always" opcode. The idiom for an unconditional relative jump is `beq $0, $0, off`. */ * MIPS I R3000A has no "branch always" opcode. The idiom for an unconditional relative jump is `beq $0, $0, off`. */
#define jump_rel(off) branch_equal(R_0, R_0, (off)) #define jump_rel(off) branch_equal(R_0, R_0, (off))
@@ -411,6 +417,7 @@ enum { _BitOffsets = 0
#define div_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_div) #define div_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_div)
#define div_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_divu) #define div_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_divu)
// TODO(Ed): Change convention of 'self' to ds for (destination is source)?
#define add_u_self(rd_rs, rt) add_u(rd_rs, rd_rs, rt) #define add_u_self(rd_rs, rt) add_u(rd_rs, rd_rs, rt)
/* --- Arithmetic I-type (immediate) --- */ /* --- Arithmetic I-type (immediate) --- */
@@ -458,6 +465,7 @@ enum { _BitOffsets = 0
// li_s — load signed 16-bit immediate into GPR (addiu rt, $0, imm — sign-extends). // li_s — load signed 16-bit immediate into GPR (addiu rt, $0, imm — sign-extends).
#define li_s(rt, imm) add_ui((rt), R_0, (imm)) #define li_s(rt, imm) add_ui((rt), R_0, (imm))
// #define load_imm_s(rt, imm) add_ui((rt), R_0, (imm))
#define load_imm_1w(rt, imm) add_ui((rt), R_0, (imm)) #define load_imm_1w(rt, imm) add_ui((rt), R_0, (imm))
#define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm)) #define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm))
+17 -16
View File
@@ -11,18 +11,18 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(pad_atom_c);
#pragma region MACs (Mips Atom Components) #pragma region MACs (Mips Atom Components)
FI_ Slice_MipsCode ac_pad_set_centered_axes(AtomBuilder_R ab, U4 r_state, U4 r_scratch) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_centered_axes, ab, { FI_ Slice_MipsCode ac_pad_set_centered_axes(AtomBuilder_R ab, Reg state, Reg scratch) atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_upper_i(r_scratch, (PadAxis_Centered_Word >> 16) & 0xFFFF), load_upper_i(scratch, (PadAxis_Centered >> 16) & 0xFFFF),
or_i_self( r_scratch, PadAxis_Centered_Word & 0xFFFF), or_i_self( scratch, PadAxis_Centered & 0xFFFF), // mac_load_word_imm(scratch, PadAxis_Centered),
store_word( r_scratch, r_state, O_(PadState,axes)), store_word( scratch, state, O_(PadState,axes)),
}) })
FI_ Slice_MipsCode ac_pad_set_id_byte(AtomBuilder_R ab, U1 r_state, U1 r_id, U1 id_value) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_id_byte, ab, { FI_ Slice_MipsCode ac_pad_set_id_byte(AtomBuilder_R ab, Reg state, Reg r_id, U1 id_value) atom_dbg_skip MipsAtomComp_Proc_(ab, {
add_ui( r_id, R_0, id_value), add_ui( r_id, R_0, id_value),
store_byte(r_id, r_state, O_(PadState,id)), store_byte(r_id, state, O_(PadState,id)),
}) })
FI_ Slice_MipsCode ac_pad_set_status(AtomBuilder_R ab, U4 r_tmp, U1 r_state, U4 pad_status) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_status, ab, { FI_ Slice_MipsCode ac_pad_set_status(AtomBuilder_R ab, U4 r_tmp, U1 r_state, U4 pad_status) atom_dbg_skip MipsAtomComp_Proc_(ab, {
add_ui( r_tmp, R_0, pad_status), add_ui( r_tmp, R_0, pad_status),
store_word(r_tmp, r_state, O_(PadState,status)), store_word(r_tmp, r_state, O_(PadState,status)),
}) })
@@ -30,9 +30,9 @@ FI_ Slice_MipsCode ac_pad_set_status(AtomBuilder_R ab, U4 r_tmp, U1 r_state, U4
/* Invert r_buttons (active-low → active-high) and store to PadState.buttons. /* Invert r_buttons (active-low → active-high) and store to PadState.buttons.
* r_buttons must already be loaded (the caller is responsible for filling the load-delay slot of * r_buttons must already be loaded (the caller is responsible for filling the load-delay slot of
* the preceding load_half_u with an instruction that doesn't read r_buttons). */ * the preceding load_half_u with an instruction that doesn't read r_buttons). */
FI_ Slice_MipsCode ac_pad_store_inverted_buttons(AtomBuilder_R ab, U1 r_buttons, U1 r_pad_state) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_store_inverted_buttons, ab, { FI_ Slice_MipsCode ac_pad_store_inverted_buttons(AtomBuilder_R ab, U1 r_buttons, U1 r_pad_state) atom_dbg_skip MipsAtomComp_Proc_(ab, {
nor_u( r_buttons, r_buttons, R_0), nor_u( r_buttons, r_buttons, R_0),
store_half( r_buttons, r_pad_state, O_(PadState, buttons)), store_half(r_buttons, r_pad_state, O_(PadState,buttons)),
}) })
#pragma endregion MACs (Mips Atom Components) #pragma endregion MACs (Mips Atom Components)
@@ -54,12 +54,12 @@ FI_ Slice_MipsCode ac_pad_store_inverted_buttons(AtomBuilder_R ab, U1 r_buttons,
* byte_swap16(x) = (x >> 8) | (x << 8); nor(x, R_0) = ~x. store_half truncates to 16 bits so the upper-16 mask is implicit in the store. * byte_swap16(x) = (x >> 8) | (x << 8); nor(x, R_0) = ~x. store_half truncates to 16 bits so the upper-16 mask is implicit in the store.
* *
* Register use (atom-local; no wave-context touched): * Register use (atom-local; no wave-context touched):
* R_T0 = raw base (kept throughout; axes loads read raw[4..7] from R_T0) * R_T0 = raw base : Kept throughout; axes loads read raw[4..7] from R_T0.
* R_T1 = state base (kept throughout; all stores go through R_T1) * R_T1 = state base : Kept throughout; all stores go through R_T1.
* R_T2 = raw[0] status (alive across the disc/pending/id dispatch, then dead) * R_T2 = raw[0] status : Alive across the disc/pending/id dispatch, then dead.
* R_T3 = raw[1] id (alive across the id dispatch, then dead) * R_T3 = raw[1] id : Alive across the id dispatch, then dead.
* R_T4 = scratch (shifts, compares, immediate loads, store values) * R_T4 = scratch : Shifts, compares, immediate loads, store values.
* R_T5 = scratch (parallel lui+ori for the 0x80808080 axes constant + byte-swap target) * R_T5 = scratch : Parallel lui + ori for the 0x80808080 axes constant + byte-swap target.
*/ */
enum { enum {
R_PadRaw = R_T0 atom_reg atom_type(U1), R_PadRaw = R_T0 atom_reg atom_type(U1),
@@ -124,7 +124,8 @@ atom_label(id_dispatch) /* === Case 3-6: ID dispatch */
* R_T5 is then "dead" — only consumed at the analog_pad range check downstream. */ * R_T5 is then "dead" — only consumed at the analog_pad range check downstream. */
mac_pad_set_status(R_T4, R_PadState, PadStatus_Digital), mac_pad_set_status(R_T4, R_PadState, PadStatus_Digital),
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw, buttons)), /* R_T4 = raw_buttons; */ load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw, buttons)), /* R_T4 = raw_buttons; */
load_upper_i(R_T5, PadAxis_Centered_Hi), or_i_self(R_T5, PadAxis_Centered_Lo), /* fills the buttons-load's delay slot (doesn't read R_T4) */ mac_load_word_imm( R_T5, PadAxis_Centered), /* fills the buttons-load's delay slot (doesn't read R_T4) */
// load_upper_i(R_T5, PadAxis_Centered_Hi), or_i_self(R_T5, PadAxis_Centered_Lo),
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */ mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
store_word(R_T5, R_PadState, O_(PadState, axes)), /* single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y) */ store_word(R_T5, R_PadState, O_(PadState, axes)), /* single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y) */
mac_pad_set_id_byte(R_PadState, R_T4, PadRawId_Digital), mac_pad_set_id_byte(R_PadState, R_T4, PadRawId_Digital),
+1 -1
View File
@@ -36,7 +36,7 @@ NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
* $t2 = 0xB0 (BIOS B-table address) */ * $t2 = 0xB0 (BIOS B-table address) */
asm volatile( asm volatile(
asm_words( asm_words(
or_u( R_A2, R_A0, R_0), /* $a2 = $a1 = raw1 */ or_u( R_A2, R_A1, R_0), /* $a2 = $a1 = raw1 */
add_ui( R_A1, R_0, bios_pad_buffer_size), /* $a1 = 0x22 */ add_ui( R_A1, R_0, bios_pad_buffer_size), /* $a1 = 0x22 */
add_ui( R_A3, R_0, bios_pad_buffer_size), /* $a3 = 0x22 */ add_ui( R_A3, R_0, bios_pad_buffer_size), /* $a3 = 0x22 */
add_ui( R_T1, R_0, bios_init_pad_2), /* $t1 = 0x12 */ add_ui( R_T1, R_0, bios_init_pad_2), /* $t1 = 0x12 */
+1 -1
View File
@@ -83,7 +83,7 @@ typedef Enum_(U1, PadUnknownId) {
typedef Enum_(U4, PadAxisCentered) { typedef Enum_(U4, PadAxisCentered) {
PadAxis_Centered_Hi = 0x8080, PadAxis_Centered_Hi = 0x8080,
PadAxis_Centered_Lo = 0x8080, PadAxis_Centered_Lo = 0x8080,
PadAxis_Centered_Word = 0x80808080U, PadAxis_Centered = 0x80808080U,
}; };
typedef Enum_(U1, PadDeadZone) { typedef Enum_(U1, PadDeadZone) {
PadDeadZone_LowBound = 0x70, /* left_x < LowBound → active; delta = 0x80 - left_x > 0 (rightward pull) */ PadDeadZone_LowBound = 0x70, /* left_x < LowBound → active; delta = 0x80 - left_x > 0 (rightward pull) */
+1
View File
@@ -55,6 +55,7 @@ WORD_COUNT(gte_mv_to_ctrl_r, 1)
WORD_COUNT(gte_sw, 1) WORD_COUNT(gte_sw, 1)
WORD_COUNT(gte_cmdw_rtpt, 1) WORD_COUNT(gte_cmdw_rtpt, 1)
WORD_COUNT(gte_cmdw_nclip, 1) WORD_COUNT(gte_cmdw_nclip, 1)
WORD_COUNT(gte_cmdw_op, 1)
WORD_COUNT(gte_avg_sort_z3, 1) WORD_COUNT(gte_avg_sort_z3, 1)
WORD_COUNT(gte_cmdw_sqr, 1) WORD_COUNT(gte_cmdw_sqr, 1)
WORD_COUNT(gte_cmdw_gpf, 1) WORD_COUNT(gte_cmdw_gpf, 1)
+2 -2
View File
@@ -8,7 +8,7 @@
#pragma region hello_camera #pragma region hello_camera
// --- atom: pad_input_cube_rotation (60 words) --- // --- atom: pad_input_cube_rotation (61 words) ---
#define _atom_offset_dpad_left_exit_dpad_left 6 #define _atom_offset_dpad_left_exit_dpad_left 6
#define _atom_offset_dpad_right_exit_dpad_right 6 #define _atom_offset_dpad_right_exit_dpad_right 6
@@ -44,7 +44,7 @@ enum {
atom_offset_circle_z_exit_circle_z = _atom_offset_circle_z_exit_circle_z, atom_offset_circle_z_exit_circle_z = _atom_offset_circle_z_exit_circle_z,
}; };
// --- atom: cube_g4_face (76 words) --- // --- atom: cube_g4_face (75 words) ---
#define _atom_offset_cull_cube_g4_face_exit 41 #define _atom_offset_cull_cube_g4_face_exit 41
#define _atom_offset_bounds_chk_cube_g4_face_exit 24 #define _atom_offset_bounds_chk_cube_g4_face_exit 24
+197 -529
View File
@@ -10,7 +10,7 @@
# include "duffle/pad.h" # include "duffle/pad.h"
# include "duffle/word_count.metadata.h" # include "duffle/word_count.metadata.h"
# include "duffle/psyq.h" # include "duffle/psyq.h"
# include "duffle/math.atom.c" # include "duffle/math.atom.h"
# include "duffle/mips.atom.c" # include "duffle/mips.atom.c"
# include "duffle/gte.atom.c" # include "duffle/gte.atom.c"
# include "duffle/gp.atom.c" # include "duffle/gp.atom.c"
@@ -26,7 +26,7 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c);
#pragma region MACs (Mips Atom components) #pragma region MACs (Mips Atom components)
FI_ Slice_MipsCode ac_put_disp_env(AtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port) FI_ Slice_MipsCode ac_put_disp_env(AtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_put_disp_env, ab, { MipsAtomComp_Proc_(ab, {
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)). // Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR // Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port), mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
@@ -37,7 +37,7 @@ MipsAtomComp_Proc_(ac_put_disp_env, ab, {
}) })
I_ Slice_MipsCode ac_put_draw_env(AtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port) I_ Slice_MipsCode ac_put_draw_env(AtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_put_draw_env, ab, { MipsAtomComp_Proc_(ab, {
/* /*
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings. * ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
* References: * References:
@@ -94,522 +94,187 @@ MipsAtomComp_Proc_(ac_put_draw_env, ab, {
#pragma region Atom Procs #pragma region Atom Procs
// Modular Atoms // Modular Atoms
/* Scratchpad layout for the resolve_look_at bundle. #define AtomBundle_(name) Struct_(tmpl(AtomBundle,name))
* The chain atoms communicate entirely via the wave-context GPR carrier R_ResolveScratch (R_T4) + hardcoded offsets into smem.scratchpad #define AtomBundle_Len(name) S_(tmpl(AtomBundle,name))/S_(MipsAtom*)
* (PS1 hardware scratchpad at 0x1F800000). #define AtomBundleEntry_(bundle,entry) tmpl(bundle,entry)
*
* Atom 0 (input_and_sub) STAGES the C-side inputs (eye, up_in) into the scratchpad;
* AT THE SAME TIME it computes fwd = target - eye and stores it at scratch+0.
* Atoms 1-6 then read/write specific scratchpad offsets internally using
* `r_scratch + hardcoded_offset` — no tape-data pointers are passed between atoms.
* +0 fwd (atom 0 writes; atom 1 reads)
* +16 uz (atom 1 writes; atoms 2 + 4 read)
* +32 right (atom 2 writes; atom 3 reads)
* +48 ux (atom 3 writes; atoms 4 + 6 read)
* +64 up (atom 4 writes; atom 5 reads)
* +80 uy (atom 5 writes; atom 6 reads)
* +96 eye (atom 0 stages from C-side pointer; atom 6 reads)
* +128 up_in (atom 0 stages from C-side pointer; atom 2 reads)
*/
// enum { #pragma region resolve_look_at
// R_LookAt = R_T0 atom_reg atom_type(MT3_S2S4*), /* ─── resolve_look_at bundle chain atoms ──────────────────────────── */
// R_CamEye = R_T1 atom_reg atom_type(P3_S4*),
// R_CamTarget = R_T2 atom_reg atom_type(P3_S4*),
// R_WorldUp = R_T3 atom_reg atom_type(V3_S4*),
// };
enum { typedef AtomBundle_(resolve_look_at) { MipsAtom
/* Wave-context GPR carrier for the resolve_look_at bundle: the scratch base. *input_and_sub,
* Set by atom 0 (popped from tape), read by atoms 1-6 (used as pointer base). */ *normalize_fwd_uz,
R_ResolveScratch = R_T4 atom_reg atom_type(U4*), *cross_to_right,
#define R_ResolveScratch_Code R_T4_Code *normalize_right_ux,
*cross_to_up,
*normalize_up_uy,
*pop_mv_trans;
}; };
typedef Struct_(ResolveLookAtScratch) {
V3_S4 fwd;
V3_S4 uz;
V3_S4 right;
V3_S4 ux;
V3_S4 up;
V3_S4 uy;
P3_S4 eye;
P3_S4 target;
V3_S4 up_in;
};
/* Binds_ResolveLookAtSub — what the C side pushes onto the tape before input_and_sub.
* The scratchpad base is no longer pushed because R_ScratchBase (= R_SP) is a tape carrier
* preserved across atoms; the atom body reads 0x1F800000 directly from R_SP. */
typedef Struct_(Binds_ResolveLookAt) { typedef Struct_(Binds_ResolveLookAt) {
MT3_S2S4* look_at; MT3_S2S4* look_at;
P3_S4* eye; P3_S4* eye;
P3_S4* target; P3_S4* target;
V3_S4* up_in; V3_S4* up_in;
}; };
/* ─── ResolveLookAtScratch — offset schema for the resolve_look_at bundle's
* scratchpad slots (PS1 hardware scratchpad at 0x1F800000).
*
* Each slot is 16 bytes: V3_S4 is already 16 bytes (4 × S4 = x/y/z/pad).
* The struct fields are contiguous — slot i starts at offset i*16.
* Used by the assembly via O_(ResolveLookAtScratch, fld.x/y/z) which resolves to a compile-time byte offset.
* NOT a runtime struct — the struct is purely a schema for offsets; the assembly uses `r_scratch + O_(...)` to compute slot addresses at runtime.
*
* Slot producers/consumers (referenced by the resolve_look_at chain atoms):
* +0 fwd 0 writes (target - eye); atom 1 (normalize) reads
* +16 uz 1 writes (normalize fwd); atoms 2 + 4 read (cross operands)
* +32 right 2 writes (cross uz x up_in); atom 3 (normalize) reads
* +48 ux 3 writes (normalize right); atoms 4 + 6 read
* +64 up 4 writes (cross uz x ux); atom 5 (normalize) reads
* +80 uy 5 writes (normalize up); atom 6 reads
* +96 eye 0 stages (C-side input); atom 6 reads (translation column)
* +112 target reserved (currently written nowhere — kept for symmetry w/ eye)
* +128 up_in 0 stages (C-side input); atom 2 reads (cross operand)
*
* Fields use P3_S4 (point) for eye/target (RGA: affine point, implicit weight 1);
* V3_S4 (vector) for fwd/uz/right/ux/up/uy/up_in (RGA: Euclidean vector).
* P3_S4 is a storage alias of V3_S4 (see math.h comment: "Storage alias of V3_S4.
* Use P3_S4 when the value is a point.") — both are 16 bytes.
*/
typedef Struct_(ResolveLookAtScratch) {
V3_S4 fwd; /* offset +0 (16 bytes — 4 S4 fields incl. internal pad) */
V3_S4 uz; /* offset +16 (16 bytes) */
V3_S4 right; /* offset +32 (16 bytes) */
V3_S4 ux; /* offset +48 (16 bytes) */
V3_S4 up; /* offset +64 (16 bytes) */
V3_S4 uy; /* offset +80 (16 bytes) */
P3_S4 eye; /* offset +96 (16 bytes; storage alias of V3_S4) */
P3_S4 target; /* offset +112 (16 bytes; storage alias of V3_S4) */
V3_S4 up_in; /* offset +128 (16 bytes) */
};
/* ─── resolve_look_at bundle chain atoms ────────────────────────────
* 4 unique atom procs in the resolve_look_at bundle (4 chain atoms + 3 calls to generic normalize_v3s4_proc).
* All 4 chain atoms are runtime-built MipsAtom_Proc_ atoms: each function declares a static MipsCode[] body,
* then calls atombuilder_unroll() to append it to the caller's MipsAtomBuilder arena. resolve_look_at_init()
* uses this pattern to pre-build the bundle into the static arena (smem.resolve_look_at_arena).
*
* Atom roster:
* 0: resolve_look_at__input_and_sub (chain atom)
* 1: normalize_v3s4_proc (gte.atom.c) (generic normalize; called for fwd→uz)
* 2: resolve_look_at__cross_uz_up_in_to_right (chain atom)
* 3: normalize_v3s4_proc (gte.atom.c) (generic normalize; called for right→ux)
* 4: resolve_look_at__cross_uz_ux_to_up (chain atom)
* 5: normalize_v3s4_proc (gte.atom.c) (generic normalize; called for up→uy)
* 6: resolve_look_at__populate_and_translate (chain atom)
*
* The generic normalize_v3s4_proc is a parameterized 4-stage GTE normalize (SQR → mfc2 → LZCS → GPF → srav);
* it accepts scratch base + offset args so any caller (with a scratch base + struct schema) can use it.
*/
typedef Struct_(Binds_ResolveLookAtSub) { typedef Struct_(Binds_ResolveLookAtSub) {
P3_S4* target; /* U4 (C-side P3_S4* — read by atom 0 directly; NOT a scratchpad address) */ P3_S4* target;
P3_S4* eye; /* U4 (C-side P3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */ P3_S4* eye;
V3_S4* up_in; /* U4 (C-side V3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */ V3_S4* up_in;
ResolveLookAtScratch* scratchpad;
}; };
typedef Struct_(RegUse_resolve_look_at_input_and_sub) {
Reg target; Reg eye; Reg up_in;
Reg t0; Reg t1; Reg t2; Reg t3; Reg t4;
};
/* Atom 0 in the bundle: input_and_sub. Stages C-side inputs into the scratchpad and computes fwd = target - eye. */
internal MipsAtom* AtomBundleEntry_(resolve_look_at,input_and_sub)(AtomArena_R aa, RegUse_resolve_look_at_input_and_sub r)
atom_info(atom_bind(Binds_ResolveLookAtSub)) MipsAtom_Proc_(aa, {
load_word(r.target, R_TapePtr, O_(Binds_ResolveLookAtSub,target)),
load_word(r.eye, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)),
load_word(r.up_in, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)),
LdSlot_ add_ui_self(R_TapePtr, S_(Binds_ResolveLookAtSub)),
/* Atom 0 in the bundle: input_and_sub. Stages C-side inputs into the scratchpad and computes fwd = target - eye. /* Stage up_in.x/y/z into the scratchpad. R_ScratchBase = R_SP = 0x1F800000. */
* Staging work: mac_load_word_v3( r.t0, r.t1, r.t2, r.up_in, 0), LdSlot_
* * Stage eye.x/y/z → scratch (for atom 6's translation column) mac_store_word_v3(r.t0, r.t1, r.t2, R_ScratchBase, O_(ResolveLookAtScratch,up_in)),
* * Stage up_in.x/y/z → scratch (for atom 2's outer-product operand)
* * Compute fwd = target - eye, store fwd.x/y/z → scratch+0/+4/+8 (for atom 1)
* GPR codes (assigned by resolve_look_at_init):
* r_target_ptr : R_T0
* r_eye_ptr : R_T1
* r_up_in_ptr : R_T2
* r_scratch : R_T4 (R_ResolveScratch; wave-context carrier)
* r_tmp0 : R_T3 (stage eye/up_in + load eye.y)
* r_tmp1 : R_T5 (stage eye/up_in + load eye.z)
* r_tmp2 : R_T6 (stage eye/up_in + load target.x)
* r_tmp3 : R_T7 (stage eye/up_in + load target.y)
* R_AT : hardcoded (load eye.y / eye.z / target.z)
* R_V0 : hardcoded (load eye.z / target.z)
* Pool cost: 8 GPRs + R_T4 (carrier) + R_AT + R_V0 (hardcoded) = 11 GPRs.
*/
internal MipsAtom* resolve_look_at__input_and_sub_proc(AtomArena_R aa,
// TODO(Ed): We can resolve scratch at anytime its fixed to a specific address.
U4 r_scratch
, U4 r_target_ptr,U4 r_eye_ptr, U4 r_up_in_ptr
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2, U4 r_tmp3
) MipsAtom_Proc_(resolve_look_at__input_and_sub, aa, {
load_word(r_target_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,target)),
load_word(r_eye_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)),
load_word(r_up_in_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)),
load_word(r_scratch, R_TapePtr, O_(Binds_ResolveLookAtSub,scratchpad)),
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtSub)),
// Stage eye.x/y/z into the scratchpad (atom 6 reads these for the translation column). // Stage eye.x/y/z into the scratchpad (atom 6 reads these for the translation column).
mac_load_p3s4( r_tmp0, r_tmp1, r_tmp2, r_eye_ptr, 0), mac_load_word_v3( r.t0, r.t1, r.t2, r.eye, 0), LdSlot_
mac_store_p3s4(r_tmp0, r_tmp1, r_tmp2, r_scratch, O_(ResolveLookAtScratch,eye)), mac_store_word_v3(r.t0, r.t1, r.t2, R_ScratchBase, O_(ResolveLookAtScratch,eye)),
/* Stage up_in.x/y/z into the scratchpad. */
mac_load_p3s4( r_tmp0, r_tmp1, r_tmp2, r_up_in_ptr, 0),
mac_store_p3s4(r_tmp0, r_tmp1, r_tmp2, r_scratch, O_(ResolveLookAtScratch,up_in)),
/* Compute fwd = target - eye. */ /* Compute fwd = target - eye. */
mac_load_p3s4(r_tmp0, r_tmp1, r_tmp2, r_target_ptr, 0), // mac_load_p3s4(t3, R_AT, t4, r.eye, 0),
mac_load_p3s4(r_tmp3, R_AT, R_V0, r_eye_ptr, 0), mac_load_word_v3(r.t3, R_AT, r.t4, r.target, 0), LdSlot_
mac_sub_v3s4( mac_sub_s_v3_self(
r_tmp0, r_tmp1, r_tmp2, r.t3, R_AT, r.t4,
r_tmp3, R_AT, R_V0), r.t0, r.t1, r.t2),
mac_store_v3s4(r_tmp0, r_tmp1, r_tmp2, r_scratch, O_(ResolveLookAtScratch,fwd)), mac_store_word_v3(r.t3, R_AT, r.t4, R_ScratchBase, O_(ResolveLookAtScratch,fwd)),
mac_yield() mac_yield()
}) })
/* Atoms 2 + 4 in the bundle: out = a × b (GTE outer product on IR/D vectors).
* No bind pop — the three operand pointers (a, b, out) are derived in-body from r_scratch + hardcoded_offset.
* Each atom has its own variant because the offsets are baked into the body and each atom uses unique GPRs.
*
* GTE register layout (per PSX-SPX + duffle gte.h):
* IR1/2/3 = a.x/y/z (mtc2)
* VXY0 = b.x (mtc2)
* VZ0 = b.y (mtc2)
* VXY1 = b.z (mtc2)
* OP = outer product
* MAC1/2/3 = out.x/y/z (mfc2)
*
* Pool cost: r_scratch (R_T4 carrier) + 7 body GPRs + R_AT + R_V0 (hardcoded) = 10 GPRs.
*/
/* Atom 2: cross uz × up_in → right. */ typedef Struct_(Binds_ResolveLookAtPopMvTrans) {
internal MipsAtom* resolve_look_at__cross_uz_up_in_to_right_proc(AtomArena_R aa, U4 r_scratch U4 look_at; /* MT3_S2S4* — destination matrix address */
, U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */
, U4 r_d /* load b.x */
, U4 r_f, U4 r_g, U4 r_h /* r_f = &right (out ptr), r_g = &uz, r_h = &up_in */
) MipsAtom_Proc_(resolve_look_at__cross_uz_up_in_to_right, aa, {
/* FIX: build packed RT22+RT33 with proper sign extension. */
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
add_si(r_h, r_scratch, O_(ResolveLookAtScratch,up_in)), /* r_h = &up_in */
add_si(r_f, r_scratch, O_(ResolveLookAtScratch,right)), /* r_f = &right (out) */
nop,
/* Load a (uz).x/y/z into r_a/r_b/r_c. */
load_word(r_a, r_g, O_(V3_S4,x)),
load_word(r_b, r_g, O_(V3_S4,y)),
load_word(r_c, r_g, O_(V3_S4,z)),
nop,
/* Load b (up_in).x/y/z into r_d + R_AT/R_V0 (R_AT/R_V0 are hardcoded scratch). */
load_word(r_d, r_h, O_(V3_S4,x)),
load_word(R_AT, r_h, O_(V3_S4,y)),
load_word(R_V0, r_h, O_(V3_S4,z)),
nop,
/* Save the two RT control-register slots OP will clobber. We reuse
* r_g/r_h (scratch pointers, no longer needed) as the save targets. */
gte_mv_from_ctrl_r(r_g, gte_cr_RT11), /* r_g = C2 r0 (RT11|RT12) */
gte_mv_from_ctrl_r(r_h, gte_cr_RT22), /* r_h = C2 r4 (RT22|RT33) */
/* Load uz.x/uz.y/uz.z into COP2 control registers.
* OP reads D1 = RT11 from $0.low, D2 = RT22 from $2.high, D3 = RT33 from $4.high.
* RT22 is in BOTH $2.high AND $4.low (shared bit position). OP reads from $2.high.
* So set RT22 via ctc2 r_b, $2 (sets $2.high = a.y.high = RT22, $2.low = a.y.low = RT13).
* Then set RT33 via ctc2 r_c, $4 (sets $4.high = a.z.high = RT33, $4.low = a.z.low).
* The $2 and $4 writes don't clobber each other (separate registers).
* The 2nd ctc2 DOES clobber $4.low (becomes a.z.low, NOT a.y.high), but since OP
* reads RT22 from $2.high (which the 2nd ctc2 doesn't touch), D2 is still a.y.high.
* This is libpsyx's OuterProduct12 convention EXACTLY. */
gte_mv_to_ctrl_r(r_b, gte_cr_RT13), /* $2 = r_b = a.y. RT13=a.y.low, RT22=a.y.high. */
gte_mv_to_ctrl_r(r_c, gte_cr_RT22), /* $4 = r_c = a.z. RT22=a.z.low, RT33=a.z.high. */
/* Load uz into the RT diagonal. */
gte_mv_to_ctrl_r(r_a, gte_cr_RT11), /* D1 = RT11 = uz.x (low 16 of $0, sign-extended by OP). */
nop2, /* CTC2 retirement (CPU→COP2 2-slot delay) */
/* Load up_in into IR (the second operand for OP). */
gte_mv_to_data_r(r_d, C2_IR1), /* IR1 = up_in.x */
gte_mv_to_data_r(R_AT, C2_IR2), /* IR2 = up_in.y */
gte_mv_to_data_r(R_V0, C2_IR3), /* IR3 = up_in.z */
nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
gte_cmdw_outer_product, /* OP: MAC1/2/3 = uz × up_in
* MAC1 = IR3*D2 - IR2*D3 = up_in.z*uz.y.high - up_in.y*uz.z.high
* MAC2 = IR1*D3 - IR3*D1 = up_in.x*uz.z.high - up_in.z*uz.x
* MAC3 = IR2*D1 - IR1*D2 = up_in.y*uz.x - up_in.x*uz.y.high
* For up_in = (0, -fp_one, 0):
* MAC1 = 0 - (-fp_one)*uz.z.high = fp_one*uz.z.high
* MAC2 = 0 - 0 = 0
* MAC3 = (-fp_one)*uz.x - 0 = -fp_one*uz.x */
/* Restore the RT slots we clobbered. */
gte_mv_to_ctrl_r(r_g, gte_cr_RT11), /* restore C2 r0 (RT11|RT12) */
gte_mv_to_ctrl_r(r_h, gte_cr_RT22), /* restore C2 r4 (RT22|RT33) */
/* mfc2 MAC1/2/3 → r_a/r_b/r_c (out.x/y/z). */
gte_mv_from_data_r(r_a, C2_MAC1),
gte_mv_from_data_r(r_b, C2_MAC2),
gte_mv_from_data_r(r_c, C2_MAC3),
nop, /* MFC2 retirement */
/* Right-shift MAC by 12 to convert from GTE's S12.20 fixed-point scale back to libpsyx OuterProduct12 convention (S12.0, fp_one=4096=1<<12).
* Without this, MAC values (~16M for unit-vector cross products) overflow the GTE's 16-bit IR registers when atom 3 normalizes via mtc2. */
shift_aright(r_a, r_a, 12),
shift_aright(r_b, r_b, 12),
shift_aright(r_c, r_c, 12),
/* Store out.x/y/z to r_f (out ptr = scratch+32). */
store_word(r_a, r_f, O_(V3_S4,x)),
store_word(r_b, r_f, O_(V3_S4,y)),
store_word(r_c, r_f, O_(V3_S4,z)),
mac_yield()
})
/* Atom 4: cross uz × ux → up. */
internal MipsAtom* resolve_look_at__cross_uz_ux_to_up_proc(AtomArena_R aa, U4 r_scratch
, U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */
, U4 r_d /* load b.x */
, U4 r_f, U4 r_g, U4 r_h /* r_f = &up (out ptr), r_g = &uz, r_h = &ux */
) MipsAtom_Proc_(resolve_look_at__cross_uz_ux_to_up, aa, {
/* Compute the three scratch pointers from r_scratch. */
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
add_si(r_h, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_h = &ux */
add_si(r_f, r_scratch, O_(ResolveLookAtScratch,up)), /* r_f = &up (out) */
nop,
/* Load a (uz).x/y/z into r_a/r_b/r_c. */
load_word(r_a, r_g, O_(V3_S4,x)),
load_word(r_b, r_g, O_(V3_S4,y)),
load_word(r_c, r_g, O_(V3_S4,z)),
nop,
/* Load b (ux).x/y/z into r_d + R_AT/R_V0. */
load_word(r_d, r_h, O_(V3_S4,x)),
load_word(R_AT, r_h, O_(V3_S4,y)),
load_word(R_V0, r_h, O_(V3_S4,z)),
nop,
/* OP reads D1/D2/D3 from RT11/RT22/RT33 ($0/$2/$4), not V0/V1/V2.
* Mirror atom 1: cfc2 RT save, ctc2 RT diagonal from uz, mtc2 IR from ux,
* ctc2 RT restore. */
/* Save the two RT control-register slots OP will clobber (reusing
* r_g/r_h — they're no longer needed as scratch pointers). */
gte_mv_from_ctrl_r(r_g, gte_cr_RT11), /* r_g = C2 $0 (RT11|RT12) */
gte_mv_from_ctrl_r(r_h, gte_cr_RT22), /* r_h = C2 $4 (RT22|RT33) */
/* Load uz into the RT diagonal — same packing as atom 1.
* OP reads D1 = RT11 from $0.low, D2 = RT22 from $2.high, D3 = RT33 from $4.high.
* RT22 is shared between $2.high and $4.low — the ctc2 sequence to $2 then $4
* sets RT22 to uz.y.high (via $2), then to uz.z.low (via $4). OP reads
* RT22 from $2.high which the second ctc2 doesn't touch, so D2 stays uz.y.high.
* (This is libpsyx OuterProduct12 convention EXACTLY.) */
gte_mv_to_ctrl_r(r_b, gte_cr_RT13), /* $2 = uz.y. RT13=uz.y.low, RT22=uz.y.high. */
gte_mv_to_ctrl_r(r_c, gte_cr_RT22), /* $4 = uz.z. RT22=uz.z.low, RT33=uz.z.high. */
gte_mv_to_ctrl_r(r_a, gte_cr_RT11), /* $0 = uz.x. RT11=uz.x. */
nop2, /* CTC2 retirement (CPU→COP2 2-slot delay) */
/* Load ux into the IR registers (the second operand for OP). */
gte_mv_to_data_r(r_d, C2_IR1), /* IR1 = ux.x */
gte_mv_to_data_r(R_AT, C2_IR2), /* IR2 = ux.y */
gte_mv_to_data_r(R_V0, C2_IR3), /* IR3 = ux.z */
nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
gte_cmdw_outer_product,
/* Restore the RT slots we clobbered. */
gte_mv_to_ctrl_r(r_g, gte_cr_RT11), /* restore C2 $0 (RT11|RT12) */
gte_mv_to_ctrl_r(r_h, gte_cr_RT22), /* restore C2 $4 (RT22|RT33) */
gte_mv_from_data_r(r_a, C2_MAC1),
gte_mv_from_data_r(r_b, C2_MAC2),
gte_mv_from_data_r(r_c, C2_MAC3),
nop,
/* Right-shift MAC by 12 to convert from GTE's S12.20 scale back to libpsyx
* OuterProduct12 convention (S12.0, fp_one=4096). See atom 1 for rationale. */
shift_aright(r_a, r_a, 12),
shift_aright(r_b, r_b, 12),
shift_aright(r_c, r_c, 12),
store_word(r_a, r_f, O_(V3_S4,x)),
store_word(r_b, r_f, O_(V3_S4,y)),
store_word(r_c, r_f, O_(V3_S4,z)),
mac_yield()
})
typedef Struct_(Binds_ResolveLookAtPopAndTrans) {
U4 look_at; /* U4 (MT3_S2S4* — destination matrix address) */
}; };
/* Atom 6 in the bundle: write look_at->m[][] from ux/uy/uz, then compute the translation column t[] = R * (-eye). typedef Struct_(RegUse_resolve_look_at__pop_mv_trans) {
Reg look_at;
Reg_(V3_S4) row; /* populate phase: load ux/uy/uz */
union { Reg ux, v_x; } t6; /* populate addr (canonical) → matrix_vector v_x */
union { Reg uy, v_y; } t7; /* populate uy → matrix_vector v_y */
union { Reg uz, v_z; } t8; /* populate uz → matrix_vector v_z */
Reg eye; /* matrix_vector phase: load -eye */
};
/* Atom 6 (fused): write look_at->m[][] from ux/uy/uz as packed S2 (populate),
* ctc2 RT chain into C2[0..4] (matrix_vector), MVMVA RT*(-eye)>>12, store off
* directly to look_at->t[] (trans_matrix). Replaces the previous 3 separate atoms
* (populate + matrix_vector + trans_matrix).
* *
* GPR codes (assigned by resolve_look_at_init): * MT3_S2S4 { A3x3_S2 m; A3_S4 t; }
* r_look_at : MT3_S2S4* (popped from tape; output matrix destination) * m[][] is S2 packed (9 × 2 = 18 bytes at offset 0)
* r_pux : pointer to ux (offset O_(ResolveLookAtScratch,ux)) * t[0..2] is S4 (3 × 4 = 12 bytes at offset 18)
* r_puy : pointer to uy (offset O_(ResolveLookAtScratch,uy))
* r_puz : pointer to uz (offset O_(ResolveLookAtScratch,uz))
* r_peye : pointer to eye (offset O_(ResolveLookAtScratch,eye))
* r_tmp0/1/2 : atom-local scratch (load + MVMVA + store temps)
* *
* 4 pointer regs (r_pux/r_puy/r_puz/r_peye) are DEDICATED — they hold the scratch addresses for the entire body. * C11 ApplyMatrixLV semantics (gte.atom.c ac_apply_matrix_lv; libgte reference):
* They are computed in-body via `add_si(r_px, r_scratch, O_(ResolveLookAtScratch, field))` so no tape-data pointer is needed.
*
* Struct layout (per duffle/math.h):
* MT3_S2S4 { A3x3_S2 m; A3_S4 t; } → m[][] is S2 packed (9 × 2 = 18 bytes at offset 0)
* t[0/1/2] is S4 (3 × 4 = 12 bytes at offset 18)
*
* Translation column: GTE MVMVA with the world rotation matrix pre-set
* (helper emits set_gte_world before the bundle, per the bundle design).
* MVMVA computes R * pos (with cv=0/mx=0/sf=0/v=0); MAC1/2/3 = R * (-eye).
* Pool cost: r_look_at (1) + r_scratch (R_T4 carrier) + 4 ptr regs + 3 tmp regs = 9 GPRs.
*/
internal MipsAtom* resolve_look_at__populate_proc(AtomArena_R aa
, U4 r_look_at
, U4 r_scratch
, U4 r_pux, U4 r_puy, U4 r_puz
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2
) MipsAtom_Proc_(resolve_look_at__populate, aa, {
/* Pop look_at* (the matrix output) — advance R_TapePtr by 4 bytes. */
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
/* Compute the 3 scratch pointers in their dedicated GPRs (eye isn't needed by 6a — 6b reads it). */
add_si(r_pux, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_pux = &ux */
add_si(r_puy, r_scratch, O_(ResolveLookAtScratch,uy)), /* r_puy = &uy */
add_si(r_puz, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_puz = &uz */
nop,
/* ── m[0] = (S2)ux ── */
load_word(r_tmp0, r_pux, O_(V3_S4,x)),
load_word(r_tmp1, r_pux, O_(V3_S4,y)),
load_word(r_tmp2, r_pux, O_(V3_S4,z)),
nop,
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[0][0])),
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[0][1])),
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[0][2])),
/* ── m[1] = (S2)uy ── */
load_word(r_tmp0, r_puy, O_(V3_S4,x)),
load_word(r_tmp1, r_puy, O_(V3_S4,y)),
load_word(r_tmp2, r_puy, O_(V3_S4,z)),
nop,
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[1][0])),
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[1][1])),
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[1][2])),
/* ── m[2] = (S2)uz ── */
load_word(r_tmp0, r_puz, O_(V3_S4,x)),
load_word(r_tmp1, r_puz, O_(V3_S4,y)),
load_word(r_tmp2, r_puz, O_(V3_S4,z)),
nop,
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[2][0])),
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[2][1])),
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[2][2])),
/* Zero t[0..2] — atom 6c writes the final values here. */
store_word(R_0, r_look_at, O_(MT3_S2S4,t[0])),
store_word(R_0, r_look_at, O_(MT3_S2S4,t[1])),
store_word(R_0, r_look_at, O_(MT3_S2S4,t[2])),
mac_yield()
})
/* Atom 6b in the bundle: matrix-vector product off = R * (-eye) >> 12.
* Uses RTPS with V0 loaded from scratch via lwc2. The RT matrix is
* pre-loaded by atom 6a.5 (resolve_look_at__load_rt).
* Stores off to scratch+96 (overwriting the packed pos).
*
* GPR codes (assigned by resolve_look_at_init):
* r_scratch : R_ResolveScratch (R_T4) — scratch base
* r_peye : pointer to eye (slot +96, reused as off destination)
* r_tmp0/1/2: -eye + GTE transfer scratch
*
* Pool cost: r_scratch (carrier) + 1 ptr reg + 3 tmp regs = 5 GPRs.
*/
internal MipsAtom* resolve_look_at__matrix_vector_proc(AtomArena_R aa
, U4 r_scratch
, U4 r_peye
, U4 r_look_at
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2
) MipsAtom_Proc_(resolve_look_at__matrix_vector, aa, {
/* === EXACT C11 ApplyMatrixLV replication ===
* The C11 does:
* 1. ctc2 RT matrix (5 ctc2s to C2[0..4]) * 1. ctc2 RT matrix (5 ctc2s to C2[0..4])
* 2. lw v.x/y/z from memory * 2. lw -eye from memory
* 3. S15 decomposition (negu + sra 15 + negu + andi 0x7FFF + negu) * 3. S15 decomposition (eliminated here — the fused body takes the >>12 path
* 4. mtc2 HIGH bits to IR1/2/3, nop, MVMVA pass1 (sf=0, mx=0, v=3, cv=3) * directly via mtc2 IR + MVMVA pass2, matching the libgte canonical output)
* 5. mfc2 MACs * 4. mtc2 to IR1/2/3, nop2, MVMVA pass2 (sf=1, mx=0, v=3, cv=3)
* 6. mtc2 LOW bits to IR1/2/3, nop, MVMVA pass2 (sf=1, mx=0, v=3, cv=3) * 5. mfc2 MACs → off
* 7. mfc2 MACs * 6. store off to look_at->t[] (skip scratch.eye intermediate)
* 8. Combine: (pass1 << 3) + pass2
*
* For S16-fitting pos (|pos| < 32768), pos >> 15 = 0, so pass1 = 0.
* The combine simplifies: result = 0 + pass2 = pass2.
* So we skip the S15 decomposition and just do pass 2 directly.
* We still use v=3 (IR input) and mx=0 (RT matrix) like the C11. */
/* Pop look_at* from tape. */
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
/* r_peye = &eye (slot +96, reused as off destination). */
add_si(r_peye, r_scratch, O_(ResolveLookAtScratch,eye)),
nop,
/* === Load RT matrix from look_at into C2[0..4] via ctc2 ===
* Exact s ame sequence as set_gte_mt3s2s4 / C11's ApplyMatrixLV. */
load_word( r_tmp0, r_look_at, 0), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT11),
load_word( r_tmp0, r_look_at, 4), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT12),
load_word( r_tmp0, r_look_at, 8), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT13),
load_word( r_tmp0, r_look_at, 12), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT21),
load_half_u(r_tmp0, r_look_at, 16), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT22),
nop2, /* CTC2 retirement (2 slots × 5 ctc2s) */
/* Load pos = -eye after the matrix load releases r_tmp0. */
load_word(r_tmp0, r_peye, O_(P3_S4,x)),
load_word(r_tmp1, r_peye, O_(P3_S4,y)),
load_word(r_tmp2, r_peye, O_(P3_S4,z)),
nop,
sub_u(r_tmp0, R_0, r_tmp0), /* pos.x = -eye.x */
sub_u(r_tmp1, R_0, r_tmp1),
sub_u(r_tmp2, R_0, r_tmp2),
/* === mtc2 pos (as S16) to IR1/2/3 ===
* The GTE takes low 16 bits. pos fits in S16. For negative pos, the
* 32-bit sign-extended value's low 16 bits = correct S16. */
/* Mask pos to 16 bits to be safe. For S16-fitting pos, pos & 0xFFFF
* gives the correct S16 value (sign bit preserved). */
/* r_tmp0/1/2 already have pos values. */
gte_mv_to_data_r(r_tmp0, C2_IR1),
gte_mv_to_data_r(r_tmp1, C2_IR2),
gte_mv_to_data_r(r_tmp2, C2_IR3),
nop2, /* MTC2 retirement (2 slots) */
/* === MVMVA pass 2 EXACT C11 command: 0x4A49E012 ===
* sf=1, mx=0 (RT), v=3 (IR), cv=3. Reads RT × IR >> 12. */
gte_cmdw_mvmva_c11_pass2_exact,
nop, /* GTE interlock */
/* === mfc2 MAC1/2/3 → r_tmp0/1/2 === */
gte_mv_from_data_r(r_tmp0, C2_MAC1),
gte_mv_from_data_r(r_tmp1, C2_MAC2),
gte_mv_from_data_r(r_tmp2, C2_MAC3),
nop,
/* === Store off → scratch+96 (overwriting pos) === */
store_word(r_tmp0, r_peye, O_(V3_S4,x)),
store_word(r_tmp1, r_peye, O_(V3_S4,y)),
store_word(r_tmp2, r_peye, O_(V3_S4,z)),
mac_yield()
})
/* Atom 6c in the bundle: copy scratch+96 (off, written by atom 6b) → look_at->t[].
* Uses mac_trans_matrix component (m->t = v, libgte TransMatrix semantics = struct copy).
* *
* GPR codes (assigned by resolve_look_at_init): * GPR codes (assigned by resolve_look_at_init):
* r_look_at : MT3_S2S4* (popped from tape; output matrix destination) * r_scratch : R_ResolveScratch (R_T4 carrier)
* r_scratch : R_ResolveScratch (R_T4) — scratch base * r_look_at : ralloc() — also serves as the off-dst in the trans_matrix phase
* r_off_ptr : pointer to off (= &scratch.eye, reused slot) * r_row : V3_S4, reused for ux/uy/uz loads in populate phase
* r_tmp0 : transfer reg for mac_trans_matrix * r_eye : ralloc() — &scratch.eye, used for -eye load in matrix_vector phase
* r_v_x/v_y/v_z : ralloc() — populate scratch addrs (ux/uy/uz), reused as
* ctc2 transfer + MVMVA -eye temp in matrix_vector phase
* (v_x/v_y/v_z alias ux/uy/uz via the union; lifetime ends for ux/uy/uz after
* populate's mac_load_v3s4, so reusing for v.x/v.y/v.z is safe)
* Pool cost: 1 carrier + 1 look_at + 3 row + 1 eye + 3 aliased = 9 GPRs
* *
* Pool cost: r_look_at (1) + r_scratch (carrier) + r_off_ptr + 1 clobber = 4 GPRs. * Net word savings vs the previous 3-atom flow: ~15 words + 2 mac_yields + 1 tape pop.
* - 2 mac_yields (trans_matrix's + matrix_vector's) → fused into one yield
* - 1 redundant tb_data (look_at was pushed 2x; now once)
* - mac_trans_mt3s3s4 (6 words) → replaced by direct mac_store_v3s4
* - mac_store_v3s4 to scratch.eye (3 words intermediate) → eliminated
* - add_si for r_off_ptr (2 words) → eliminated
* - mac_store_v3s4 zero-store of t[] (3 words) → eliminated (matrix_vector writes
* off directly; no consumer needed the zero first)
* - 1 set_gte_mt3s2s4 ctc2 chain (13 baked words) → eliminated (matrix_vector
* has its own ctc2 RT chain; cube rendering atoms reload C2 state themselves)
*/ */
I_ MipsAtom* resolve_look_at__trans_matrix_proc(AtomArena_R aa internal MipsAtom* resolve_look_at__pop_mv_trans(AtomArena_R aa,
, U4 r_look_at RegUse_resolve_look_at__pop_mv_trans r
, U4 r_scratch ) MipsAtom_Proc_(aa, {
, U4 r_off_ptr /* --- Tape pop: look_at pointer --- */
, U4 r_tmp0 load_word(r.look_at, R_TapePtr, O_(Binds_ResolveLookAtPopMvTrans,look_at)),
) MipsAtom_Proc_(resolve_look_at__trans_matrix, aa, { LdSlot_ add_ui_self(R_TapePtr, S_(Binds_ResolveLookAtPopMvTrans)),
/* Pop look_at* from tape. */
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
/* r_off_ptr = &off (= &scratch.eye since atom 6b overwrote eye with off). */ /* --- Scratch addresses for ux/uy/uz/eye (populate phase; t6/t7/t8 alias ux/uy/uz).
add_si(r_off_ptr, r_scratch, O_(ResolveLookAtScratch,eye)), * R_ScratchBase (= R_SP) holds 0x1F800000; no per-atom bake is required because
nop, * R_SP is a tape carrier preserved across atoms. --- */
add_si(r.t6.ux, R_ScratchBase, O_(ResolveLookAtScratch, ux)), LdSlot_
add_si(r.t7.uy, R_ScratchBase, O_(ResolveLookAtScratch, uy)),
add_si(r.t8.uz, R_ScratchBase, O_(ResolveLookAtScratch, uz)),
add_si(r.eye, R_ScratchBase, O_(ResolveLookAtScratch, eye)),
/* Copy off → look_at.t[] (mac_trans_matrix: m->t = v). */ /* --- POPULATE phase: write look_at->m[][] from ux/uy/uz as packed S2 --- */
mac_trans_matrix(r_look_at, r_off_ptr, r_tmp0), mac_load_v3s4(r.row, r.t6.ux, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4, m[0])),
mac_load_v3s4(r.row, r.t7.uy, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4, m[1])),
mac_load_v3s4(r.row, r.t8.uz, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4, m[2])),
/* --- MATRIX-VECTOR phase: ctc2 RT chain + MVMVA RT*(-eye)>>12 --- */
/* RT packing (per libgte ApplyMatrixLV convention; see gte.h:217-220 +
* atom_6b_disasm_comparison.md:28-32):
* C2[0] = (RT12<<16)|RT11 ← ctc2 RT11 from m[0][0..1] packed word
* C2[1] = (RT21<<16)|RT13 ← ctc2 RT12 from m[0][2..3] packed word
* C2[2] = (RT23<<16)|RT22 ← ctc2 RT13 from m[1][1..2] packed word
* C2[3] = (RT32<<16)|RT31 ← ctc2 RT21 from m[2][0..1] packed word
* C2[4] = (RT33<<16)|junk ← ctc2 RT22 from m[2][2] (half)
* Each ctc2 writes a WHOLE 32-bit C2 slot; the "macro name" identifies
* which C2 register, not which 16-bit half. */
load_word( r.t6.v_x, r.look_at, O_(MT3_S2S4, m[0][0])), /* RT11|RT12 */ LdSlot_
load_word( r.t7.v_y, r.look_at, O_(MT3_S2S4, m[0][2])), /* RT13|RT21 */ LdSlot_ gte_mv_to_ctrl_r(r.t6.v_x, gte_cr_RT11),
load_word( r.t8.v_z, r.look_at, O_(MT3_S2S4, m[1][1])), /* RT22|RT23 */ LdSlot_ gte_mv_to_ctrl_r(r.t7.v_y, gte_cr_RT12),
load_word( r.t6.v_x, r.look_at, O_(MT3_S2S4, m[2][0])), /* RT31|RT32 */ LdSlot_ gte_mv_to_ctrl_r(r.t8.v_z, gte_cr_RT13),
load_half_u(r.t7.v_y, r.look_at, O_(MT3_S2S4, m[2][2])), /* RT33 */ LdSlot_ gte_mv_to_ctrl_r(r.t6.v_x, gte_cr_RT21),
/* pos = -eye. The three loads also retire the last CTC2. */ gte_mv_to_ctrl_r(r.t7.v_y, gte_cr_RT22),
GteDelay_ mac_load_word_v3(r.t6.v_x, r.t7.v_y, r.t8.v_z, r.eye, 0), LdSlot_
mac_sub_s_v3(r.t6.v_x, r.t7.v_y, r.t8.v_z, R_0, R_0, R_0, r.t6.v_x, r.t7.v_y, r.t8.v_z),
/* mtc2 pos (as S16) to IR1/2/3. The GTE takes low 16 bits. pos fits in S16. For negative pos, the 32-bit sign-extended value's low 16 bits = correct S16. */
gte_mv_to_data_r(r.t6.v_x, C2_IR1),
gte_mv_to_data_r(r.t7.v_y, C2_IR2),
gte_mv_to_data_r(r.t8.v_z, C2_IR3),
GteDelay_ nop2,
/* MVMVA pass 2 — C11 ApplyMatrixLV command.
* sf=1, mx=0 (RT), v=3 (IR), cv=3. Reads RT × IR >> 12. */
gte_cmdw_mvmva_c11_pass2, GteDelay_ nop,
mac_gte_mv_from_data_r_mac123(r.t6.v_x, r.t7.v_y, r.t8.v_z), GteDelay_ nop,
/* --- TRANS-MATRIX phase: store off directly to look_at->t[] (skip scratch.eye intermediate) --- */
mac_store_word_v3(r.t6.v_x, r.t7.v_y, r.t8.v_z, r.look_at, O_(MT3_S2S4, t)),
mac_yield() mac_yield()
}) })
#pragma endregion resolve_look_at
#pragma endregion Atom Procs #pragma endregion Atom Procs
@@ -628,7 +293,7 @@ internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
) { ) {
/* display[0] = (0, 0, 320, 240); rest of struct zeroed. */ /* display[0] = (0, 0, 320, 240); rest of struct zeroed. */
add_ui(R_ScreenX, R_0, ScreenRes_X), add_ui(R_ScreenY, R_0, ScreenRes_Y), add_ui(R_ScreenX, R_0, ScreenRes_X), add_ui(R_ScreenY, R_0, ScreenRes_Y),
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area.width) + OA_(DoubleBuffer,display,0)), mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area.width) + O_(DoubleBuffer,display[0])),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,display_area) + O_(DoubleBuffer,display[0])), store_word(R_0, R_ScreenBuf, O_(DisplayEnv,display_area) + O_(DoubleBuffer,display[0])),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + O_(DoubleBuffer,display[0])), store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + O_(DoubleBuffer,display[0])),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + O_(DoubleBuffer,display[0])), store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + O_(DoubleBuffer,display[0])),
@@ -741,15 +406,15 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
load_word(R_PadStateT5, R_TapePtr, O_(Binds_PadApplyInput,state)), load_word(R_PadStateT5, R_TapePtr, O_(Binds_PadApplyInput,state)),
load_word(R_CubeRot, R_TapePtr, O_(Binds_PadApplyInput,cube_rot)), load_word(R_CubeRot, R_TapePtr, O_(Binds_PadApplyInput,cube_rot)),
load_word(R_FloorRot, R_TapePtr, O_(Binds_PadApplyInput,floor_rot)), load_word(R_FloorRot, R_TapePtr, O_(Binds_PadApplyInput,floor_rot)),
add_ui_self( R_TapePtr, S_(Binds_PadApplyInput)), LdSlot_ add_ui_self( R_TapePtr, S_(Binds_PadApplyInput)),
/* Load pad[0].buttons into R_T0. */ /* Load pad[0].buttons into R_T0. */
load_word(R_T0, R_PadStateT5, O_(PadState,buttons)), nop, load_word(R_T0, R_PadStateT5, O_(PadState,buttons)), LdSlot_ nop,
// Note(Ed): Potential op with delay slot? // Note(Ed): Potential op with delay slot?
/* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */ /* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)), and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)), BdSlot_
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */ load_half( R_T4, R_CubeRot, O_(V3_S2,y)), LdSlot_
load_half( R_T3, R_FloorRot, O_(V3_S2,y)), load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
add_si( R_T4, R_T4, 30), add_si( R_T4, R_T4, 30),
add_si( R_T3, R_T3, 5), add_si( R_T3, R_T3, 5),
@@ -758,8 +423,8 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
atom_label(exit_dpad_left) atom_label(exit_dpad_left)
/* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */ /* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)), and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)), BdSlot_
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */ load_half( R_T4, R_CubeRot, O_(V3_S2,y)), LdSlot_
load_half( R_T3, R_FloorRot, O_(V3_S2,y)), load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
add_si( R_T4, R_T4, -30), add_si( R_T4, R_T4, -30),
add_si( R_T3, R_T3, -5), add_si( R_T3, R_T3, -5),
@@ -769,7 +434,7 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
/* Analog left-stick X: dead zone 0x70..0x90. /* Analog left-stick X: dead zone 0x70..0x90.
* Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */ * Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), LdSlot_ //?
/* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly). /* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly).
* set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */ * set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */
@@ -778,14 +443,14 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
atom_label(dead_check_upper) atom_label(dead_check_upper)
/* left_x >= 0x70 → check upper bound. */ /* left_x >= 0x70 → check upper bound. */
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), /* reload */ load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), /* reload */ LdSlot_ //?
add_ui( R_T4, R_0, PadDeadZone_HighBound), add_ui( R_T4, R_0, PadDeadZone_HighBound),
/* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */ /* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)), set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)), BdSlot_
add_ui( R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_high_active */ add_ui( R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_high_active */
jump_rel(atom_offset(dead_zone_skip, exit_stick)), jump_rel(atom_offset(dead_zone_skip, exit_stick)),
mac_yield_load(), BdSlot_ mac_yield_load(), LdSlot_
atom_label(dead_low_active) atom_label(dead_low_active)
/* R_T3 = left_x (from line 632 lbu; not clobbered between dead_zone_low_check branch + its BD-slot `add_ui R_T4, 0x80`). /* R_T3 = left_x (from line 632 lbu; not clobbered between dead_zone_low_check branch + its BD-slot `add_ui R_T4, 0x80`).
@@ -796,18 +461,18 @@ atom_label(dead_low_active)
/* R_T4 = cube_delta */ /* R_T4 = cube_delta */
shift_aright(R_T4, R_T3, 2), shift_aright(R_T4, R_T3, 2),
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), nop, load_half( R_T0, R_CubeRot, O_(V3_S2,y)), LdSlot_ nop,
add_u( R_T0, R_T0, R_T4), add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_CubeRot, O_(V3_S2,y)), store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
/* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap; /* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap;
* doesn't read R_T0; R_T4 settles by the subsequent add_u). */ * doesn't read R_T0; R_T4 settles by the subsequent add_u). */
load_half( R_T0, R_FloorRot, O_(V3_S2,y)), load_half( R_T0, R_FloorRot, O_(V3_S2,y)), LdSlot_
shift_aright(R_T4, R_T3, 5), shift_aright(R_T4, R_T3, 5),
add_u( R_T0, R_T0, R_T4), add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_FloorRot, O_(V3_S2,y)), store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
jump_rel(atom_offset(end_low, exit_stick)), jump_rel(atom_offset(end_low, exit_stick)),
mac_yield_load(), BdSlot_ mac_yield_load(), LdSlot_
atom_label(dead_high_active) atom_label(dead_high_active)
/* R_T3 = left_x (from line 641 lbu in dead_check_upper; not clobbered between dead_zone_high_check branch + its BD-slot `add_ui R_T4, 0x80`). /* R_T3 = left_x (from line 641 lbu in dead_check_upper; not clobbered between dead_zone_high_check branch + its BD-slot `add_ui R_T4, 0x80`).
@@ -817,18 +482,18 @@ atom_label(dead_high_active)
/* delta = 0x80 - left_x (signed negative). */ /* delta = 0x80 - left_x (signed negative). */
shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */ shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), nop, load_half( R_T0, R_CubeRot, O_(V3_S2,y)), LdSlot_ nop,
add_u( R_T0, R_T0, R_T4), add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_CubeRot, O_(V3_S2,y)), store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
/* R_T4 = floor_delta (signed) — moved into the load-delay slot of the floor load below. */ /* R_T4 = floor_delta (signed) — moved into the load-delay slot of the floor load below. */
load_half( R_T0, R_FloorRot, O_(V3_S2,y)), load_half( R_T0, R_FloorRot, O_(V3_S2,y)), LdSlot_
shift_aright(R_T4, R_T3, 5), shift_aright(R_T4, R_T3, 5),
add_u( R_T0, R_T0, R_T4), add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_FloorRot, O_(V3_S2,y)), store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
atom_label(no_jump_fallthrough) atom_label(no_jump_fallthrough)
mac_yield_load(), mac_yield_load(), LdSlot_
atom_label(exit_stick) atom_label(exit_stick)
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the dead-zone/exit branch. */ /* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the dead-zone/exit branch. */
@@ -850,40 +515,43 @@ internal MipsAtom_(pad_input_cam) atom_info(atom_bind(Binds_PadInputCam)
/* Bind pop: state → R_CamPadState (R_T5), cam → R_Cam (R_T4), advance R_TapePtr by 8. */ /* Bind pop: state → R_CamPadState (R_T5), cam → R_Cam (R_T4), advance R_TapePtr by 8. */
load_word(R_CamPadState, R_TapePtr, O_(Binds_PadInputCam,state)), load_word(R_CamPadState, R_TapePtr, O_(Binds_PadInputCam,state)),
load_word(R_Cam, R_TapePtr, O_(Binds_PadInputCam,cam)), load_word(R_Cam, R_TapePtr, O_(Binds_PadInputCam,cam)),
add_ui_self( R_TapePtr, S_(Binds_PadInputCam)), LdSlot_ add_ui_self( R_TapePtr, S_(Binds_PadInputCam)),
/* Load pad[0].buttons into R_T0; nop fills the load-delay slot. */ /* Load pad[0].buttons into R_T0; nop fills the load-delay slot. */
load_word(R_T0, R_CamPadState, O_(PadState,buttons)), load_word(R_T0, R_CamPadState, O_(PadState,buttons)), LdSlot_
load_word(R_T1, R_Cam, O_(Camera,pos.x)), // BD-Slot. load_word(R_T1, R_Cam, O_(Camera,pos.x)),
// D-pad Left → cam.pos.x -= 50. and_i fulfills BD-slot for load on R_Cam. // D-pad Left → cam.pos.x -= 50. and_i fulfills BD-slot for load on R_Cam.
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(left_x, exit_left_x)), mac_yield_load(), LdSlot_ and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(left_x, exit_left_x)), BdSlot_ mac_yield_load(), LdSlot_
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.x)), add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
atom_label(exit_left_x) atom_label(exit_left_x)
/* D-pad Right → cam.pos.x += 50. Reuses R_T1 from Left. */ /* D-pad Right → cam.pos.x += 50. Reuses R_T1 from Left. */
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(right_x, exit_right_x)), nop, and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(right_x, exit_right_x)), BdSlot_ nop,
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.x)), add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
atom_label(exit_right_x) atom_label(exit_right_x)
/* D-pad Up → cam.pos.y -= 50. Load pos.y BEFORE the andi. */ /* D-pad Up → cam.pos.y -= 50. Load pos.y BEFORE the andi. */
load_word(R_T1, R_Cam, O_(Camera,pos.y)), load_word(R_T1, R_Cam, O_(Camera,pos.y)), LdSlot_
and_i(R_T3, R_T0, Pad_Up), branch_le_zero(R_T3, atom_offset(up_y, exit_up_y)), nop, and_i(R_T3, R_T0, Pad_Up), branch_le_zero(R_T3, atom_offset(up_y, exit_up_y)), BdSlot_ nop,
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.y)), add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
atom_label(exit_up_y) atom_label(exit_up_y)
/* D-pad Down → cam.pos.y += 50. Reuses R_T1 from Up. */ /* D-pad Down → cam.pos.y += 50. Reuses R_T1 from Up. */
and_i(R_T3, R_T0, Pad_Down), branch_le_zero(R_T3, atom_offset(down_y, exit_down_y)), nop, and_i(R_T3, R_T0, Pad_Down), branch_le_zero(R_T3, atom_offset(down_y, exit_down_y)), BdSlot_ nop,
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.y)), add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
atom_label(exit_down_y) atom_label(exit_down_y)
/* D-pad Cross → cam.pos.z -= 50. Load pos.z BEFORE the andi. */ /* D-pad Cross → cam.pos.z -= 50. Load pos.z BEFORE the andi. */
load_word(R_T1, R_Cam, O_(Camera,pos.z)), load_word(R_T1, R_Cam, O_(Camera,pos.z)), LdSlot_
and_i(R_T3, R_T0, Pad_Cross), branch_le_zero(R_T3, atom_offset(cross_z, exit_cross_z)), nop, and_i(R_T3, R_T0, Pad_Cross), branch_le_zero(R_T3, atom_offset(cross_z, exit_cross_z)), BdSlot_ nop,
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.z)), add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
atom_label(exit_cross_z) atom_label(exit_cross_z)
/* D-pad Circle → cam.pos.z += 50. Reuses R_T1 from Cross. */ /* D-pad Circle → cam.pos.z += 50. Reuses R_T1 from Cross. */
and_i(R_T3, R_T0, Pad_Circle), branch_le_zero(R_T3, atom_offset(circle_z, exit_circle_z)), nop, and_i(R_T3, R_T0, Pad_Circle), branch_le_zero(R_T3, atom_offset(circle_z, exit_circle_z)), BdSlot_ nop,
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.z)), add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
atom_label(exit_circle_z) atom_label(exit_circle_z)
mac_yield_tail(), mac_yield_tail(),
}; };
@@ -913,7 +581,7 @@ internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)), load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)), load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)), load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
add_ui_self( R_TapePtr, S_(Binds_CubeTri)), LdSlot_ add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
mac_yield() mac_yield()
}; };
@@ -926,20 +594,20 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)), load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)), load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)), load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)), // load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2), LdSlot_ mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2), GteDelay_ load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)), LdSlot_
nop2, gte_cmdw_rotate_translate_perspective_triple, // required cpu -> gte delay slot GteDelay_ nop, gte_cmdw_rotate_translate_perspective_triple,
gte_cmdw_nclip, gte_cmdw_nclip,
gte_mv_from_data_r(R_T0, C2_MAC0), nop, gte_mv_from_data_r(R_T0, C2_MAC0), GteDelay_ nop,
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)), branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
/* BD-slot: Write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer). /* BD-slot: Write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded — * If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
* harmless because the OT entry that points to this prim is created later. */ * harmless because the OT entry that points to this prim is created later. */
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)), BdSlot_ store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase), shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)), load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)), LdSlot_
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
mac_gte_store_g4_p012(R_PrimCursor), mac_gte_store_g4_p012(R_PrimCursor),
@@ -951,7 +619,7 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
add_ui( R_AT, R_0, OrderingTbl_Len), add_ui( R_AT, R_0, OrderingTbl_Len),
set_lt_u( R_AT, R_T1, R_AT), set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop, branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), BdSlot_ nop,
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_G4)), mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_G4)),
mac_format_g4_color(R_PrimCursor, mac_format_g4_color(R_PrimCursor,
/* c0 magenta */ 0xFF, 0x00, 0xFF, /* c0 magenta */ 0xFF, 0x00, 0xFF,
@@ -983,7 +651,7 @@ MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(f
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)), load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)), load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)), load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
add_ui_self( R_TapePtr, S_(Binds_FloorTri)), LdSlot_ add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
mac_yield() mac_yield()
}; };
@@ -1030,7 +698,7 @@ internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitive
, atom_writes(R_TapePtr) , atom_writes(R_TapePtr)
){ ){
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)), load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)), load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)), LdSlot_
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)), add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
/* Calculate byte offset and store directly back to RAM */ /* Calculate byte offset and store directly back to RAM */
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
+113 -282
View File
@@ -1,7 +1,7 @@
#pragma region Vendors #pragma region Vendors
#include <stdio.h> #include <stdio.h>
#include <stdlib.h> #include <stdlib.h>
#include <assert.h> // #include <assert.h>
// #include "libgpu.h" // #include "libgpu.h"
// #include "libetc.h" // #include "libetc.h"
// #include "libgte.h" // #include "libgte.h"
@@ -32,7 +32,7 @@
#pragma region Duffle TUs #pragma region Duffle TUs
#include "duffle/pad.c" #include "duffle/pad.c"
#include "duffle/math.atom.c" #include "duffle/math.atom.h"
#include "duffle/mips.atom.c" #include "duffle/mips.atom.c"
#include "duffle/gte.atom.c" #include "duffle/gte.atom.c"
#include "duffle/gp.atom.c" #include "duffle/gp.atom.c"
@@ -53,15 +53,13 @@
#pragma endregion Hello Joypad TUs #pragma endregion Hello Joypad TUs
enum { enum {
Scratchpad_Loc = 0x1F800000,
};
#define C_scratch(type) C_(type, Scratchpad_Loc)
enum {
Scratchpad_Len = 1024,
MemTape_Len = 512, MemTape_Len = 512,
ResolveLookAtArena_Words = 1024, ResolveLookAtArena_Words = 1024,
ResolveLookAtArena_Size = ResolveLookAtArena_Words * S_(MipsCode), ResolveLookAtArena_Size = ResolveLookAtArena_Words * S_(MipsCode),
CT_InitAtomMem_Words = Kilo_(4),
CT_InitAtomMem_Size = CT_InitAtomMem_Words * S_(MipsCode),
}; };
typedef Struct_(SMemory) { typedef Struct_(SMemory) {
PrimitiveArena primitives; PrimitiveArena primitives;
@@ -85,8 +83,12 @@ typedef Struct_(SMemory) {
// TODO(Ed): We don't need this we can just cast at any point an address to a desired view of scratchpad, we have the address. // TODO(Ed): We don't need this we can just cast at any point an address to a desired view of scratchpad, we have the address.
U4_V scratchpad; // d-cache U4_V scratchpad; // d-cache
U1 ct_init_atom_mem[CT_InitAtomMem_Size];
MipsAtom* normalize_v3s4;
MipsAtom* gte_cross_v3s4;
U1 resolve_look_at_mem[ResolveLookAtArena_Size]; U1 resolve_look_at_mem[ResolveLookAtArena_Size];
MipsAtom* resolve_look_at_atom_addrs[10]; MipsAtom* resolve_look_at_bundle[AtomBundle_Len(resolve_look_at)];
}; };
global SMemory smem; global SMemory smem;
extern SMemory smem; extern SMemory smem;
@@ -131,284 +133,123 @@ I_ void resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4*
} }
FI_ void camera_look_at_c11(Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at_c11(& c->look_at, & c->pos, target, up_in); } FI_ void camera_look_at_c11(Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at_c11(& c->look_at, & c->pos, target, up_in); }
/* Pre-build all 7 chain atoms of the resolve_look_at bundle into the static arena. internal void compile_init_atoms(void) {
* Called ONCE from main() before the frame loop. AtomArena ab = atomarena_make(slice_ut_arr(smem.ct_init_atom_mem));
* After this returns, the smem.resolve_look_at_atom_addrs[] array contains valid MIPS atom pointers RegFile rf = regfile(regfile_abi_mask);
* for the frame-time bundle helper to emit via tb_emit(tb, captured_addr). #define ralloc() regfile_alloc(& rf)
* #define ralloc_v3() { ralloc(), ralloc(), ralloc() }
* 4 unique procs in hello_camera.atom.c (chain atoms 0, 2, 4, 6); atoms 1, 3, 5
* share the GENERIC normalize_v3s4_proc from gte.atom.c (called 3x with different smem.gte_cross_v3s4 = gte_cross_v3s4(& ab,
* O_(ResolveLookAtScratch,...) offsets): RegUse_(gte_cross_v3s4) {
* 0: resolve_look_at__input_and_sub_proc .a = ralloc_v3(),
* 1: normalize_v3s4_proc (fwd → uz; offsets 0, 16) .b = ralloc_v3(),
* 2: resolve_look_at__cross_uz_up_in_to_right_proc .x = ralloc(),
* 3: normalize_v3s4_proc (right → ux; offsets 32, 48) .y = ralloc(),
* 4: resolve_look_at__cross_uz_ux_to_up_proc .z = ralloc(),
* 5: normalize_v3s4_proc (up → uy; offsets 64, 80) });
* 6: resolve_look_at__populate_and_translate_proc regfile_reset(& rf);
*
* Task 12.16 promotion: the bundle-specific resolve_look_at__chain_normalize_proc smem.normalize_v3s4 = build_normalize_v3s4(& ab,
* has been promoted to the generic normalize_v3s4_proc (gte.atom.c), which now RegUse_(build_normalize_v3s4) {
* takes r_scratch + r_src_offset + r_dst_offset as U4 parameters. The 3 callers .scratch = ralloc(),
* pass O_(ResolveLookAtScratch,...) macros as offset args. The metaprogram emits .src_ptr = ralloc(),
* one set of `atom_offset__normalize_v3s4__srav_path__aligned_done` defs .dst_ptr = ralloc(),
* (namespaced by atom name) in duffle/gen/offsets.h, shared by all 3 callers. .recip_est = ralloc(),
* .norm = ralloc(),
* GPR pool per atom: 10 free GPRs (R_T0..R_T3 + R_T5..R_T7 + R_V0 + R_V1 + R_AT). .shift = ralloc(),
* R_T4 is reserved as the wave-context carrier (R_ResolveScratch). .src_x = ralloc(),
*/ // .shift_count = ralloc(), /* dedicated slot for stage-3 → stage-4 shift count */
/* === EXPLICIT REGISTER ALLOCATION TRACKER === .t3 = ralloc(),
* Every GPR used by every atom is tracked below. NO GPR is assigned to .t4 = ralloc(),
* two atoms at overlapping lifetimes. The tape runtime preserves R_T8/R_T9 .t5 = ralloc(),
* (R_AtomJmp/R_TapePtr) and clobbers R_T0-R_T7, R_AT, R_V0, R_V1. });
* R_T4 is reserved as R_ResolveScratch (wave-context carrier). regfile_reset(& rf);
*
* GPR pool: R_T0($8), R_T1($9), R_T2($10), R_T3($11), R_T5($13), assert(ab.used <= CT_InitAtomMem_Size);
* R_T6($14), R_T7($15), R_V0($2), R_V1($3), R_AT($1) #undef ralloc
* Reserved: R_T4($12) = R_ResolveScratch #undef ralloc_v3
* Tape: R_T8($24) = R_AtomJmp, R_T9($25) = R_TapePtr (preserved) }
*
* === ATOM 0: input_and_sub (stages eye/up_in, computes fwd) === internal void compile_resolve_look_at(void) {
* Pop tape → R_T0(target), R_T1(eye), R_T2(up_in).
* Use R_T3,R_T5,R_T6,R_T7 as temps.
* NO conflict with other atoms (each atom has independent lifetime).
*
* === ATOM 1: normalize fwd→uz ===
* r_src_offset=0, r_dst_offset=16.
* r_src_ptr=R_T0, r_dst_ptr=R_T1, r_tmp=R_T2 (preserved for stage 4).
* r_mac1=R_T3, r_mac2=R_T5, r_recip=R_T6, r_lzcr=R_T7, r_shift=R_V0, r_branch=R_V1.
*
* === ATOM 2: cross uz×up_in→right ===
* r_a=R_T0, r_b=R_T1, r_c=R_T2, r_d=R_T3, r_f(out)=R_T5, r_g=R_T6, r_h=R_T7.
*
* === ATOM 3: normalize right→ux ===
* Same GPR pool as atom 1.
*
* === ATOM 4: cross uz×ux→up ===
* r_a=R_T0, r_b=R_T1, r_c=R_T2, r_d=R_T3, r_f(out)=R_T5, r_g=R_T6, r_h=R_T7.
*
* === ATOM 5: normalize up→uy ===
* Same GPR pool as atom 1.
*
* === ATOM 6a: populate (m[][] from ux/uy/uz, t[]=0) ===
* r_look_at=R_T0 (pop tape), r_scratch=R_T4.
* r_pux=R_T1, r_puy=R_T3, r_puz=R_T5.
* r_tmp0=R_T2, r_tmp1=R_T6, r_tmp2=R_V0.
*
* === ATOM 6a.5: set_gte_mt3s2s4 (ctc2 RT matrix) ===
* BAKED atom. Uses R_T3 internally (hardcoded in gte.atom.c).
* NO conflict — different GPR pool, and the atom body hardcodes R_T3
* as the matrix pointer. We DON'T need to assign R_T3 to atom 6a.5
* because it's a baked atom with its own GPR usage.
*
* === ATOM 6b: matrix_vector (RT * (-eye) >> 12) ===
* r_look_at=R_T0 (pop tape), r_scratch=R_T4.
* r_peye=R_T1.
* r_tmp0=R_T2, r_tmp1=R_T3, r_tmp2=R_T5.
* Uses mac_apply_matrix_lv which internally uses these temps.
*
* === ATOM 6c: trans_matrix (off → look_at->t[]) ===
* r_look_at=R_T0 (pop tape), r_scratch=R_T4.
* r_off_ptr=R_T1.
* r_tmp0=R_T2.
*
* === CONFLICT CHECK ===
* All atoms use the same GPR pool R_T0-R_T3, R_T5-R_T7, R_V0-R_V1.
* But atoms are SEQUENTIAL — each atom's lifetime is disjoint from
* the next atom's lifetime. The tape yield handshake between atoms
* preserves R_TapePtr (R_T9) and R_AtomJmp (R_T8).
*
* The GPR pool is SHARED across atoms (they run sequentially, not
* concurrently). Each atom's build call assigns specific R_T* codes
* for that atom's body. The same R_T* code can be reused across atoms
* because the previous atom's body has already yielded.
*/
internal void resolve_look_at_init(void) {
/* Wrap the static arena in a MipsAtomBuilder. */
AtomArena ab = atomarena_make(slice_ut_arr(smem.resolve_look_at_mem)); AtomArena ab = atomarena_make(slice_ut_arr(smem.resolve_look_at_mem));
AtomBundle_resolve_look_at_R bundle = C_(void*, smem.resolve_look_at_bundle);
/* === ATOM 0: input_and_sub === */ /* R_ScratchBase (= R_SP) is a tape carrier preserved across atoms; no carrier
U4 const r_target_ptr = R_T0; /* tape pop → target */ * pin is needed in the regfile. The standard 24-register pool is sufficient. */
U4 const r_eye_ptr = R_T1; /* tape pop → eye */ RegFile rf = regfile(regfile_abi_mask);
U4 const r_up_in_ptr = R_T2; /* tape pop → up_in */ #define ralloc() regfile_alloc(& rf)
U4 const r_tmp0_0 = R_T3; #define ralloc_v3() { ralloc(), ralloc(), ralloc() }
U4 const r_tmp1_0 = R_T5;
U4 const r_tmp2_0 = R_T6;
U4 const r_tmp3_0 = R_T7;
smem.resolve_look_at_atom_addrs[0] = resolve_look_at__input_and_sub_proc(& ab,
R_ResolveScratch,
r_target_ptr, r_eye_ptr, r_up_in_ptr,
r_tmp0_0, r_tmp1_0, r_tmp2_0, r_tmp3_0);
/* === ATOM 1: normalize fwd→uz === */ bundle->input_and_sub = AtomBundleEntry_(resolve_look_at, input_and_sub)(& ab,
U4 const r_src_offset_1 = O_(ResolveLookAtScratch, fwd); RegUse_(resolve_look_at_input_and_sub) {
U4 const r_dst_offset_1 = O_(ResolveLookAtScratch, uz); .target = ralloc(),
U4 const r_src_ptr_1 = R_T0; .eye = ralloc(),
U4 const r_dst_ptr_1 = R_T1; .up_in = ralloc(),
U4 const r_tmp_1 = R_T2; .t0 = ralloc(),
U4 const r_mac1_1 = R_T3; .t1 = ralloc(),
U4 const r_mac2_1 = R_T5; .t2 = ralloc(),
U4 const r_recip_1 = R_T6; .t3 = ralloc(),
U4 const r_lzcr_1 = R_T7; .t4 = ralloc(),
U4 const r_shift_1 = R_V0; });
U4 const r_branch_1 = R_V1; regfile_reset(& rf);
smem.resolve_look_at_atom_addrs[1] = normalize_v3s4_proc(& ab,
R_ResolveScratch,
r_src_offset_1, r_dst_offset_1,
r_src_ptr_1, r_dst_ptr_1, r_tmp_1,
r_mac1_1, r_mac2_1, r_recip_1, r_lzcr_1,
r_shift_1, r_branch_1);
/* === ATOM 2: cross uz×up_in→right === */ bundle->normalize_fwd_uz = smem.normalize_v3s4;
U4 const r_a_2 = R_T0; bundle->cross_to_right = smem.gte_cross_v3s4;
U4 const r_b_2 = R_T1; bundle->normalize_right_ux = smem.normalize_v3s4;
U4 const r_c_2 = R_T2; bundle->cross_to_up = smem.gte_cross_v3s4;
U4 const r_d_2 = R_T3; bundle->normalize_up_uy = smem.normalize_v3s4;
U4 const r_f_2 = R_T5; /* out ptr (HARDCODED in body: scratch+32) */
U4 const r_g_2 = R_T6; /* a ptr = scratch+16 */
U4 const r_h_2 = R_T7; /* b ptr = scratch+128 */
smem.resolve_look_at_atom_addrs[2] = resolve_look_at__cross_uz_up_in_to_right_proc(& ab,
R_ResolveScratch,
r_a_2, r_b_2, r_c_2, r_d_2, r_f_2, r_g_2, r_h_2);
/* === ATOM 3: normalize right→ux === */ bundle->pop_mv_trans = resolve_look_at__pop_mv_trans(& ab,
U4 const r_src_offset_3 = O_(ResolveLookAtScratch, right); RegUse_(resolve_look_at__pop_mv_trans){
U4 const r_dst_offset_3 = O_(ResolveLookAtScratch, ux); .look_at = ralloc(),
U4 const r_src_ptr_3 = R_T0; .eye = ralloc(),
U4 const r_dst_ptr_3 = R_T1; .row = ralloc_v3(),
U4 const r_tmp_3 = R_T2; .t6 = ralloc(),
U4 const r_mac1_3 = R_T3; .t7 = ralloc(),
U4 const r_mac2_3 = R_T5; .t8 = ralloc(),
U4 const r_recip_3 = R_T6; });
U4 const r_lzcr_3 = R_T7;
U4 const r_shift_3 = R_V0;
U4 const r_branch_3 = R_V1;
smem.resolve_look_at_atom_addrs[3] = normalize_v3s4_proc(& ab,
R_ResolveScratch,
r_src_offset_3, r_dst_offset_3,
r_src_ptr_3, r_dst_ptr_3, r_tmp_3,
r_mac1_3, r_mac2_3, r_recip_3, r_lzcr_3,
r_shift_3, r_branch_3);
/* === ATOM 4: cross uz×ux→up === */
U4 const r_a_4 = R_T0;
U4 const r_b_4 = R_T1;
U4 const r_c_4 = R_T2;
U4 const r_d_4 = R_T3;
U4 const r_f_4 = R_T5; /* out ptr (HARDCODED: scratch+64) */
U4 const r_g_4 = R_T6; /* a ptr = scratch+16 */
U4 const r_h_4 = R_T7; /* b ptr = scratch+48 */
smem.resolve_look_at_atom_addrs[4] = resolve_look_at__cross_uz_ux_to_up_proc(& ab,
R_ResolveScratch,
r_a_4, r_b_4, r_c_4, r_d_4, r_f_4, r_g_4, r_h_4);
/* === ATOM 5: normalize up→uy === */
U4 const r_src_offset_5 = O_(ResolveLookAtScratch, up);
U4 const r_dst_offset_5 = O_(ResolveLookAtScratch, uy);
U4 const r_src_ptr_5 = R_T0;
U4 const r_dst_ptr_5 = R_T1;
U4 const r_tmp_5 = R_T2;
U4 const r_mac1_5 = R_T3;
U4 const r_mac2_5 = R_T5;
U4 const r_recip_5 = R_T6;
U4 const r_lzcr_5 = R_T7;
U4 const r_shift_5 = R_V0;
U4 const r_branch_5 = R_V1;
smem.resolve_look_at_atom_addrs[5] = normalize_v3s4_proc(& ab,
R_ResolveScratch,
r_src_offset_5, r_dst_offset_5,
r_src_ptr_5, r_dst_ptr_5, r_tmp_5,
r_mac1_5, r_mac2_5, r_recip_5, r_lzcr_5,
r_shift_5, r_branch_5);
/* === ATOM 6a: populate (m[][] from ux/uy/uz, t[]=0) === */
U4 const r_look_at_6a = R_T0; /* tape pop → look_at* */
U4 const r_scratch_6a = R_ResolveScratch;
U4 const r_pux_6a = R_T1;
U4 const r_puy_6a = R_T3;
U4 const r_puz_6a = R_T5;
U4 const r_tmp0_6a = R_T2;
U4 const r_tmp1_6a = R_T6;
U4 const r_tmp2_6a = R_V0;
smem.resolve_look_at_atom_addrs[6] = resolve_look_at__populate_proc(& ab,
r_look_at_6a, r_scratch_6a,
r_pux_6a, r_puy_6a, r_puz_6a,
r_tmp0_6a, r_tmp1_6a, r_tmp2_6a);
/* === ATOM 6a.5: set_gte_mt3s2s4 (BAKED — ctc2 RT matrix) ===
* This is a BAKED atom from gte.atom.c. Its body hardcodes R_T3 as
* the matrix pointer (popped from tape). It does NOT need GPR
* assignment from us — it has its own internal GPR usage.
* We just take its address. */
smem.resolve_look_at_atom_addrs[7] = (MipsAtom*) & set_gte_mt3s2s4;
/* === ATOM 6b: matrix_vector (RT * (-eye) >> 12) ===
* Uses mac_apply_matrix_lv component macro which internally uses
* r_t0 for the RT matrix load + V0 load, then r_t0/r_t1/r_t2
* for the mfc2/store. We pass our GPRs. */
U4 const r_scratch_6b = R_ResolveScratch;
U4 const r_peye_6b = R_T1; /* scratch+96 (packed V0 dst, then off dst) */
U4 const r_look_at_6b = R_T0; /* tape pop → look_at* */
U4 const r_tmp0_6b = R_T2;
U4 const r_tmp1_6b = R_T3;
U4 const r_tmp2_6b = R_T5;
smem.resolve_look_at_atom_addrs[8] = resolve_look_at__matrix_vector_proc(& ab,
r_scratch_6b, r_peye_6b, r_look_at_6b,
r_tmp0_6b, r_tmp1_6b, r_tmp2_6b);
/* === ATOM 6c: trans_matrix (off → look_at->t[]) === */
U4 const r_look_at_6c = R_T0; /* tape pop → look_at* */
U4 const r_scratch_6c = R_ResolveScratch;
U4 const r_off_ptr_6c = R_T1; /* &scratch.eye (= off dst) */
U4 const r_tmp0_6c = R_T2;
smem.resolve_look_at_atom_addrs[9] = resolve_look_at__trans_matrix_proc(& ab,
r_look_at_6c, r_scratch_6c, r_off_ptr_6c, r_tmp0_6c);
/* Sanity check: arena didn't overflow. */ /* Sanity check: arena didn't overflow. */
assert(ab.used <= ResolveLookAtArena_Size); assert(ab.used <= ResolveLookAtArena_Size);
#undef ralloc
} }
/* Emit the resolve_look_at bundle into the tape. Called once per frame from update(). /* Emit the resolve_look_at bundle into the tape. Called once per frame from update(). */
* The 7 chain atoms are pre-built at init time (resolve_look_at_init) and referenced by address via smem.resolve_look_at_atom_addrs[]. I_ void resolve_look_at(TapeBuilder_R tb
* Per-frame work: 7 tb_emit (atom pointer emissions) + 5 tb_data (C-side pointers for atom 0 + look_at for atom 6).
*
* Binds_ contract (the field-name labels are for human readability):
* Atom 0 input_and_sub target(4) eye(4) up_in(4) scratch_base(4) = 4 words
* Atoms 1-5 (no tape data — atom uses r_scratch + offset internally)
* Atom 6 populate_and_translate look_at(4) = 1 word
* ----
* 5 tb_data words total per frame.
*/
I_ void resolve_look_at(
TapeBuilder_R tb
, MT3_S2S4* look_at , MT3_S2S4* look_at
, P3_S4* eye , P3_S4* eye
, P3_S4* target , P3_S4* target
, V3_S4* up_in , V3_S4* up_in
){ ){
tb_emit(tb, smem.resolve_look_at_atom_addrs[0]); { /* Typed view of the scratchpad for field-address arithmetic. */
ResolveLookAtScratch* sp = C_scratch(ResolveLookAtScratch*);
AtomBundle_resolve_look_at_R bundle = C_(void*, smem.resolve_look_at_bundle);
tb_emit(tb, bundle->input_and_sub); {
tb_data(tb, u4_(target)); tb_data(tb, u4_(target));
tb_data(tb, u4_(eye)); tb_data(tb, u4_(eye));
tb_data(tb, u4_(up_in)); tb_data(tb, u4_(up_in));
tb_data(tb, u4_(smem.scratchpad));
} }
tb_emit(tb, bundle->normalize_fwd_uz); {
tb_emit(tb, smem.resolve_look_at_atom_addrs[1]); { } tb_data(tb, u4_(O_(ResolveLookAtScratch, fwd) | (O_(ResolveLookAtScratch, uz) << 16)));
tb_emit(tb, smem.resolve_look_at_atom_addrs[2]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[3]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[4]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[5]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[6]); {
tb_data(tb, u4_(look_at));
} }
tb_emit(tb, smem.resolve_look_at_atom_addrs[7]); { tb_emit(tb, bundle->cross_to_right); {
tb_data(tb, u4_(look_at)); tb_data(tb, u4_(& sp->uz));
tb_data(tb, u4_(& sp->up_in));
tb_data(tb, u4_(& sp->right));
} }
tb_emit(tb, smem.resolve_look_at_atom_addrs[8]); { tb_emit(tb, bundle->normalize_right_ux); {
tb_data(tb, u4_(look_at)); tb_data(tb, u4_(O_(ResolveLookAtScratch, right) | (O_(ResolveLookAtScratch, ux) << 16)));
} }
tb_emit(tb, smem.resolve_look_at_atom_addrs[9]); { tb_emit(tb, bundle->cross_to_up); {
tb_data(tb, u4_(& sp->uz));
tb_data(tb, u4_(& sp->ux));
tb_data(tb, u4_(& sp->up));
}
tb_emit(tb, bundle->normalize_up_uy); {
tb_data(tb, u4_(O_(ResolveLookAtScratch, up) | (O_(ResolveLookAtScratch, uy) << 16)));
}
tb_emit(tb, bundle->pop_mv_trans); {
tb_data(tb, u4_(look_at)); tb_data(tb, u4_(look_at));
} }
} }
@@ -425,9 +266,9 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
tb_emit_(pad_bios_snapshot); tb_emit_(pad_bios_snapshot);
tb_data_(raw, & smem.pad_raw[0]); tb_data_(raw, & smem.pad_raw[0]);
tb_data_(state, & smem.pad[0]); tb_data_(state, & smem.pad[0]);
tb_emit_(pad_bios_snapshot); // tb_emit_(pad_bios_snapshot);
tb_data_(raw, & smem.pad_raw[1]); // tb_data_(raw, & smem.pad_raw[1]);
tb_data_(state, & smem.pad[1]); // tb_data_(state, & smem.pad[1]);
tb_emit_(pad_input_cam); tb_emit_(pad_input_cam);
tb_data_(state, & smem.pad[0]); tb_data_(state, & smem.pad[0]);
@@ -448,12 +289,6 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
gknown V3_S4_R acc = & smem.cube.accel; gknown V3_S4_R acc = & smem.cube.accel;
add_v3s4(vel, acc[0]); add_v3s4(vel, acc[0]);
add_v3s4_fp(pos, vel[0]); add_v3s4_fp(pos, vel[0]);
// vel->x += acc->x;
// vel->y += acc->y;
// vel->z += acc->z;
// pos->x += vel->x;
// pos->y += vel->y;
// pos->z += vel->z;
if (pos->y + 150 > smem.floor.pos.y) vel->y *= -1; if (pos->y + 150 > smem.floor.pos.y) vel->y *= -1;
@@ -486,9 +321,6 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
gte_matrix_set_rotation (& smem.tform_view); gte_matrix_set_rotation (& smem.tform_view);
gte_matrix_set_translation(& smem.tform_view); gte_matrix_set_translation(& smem.tform_view);
// gte_matrix_set_rotation (& smem.tform_world);
// gte_matrix_set_translation(& smem.tform_world);
U4 prim_base = u4_(pa->buf[smem.active_buf_id]); U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
U4 prim_cursor = prim_base + pa->used; U4 prim_cursor = prim_base + pa->used;
@@ -508,7 +340,7 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
tb_data(& tb, u4_(& pa->used)); tb_data(& tb, u4_(& pa->used));
tb_data(& tb, prim_base); tb_data(& tb, prim_base);
} }
tape_run_a02_s07(tb_slice(tb));// Fire off the tape (bigger-clobber variant). tape_run(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
// smem.cube.rot.y += 30; // smem.cube.rot.y += 30;
} }
@@ -550,7 +382,7 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
tb_data(& tb, u4_(& pa->used)); tb_data(& tb, u4_(& pa->used));
tb_data(& tb, prim_base); tb_data(& tb, prim_base);
} }
tape_run_a02_s07(tb_slice(tb));// Fire off the tape (bigger-clobber variant). tape_run(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
// C-side state (pa->used) has already been updated by the tape! // C-side state (pa->used) has already been updated by the tape!
// smem.floor.rot.y += 5; // smem.floor.rot.y += 5;
@@ -602,8 +434,8 @@ int main(void)
/* Direct BIOS: poll both ports during VBlank. */ /* Direct BIOS: poll both ports during VBlank. */
pad_bios_init_start(& smem.pad_raw[0], & smem.pad_raw[1]); pad_bios_init_start(& smem.pad_raw[0], & smem.pad_raw[1]);
/* Pre-build the resolve_look_at bundle atoms into the static arena. */ compile_init_atoms();
resolve_look_at_init(); compile_resolve_look_at();
/* Pinned registers for the GPU init atom. */ /* Pinned registers for the GPU init atom. */
register U4* io_base_addr rgcc(R_IO_BaseAddr) = u4_r(IO_BASE_ADDR); register U4* io_base_addr rgcc(R_IO_BaseAddr) = u4_r(IO_BASE_ADDR);
@@ -624,4 +456,3 @@ int main(void)
return 0; return 0;
} }
GCC_OPTIMIZATION_ENABLE GCC_OPTIMIZATION_ENABLE
+2 -2
View File
@@ -25,7 +25,7 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c);
#pragma region MACs (Mips Atom components) #pragma region MACs (Mips Atom components)
FI_ Slice_MipsCode ac_put_disp_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port) FI_ Slice_MipsCode ac_put_disp_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_put_disp_env, ab, { MipsAtomComp_Proc_(ab, {
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)). // Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR // Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port), mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
@@ -36,7 +36,7 @@ MipsAtomComp_Proc_(ac_put_disp_env, ab, {
}) })
FI_ Slice_MipsCode ac_put_draw_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port) FI_ Slice_MipsCode ac_put_draw_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_put_draw_env, ab, { MipsAtomComp_Proc_(ab, {
/* /*
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings. * ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
* References: * References:
+1
View File
@@ -532,6 +532,7 @@ function build-hello_camera {
$compile_args = @() $compile_args = @()
$compile_args += $f_debug $compile_args += $f_debug
$compile_args += ($f_define + 'BUILD_DEBUG')
$compile_args += $f_optimize_none $compile_args += $f_optimize_none
# $compile_args += $f_optimize_intrinsics # $compile_args += $f_optimize_intrinsics
# $compile_args += $f_optimize_size # $compile_args += $f_optimize_size
+35 -2671
View File
File diff suppressed because it is too large Load Diff
+912
View File
@@ -0,0 +1,912 @@
--- duffle_emit.lua — project_emission + decl finders.
local scan = require("duffle_scan")
local isa = require("duffle_isa")
local M = {}
for k, v in pairs(scan) do M[k] = v end
for k, v in pairs(isa) do M[k] = v end
-- Section 8: Cross-source component-body index + word-event expansion
-- ════════════════════════════════════════════════════════════════════════════
--
-- Shared, memoized helpers: a single emitted-word event stream that every downstream pass reads from,
-- built once from the pre-tokenized bodies.
--- @class ComponentBodyEntry
--- @field body_tokens table -- pre-tokenized {{tok=string, rel=integer}, ...}
--- @field body_off integer -- byte offset of body[1] in `source`
--- @field line_of fun(pos:integer):integer -- byte-offset → 1-based line number in `source`
--- @field source string -- absolute path of the source containing the declaration
--- @field declaration integer -- 1-based line number of the MipsAtomComp_(ac_X) declaration
--- @field kind string -- "comp_bare" | "comp_proc"
-- The cross-source component-body index is owned by the corpus (`corpus.component_body_index`, populated by `passes/components.lua`).
-- Consumers (`passes/static_analysis.lua`, `passes/emission_model.lua`) read it directly; per-pass memoization helpers stay out of scope.
-- ASCII byte constants used by split_call_args (kept local to keep Section 8 self-contained).
local E_BYTE_OPEN_PAREN = 0x28
local E_BYTE_OPEN_BRACE = 0x7B
local E_BYTE_OPEN_BRACK = 0x5B
local E_BYTE_DQUOTE = 0x22
local E_BYTE_SQUOTE = 0x27
local E_BYTE_COMMA = 0x2C
-- Map an open-delimiter byte to its matching close string for read_balanced.
local E_OPEN_CLOSE = {
[E_BYTE_OPEN_PAREN] = ")",
[E_BYTE_OPEN_BRACE] = "}",
[E_BYTE_OPEN_BRACK] = "]",
}
--- Split the INSIDE of a `f(...)` call on top-level commas.
--- Honors nested parens / braces / brackets and skips strings / comments.
--- Returns a list of trimmed argument strings in source order.
--- (Mirrors split_top_level_commas but for paren-body args; intentionally distinct so a caller's brace-body split isn't confused with an arg list.)
--- @param inner string
--- @return string[]
local function split_call_args(inner)
local args = {}
if not inner or inner == "" then return args end
local pos = 1
local len = #inner
local start = 1
while pos <= len do
local c = inner:byte(pos)
local close = E_OPEN_CLOSE[c]
if close then
local _, after = M.read_balanced(inner, string.char(c), close, pos)
pos = after
elseif c == E_BYTE_DQUOTE or c == E_BYTE_SQUOTE then
pos = M.skip_str_or_cmt(inner, pos)
elseif c == E_BYTE_COMMA then
args[#args + 1] = M.trim(inner:sub(start, pos - 1))
start = pos + 1
pos = pos + 1
else
pos = pos + 1
end
end
if start <= len then args[#args + 1] = M.trim(inner:sub(start, len)) end
return args
end
--- Extract the leading identifier + top-level args list from a token string.
--- Returns (ident, args). For tokens without a `(...)` call, args is `{}`.
--- @param tok string
--- @return string, string[]
local function token_ident_and_args(tok)
local ident, after = M.read_ident(tok, 1)
if not ident then return "?", {} end
local paren_pos = M.skip_ws_and_cmt(tok, after)
if tok:sub(paren_pos, paren_pos) ~= "(" then return ident, {} end
local inner = M.read_parens(tok, paren_pos)
if not inner then return ident, {} end
return ident, split_call_args(inner)
end
-- The macro-name prefix that marks a `mac_X(...)` component invocation.
local E_MAC_PREFIX = "mac_"
local E_MAC_PREFIX_LEN = 4
--- Expand a body entry into the flat sequence of emitted machine-word events.
---
--- Semantics (one event per emitted machine word):
--- * Direct one-word encoders `load_word`, `add_ui`, `nop`, `gte_lw`, ...: One event with `ident` = leading ident, `args` = parsed top-level args.
--- * `nop2` (2-word pseudo-instruction): Two events, both with `ident = "nop"` so the recognized "this slot is a no-op" semantic is visible to downstream analyses.
--- * Any other N-word token in `word_counts`: N events sharing the same `ident` + `args` so useful CPU words retire slots in the cycle budget.
--- * Known `mac_X(...)` calls: Recursively expand the indexed component body, including nested components. Every event from the expansion carries:
--- - `source` / `line` = the COMPONENT'S source path + the line of the token within the component body (i.e. "definition site").
--- - `call_source` / `call_line` = the ROOT atom's source path + call-site line, PRESERVED across recursion so nested events still point at the original root.
--- * Unknown `mac_X` (not in `component_index`): fall back to `word_counts[ident]` if present; otherwise emit one opaque event so the cycle budget accounts for the word.
--- * Marker Tokens (`atom_label(...)` / `atom_offset(...)`): Zero events (they are pure metaprogram hints).
---
--- Cycle protection: a per-expansion `visiting` set tracks components currently on the expansion stack;
--- a re-entry produces a deterministic `{kind = "cycle", ...}` error and aborts that branch (does NOT hang, does NOT recurse).
---
--- Pure: reads `body_entry` / `component_index` / `word_counts`. Memoization is the caller's responsibility.
--- Callers wanting `word_events` / `word_event_errors` precomputed for many atoms should memoize them per atom.
--- @param body_entry table -- `{body_tokens, body_off, line_of, source, declaration}` (declaration = root atom's atom.line)
--- @param component_index table -- the bare-name → ComponentBodyEntry map from M.get_component_body_index
--- @param word_counts table -- macro name → emitted-word count (from `ctx.shared.word_counts`)
--- @return WordEvent[], WordEventError[]
-- ════════════════════════════════════════════════════════════════════════════
-- Section 11: project_emission (per-atom emission projection)
-- ════════════════════════════════════════════════════════════════════════════
--
-- Per-atom emission projection is owned by `passes/emission_model.lua`.
-- The projection is built from the root atom body only; invocation ancestry recursively expands nested components.
-- The items stream is the single ordered source of truth; `word_events` and `markers` are dense views over it.
--
-- The helper below operates on a body string (not a body_entry) so the pass can call it without depending on the older SourceScan / body_off conventions.
-- component_index argument is reserved for recursive component expansion.
-- word_counts table is authored-metadata + current-component count table.
--- @class EmissionProjection
--- @field items table[] -- Ordered stream of word|label|offset|invoke_begin|invoke_end
--- @field word_events table[] -- Dense view of items where kind == "word"
--- @field markers table[] -- Dense view of items where kind == "label"|"offset"
--- @field invocations InvocationRecord[] -- dense view of items where kind == "invoke_begin"|"invoke_end"
--- @field errors table[] -- Token-resolution failures surfaced without fail-loud
--- @field warnings table[] -- Opaque warnings (e.g. unknown uncounted macro)
--- @class InvocationRecord
--- Lives at `atom.paths.invocations[*]`. Constructed once at the single invocation-construction site
--- (`emit_invoke_begin` inside `_project_emission_inner`); `invoke_begin` / `invoke_end` markers in the items stream share the same `id`.
--- @field id integer -- 1-based, monotonic per-atom invocation id (0 is reserved for "no open invocation")
--- @field parent_id integer -- 0 for the outermost (root) call; otherwise the id of the immediately enclosing invocation
--- @field kind string -- "comp_bare" | "comp_proc" (component form that triggered the expansion)
--- @field component_name string -- Bare component name without the `mac_` prefix
--- @field call_text string -- Immediate `mac_X(...)` token text (or root call text for the outermost entry)
--- @field root_call_text string -- IMMUTABLE outermost `mac_X(...)` token text for every word emitted in this call's expansion
--- @field call_path string -- Source path of the call site (root atom source for direct calls, component source for nested expansions)
--- @field call_line integer -- Source line of the call site
--- @field def_path string -- Source path of the component definition
--- @field def_line integer -- Source line of the component declaration
--- @field start_pos integer -- 0-based emitted-word position of the FIRST word inside this invocation (the value of `word_idx` AT `emit_invoke_begin` time, BEFORE the first word is emitted). Words emitted inside this invocation occupy `start_pos..start_pos+#body_lines-1` (inclusive, 0-based). Downstream DWARF/provenance consumers MUST read this; do NOT reconstruct it from `start_word` (which is the 1-based items index including `invoke_begin`/`invoke_end` markers).
--- @field end_pos integer -- 0-based position of the LAST word inside this invocation (set by `emit_invoke_end` to `word_idx - 1` AFTER all body words are emitted).
--- @field start_word integer -- 1-based items index of the `invoke_begin` item
--- @field end_word integer -- 1-based items index of the `invoke_end` item (set by `emit_invoke_end`)
--- @field word_count integer -- Number of `word` items emitted between `start_word` and `end_word` (inclusive)
--- @field debug_skip boolean -- `debug_skip` stamp; true iff `corpus.components[name].debug_skip` is true at construction. Always boolean (never `nil`).
--- @field errors table[] -- Per-invocation construction errors (cycle / count_mismatch); does not include pass-level errors
-- Internal recursive walker. The items stream holds every emitted event in order; `word_events`, `markers`,
-- `invocations`, `errors`, `warnings` are dense views / side outputs appended alongside.
--
-- Output rules:
-- * `word` items record: `invocation_ids` (innermost last) and `outermost_invocation_id` (0 if no invocation is open).
-- * `invoke_begin` / `invoke_end` items are zero-width at the current word index; the same `word_index` is recorded on both.
-- * `root_call_text` is the outermost `mac_X(...)` token text for every word emitted inside a component expansion;
-- it is `nil` for direct words emitted from the root atom body.
-- * `call_text` is the IMMEDIATE top-level token spelling for the word (for nested words this is the inner `mac_X(...)` token;
-- for direct words it is the trimmed encoder token).
-- * `def_path` / `def_line` are the definition site of the current body (component source for nested words; root atom source for direct words, filled in by the pass caller).
-- * Unknown uncounted macros emit one opaque word + one warning. Unknown metadata-backed macros (entry in `word_counts`) emit the declared word count, no warning.
-- * Cycle detection uses an active DFS stack (`visiting`); a cycle appends a construction error to BOTH the projection errors and the cycle invocation's own errors,
-- then breaks out without recursing (the cycle entry still receives an invocation ID + paired `invoke_begin` / `invoke_end` items, so the boundary invariant is preserved).
-- * Component declared-count mismatch (declared vs. measured) is a construction error (kind = "count_mismatch"); recorded on the invocation record and pass-level errors list.
-- * Final boundary check: if any invocation is still open at end of walk, surface a "unbalanced" construction error.
local function _project_emission_inner(root_body_entry, ctx_table)
local items = {}
local word_events = {}
local markers = {}
local invocations = {}
local errors = {}
local warnings = {}
local word_idx = 0
local invocation_stack = {} -- stack of currently-open invocation records
local next_inv_id = 0
local reg_use_schema = ctx_table.reg_use_schema
local reg_use_param = ctx_table.reg_use_param
local atom_name = ctx_table.atom_name
local slot_readonly = {}
if reg_use_schema then
for _, slot in ipairs(reg_use_schema.slots or {}) do
slot_readonly[slot.name] = slot.readonly == true
end
end
local function apply_sub(sub_map, operand)
if not (sub_map and type(operand) == "string") then return operand end
if sub_map[operand] then return sub_map[operand] end
local dot = operand:find(".", 1, true)
if dot then
local head = operand:sub(1, dot - 1)
local mapped = sub_map[head]
if type(mapped) == "string" then
return mapped .. operand:sub(dot)
end
end
return operand
end
local function resolve_gpr_key(operand)
if type(operand) ~= "string" then return nil end
if operand:sub(1, 2) == "R_" then return operand end
if not (reg_use_schema and reg_use_param) then return nil end
local prefix = reg_use_param .. "."
if operand:sub(1, #prefix) ~= prefix then return nil end
local member_path = operand:sub(#prefix + 1)
local slot = reg_use_schema.alias_to_slot[member_path]
if not slot then return nil, member_path end
return "reguse:" .. atom_name .. ":" .. slot, nil, slot
end
local function open_invocation_ids_snapshot()
local ids = {}
for _, inv in ipairs(invocation_stack) do
ids[#ids + 1] = inv.id
end
return ids
end
local function emit_word(encoder, args, line, word_call_text,
def_source_now, def_line_now,
immediate_call_text, root_call_text_w, sub_map)
local inv_ids = open_invocation_ids_snapshot()
local outermost = inv_ids[1] or 0
-- For words emitted at the root atom body, `immediate_call_text` is nil and the walker's `word_call_text` (the word's own token, e.g. "nop") becomes the effective call_text.
-- For words emitted inside a component expansion, `immediate_call_text` is the immediate outer `mac_X(...)` token text;
-- The call that triggered the body expansion we're currently walking.
local eff_call_text = immediate_call_text or word_call_text
local eff_root_call_text = root_call_text_w
local gpr_keys = nil
if reg_use_schema or sub_map then
gpr_keys = {}
for pos, arg in ipairs(args or {}) do
local effective = apply_sub(sub_map, arg)
local key, unresolved, slot = resolve_gpr_key(effective)
gpr_keys[pos] = key
if unresolved then
errors[#errors + 1] = {
kind = "reguse_unresolved",
line = line,
msg = string.format("RegUse operand %q does not resolve in schema %q",
effective, (reg_use_schema and reg_use_schema.name) or "?"),
}
end
if key and slot and slot_readonly[slot] then
local row = M.instr(encoder)
if row and row.writes then
for _, wpos in ipairs(row.writes) do
if wpos == pos then
errors[#errors + 1] = {
kind = "reguse_const_write",
line = line,
msg = string.format("RegUse slot %q is Reg const; %s writes it",
slot, encoder),
}
end
end
end
end
end
end
if not reg_use_schema then
gpr_keys = nil
end
items[#items + 1] = {
kind = "word",
encoder = encoder,
args = args,
i = word_idx,
word_count = 1,
line = line,
call_text = eff_call_text,
root_call_text = eff_root_call_text,
invocation_ids = inv_ids,
outermost_invocation_id = outermost,
gpr_keys = gpr_keys,
}
word_events[#word_events + 1] = {
i = word_idx,
encoder = encoder,
args = args,
def_path = def_source_now or "",
def_line = def_line_now or 0,
call_text = eff_call_text,
root_call_text = eff_root_call_text,
invocation_ids = inv_ids,
outermost_invocation_id = outermost,
word_count = 1,
gpr_keys = gpr_keys,
}
word_idx = word_idx + 1
end
local function emit_marker(kind, name, target, line,
immediate_call_text, root_call_text_w,
consuming_encoder, consuming_arg_pos)
local inv_ids = open_invocation_ids_snapshot()
local outermost = inv_ids[1] or 0
-- Markers carry the open invocation stack snapshot. `call_text` / `root_call_text` belong to words, not markers — markers are zero-width and skip per-word call-site attribution.
-- `consuming_encoder` + `consuming_arg_pos` carry the surrounding control-transfer instruction context
-- (e.g. `branch_le_zero` consuming its 3rd argument, or `jump` / `call_addr` consuming their only argument).
-- `passes/offsets.lua` reads these to dispatch per-consuming-instruction offset encoding.
-- nil for top-level markers (where the marker is the entire token — no surrounding consuming instruction).
local it = {
kind = kind,
name = name,
line = line,
word_index = word_idx,
invocation_ids = inv_ids,
outermost_invocation_id = outermost,
}
if target ~= nil then it.target = target end
if consuming_encoder then it.consuming_encoder = consuming_encoder end
if consuming_arg_pos then it.consuming_arg_pos = consuming_arg_pos end
items[#items + 1] = it
markers[#markers + 1] = {
kind = kind,
name = name,
line = line,
word_index = word_idx,
target = target,
consuming_encoder = consuming_encoder,
consuming_arg_pos = consuming_arg_pos,
}
end
-- Count top-level commas in `tok` between position `from_pos` (inclusive) and `to_pos` (exclusive).
-- Tracks paren depth so commas inside nested () don't count. Skips string literals + comments.
-- Used by `emit_embedded_markers` to compute `consuming_arg_pos` for each embedded marker.
local function count_top_level_commas(tok, from_pos, to_pos)
local depth = 0
local count = 0
local i = from_pos
while i < to_pos do
local c = tok:sub(i, i)
if c == "'" or c == '"' then
local next_pos = M.skip_str_or_cmt(tok, i)
i = (next_pos > i) and next_pos or (i + 1)
elseif c == "/" and tok:sub(i + 1, i + 1) == "/" then
-- line comment: skip to end of line
local nl = tok:find("\n", i, true)
i = (nl and nl + 1) or (#tok + 1)
elseif c == "/" and tok:sub(i + 1, i + 1) == "*" then
-- block comment: skip to matching */
local close = tok:find("*/", i + 2, true)
i = (close and close + 2) or (#tok + 1)
elseif c == "(" then
depth = depth + 1
i = i + 1
elseif c == ")" then
depth = depth - 1
i = i + 1
elseif c == "," and depth == 0 then
count = count + 1
i = i + 1
else
i = i + 1
end
end
return count
end
-- Find the position of the consuming instruction's open paren (the `(` that starts the consuming instruction's argument list).
-- Returns nil if the token's leading text isn't an ident followed by `(` (e.g. the ident is at the start of a non-instruction token).
local function find_consuming_paren(tok)
local i = 1
while i <= #tok do
local c = tok:sub(i, i)
if c == "(" then return i end
if not c:match("[%w_]") and c ~= " " then return nil end
i = i + 1
end
return nil
end
local function emit_embedded_markers(tok, tok_line, consuming_encoder)
-- When called with a non-nil `consuming_encoder`, the marker is nested inside that instruction's argument list.
-- We compute each marker's arg position by counting top-level commas between the consuming instruction's `(` and the marker's start.
local consuming_paren = nil
if consuming_encoder then consuming_paren = find_consuming_paren(tok) end
local pos = 1
while pos <= #tok do
-- Trim leading whitespace and comments before each scan.
pos = M.skip_ws_and_cmt(tok, pos)
if pos > #tok then break end
local ident, after = M.read_ident(tok, pos)
if not ident then
-- Not an ident: token is a string or comment; skip or one-step.
local next_pos = M.skip_str_or_cmt(tok, pos)
pos = (next_pos > pos) and next_pos or (pos + 1)
goto continue_loop
end
if M.DELAY_MARKERS[ident] then
local arg_pos = nil
if consuming_encoder and consuming_paren then
arg_pos = count_top_level_commas(tok, consuming_paren + 1, pos) + 1
end
emit_marker("delay", ident, nil, tok_line, nil, nil, consuming_encoder, arg_pos)
pos = after
goto continue_loop
end
if ident ~= "atom_label" and ident ~= "atom_offset" then
-- Ordinary ident; nothing to emit, step past the ident only.
pos = after
goto continue_loop
end
-- Marker ident: parse the (...) arguments.
local open = M.skip_ws_and_cmt(tok, after)
local inner, after_paren = M.read_parens(tok, open)
if not inner then
-- (...) Unreadable: fall back to non-marker behavior.
pos = after
goto continue_loop
end
-- Commit: label takes 1 arg, offset takes 2.
-- For embedded markers, propagate the consuming_encoder + the marker's arg position
-- (1-based) so `passes/offsets.lua` can dispatch per-consuming-instruction offset encoding.
-- Top-level markers (no consuming_encoder) get nil for both — the offsets pass treats
-- them as branch-equivalent for backward compatibility.
local arg_pos = nil
if consuming_encoder and consuming_paren then
arg_pos = count_top_level_commas(tok, consuming_paren + 1, pos) + 1
end
local args = split_call_args(inner)
if ident == "atom_label" then emit_marker("label", args[1] or "", nil, tok_line, nil, nil, consuming_encoder, arg_pos)
else emit_marker("offset", args[1] or "", args[2] or "", tok_line, nil, nil, consuming_encoder, arg_pos)
end
pos = after_paren
::continue_loop::
end
end
local function emit_invoke_begin(inv_kind, component_name, call_text,
root_call_text, call_path, call_line)
next_inv_id = next_inv_id + 1
-- Invocation-level debug_skip stamp: Emission pass owns `atom.paths.invocations[*].debug_skip`.
-- The stamp is resolved from the `corpus.components[name]` registry (passed in via `ctx_table.components` by `emission_model.run`),
-- Unmarked components stamp `false` (not `nil`) so consumers can dispatch on the boolean without nil checks.
--
-- The walker has already found the component body in `ctx_table.component_index[component_name]`, so the matching entry MUST exist in `ctx_table.components[component_name]`
-- (both registries are populated from the same source by the components pass).
-- A missing entry is a corpus-plumbing bug; we fail loudly here rather than silently stamp `false` and mask the regression.
local components = ctx_table.components
local component_def = components and components[component_name] or nil
if not component_def then
error("duffle.emit_invoke_begin: component " .. string.format("%q", component_name)
.. " is present in `component_index` (the walker matched a `mac_" .. component_name .. "()` call) but absent from `components` (the canonical corpus.components registry). "
.. "This is a corpus-plumbing bug — the components pass must populate corpus.components[name] for every component it puts in corpus.component_body_index[name]. "
.. "The emission pass refuses to silently stamp `debug_skip = false` for a missing registry entry."
, 0
)
end
local debug_skip_stamp = component_def.debug_skip == true
local inv = {
id = next_inv_id,
parent_id = 0, -- patched below by caller
kind = inv_kind,
component_name = component_name,
call_text = call_text,
root_call_text = root_call_text,
call_path = call_path,
call_line = call_line,
def_path = nil, -- patched below after component lookup
def_line = nil,
-- 0-based emitted-word position. `word_idx` is the monotonic 0-based counter of `word` items emitted so far in this walk —
-- BEFORE this invocation's first word is emitted, it equals the position of the first word inside the invocation.
-- `start_word` (1-based items index of `invoke_begin`) is kept for items-walking consumers (Annotation pass bounds checks),
-- but DWARF / provenance rows MUST read `start_pos` because those rows are 1-based over the dense `word_events` stream (which has no `invoke_begin` items).
start_pos = word_idx,
start_word = #items + 1, -- 1-based items index of invoke_begin
end_pos = nil, -- patched by emit_invoke_end
end_word = nil, -- patched by emit_invoke_end
word_count = 0,
debug_skip = debug_skip_stamp,
errors = {},
}
invocations[#invocations + 1] = inv
items [#items + 1] = {
kind = "invoke_begin",
invocation_id = inv.id,
word_index = word_idx,
invocation_ids = open_invocation_ids_snapshot(),
}
invocation_stack[#invocation_stack + 1] = inv
return inv
end
local function emit_invoke_end(inv)
-- 0-based emitted-word position of the LAST word inside this invocation.
-- After the last body word was emitted, `word_idx` was incremented past it, so `word_idx - 1` is the 0-based position of the last word.
inv.end_pos = word_idx - 1
inv.end_word = #items + 1 -- 1-based items index of invoke_end
items[#items + 1] = {
kind = "invoke_end",
invocation_id = inv.id,
word_index = word_idx,
invocation_ids = open_invocation_ids_snapshot(),
}
for i = #invocation_stack, 1, -1 do
if invocation_stack[i] == inv then
table.remove(invocation_stack, i)
break
end
end
end
-- Resolve the per-token word count. If unresolved, surface ONE warning
-- and fall back to 1 opaque word so the cycle budget still accounts for the slot.
local function resolve_count(ident, tok_line)
local wc = ctx_table.word_counts
if wc and wc[ident] then return wc[ident] end
local canon = M.gte_canon(ident)
if canon ~= ident and wc and wc[canon] then return wc[canon] end
warnings[#warnings + 1] = {
kind = "uncounted",
line = tok_line,
msg = string.format("project_emission: opaque word emitted for %q (no entry in word_counts or component_index)",
ident),
}
return 1
end
-- Recursive walker: walk one body entry, possibly descending into components.
-- walk_parent_inv_id: Invocation ID of the enclosing call (0 for the root call).
-- walk_root_call_text: Outermost `mac_X(...)` token text (preserved across recursion).
-- walk_immediate_call_text: IMMEDIATE outer `mac_X(...)` token text for words emitted in this body — nil for the root atom body.
-- Two trackers are propagated as separate parameters so words deep inside nested expansions correctly identify both their immediate call site and the outermost call site.
local function walk_body_entry(body_entry, walk_parent_inv_id,
walk_root_call_text, walk_immediate_call_text)
local tokens = body_entry.body_tokens or {}
local body_off = body_entry.body_off or 0
local line_of = body_entry.line_of or M.LineIndex("")
local def_source = body_entry.source or ""
local def_line = body_entry.declaration or 0
local sub_map = body_entry.sub_map
-- Per-token dispatch: each matched branch returns; only the fall-through
-- "opaque word" emit handles direct encoders + mac_X-without-component.
local function process_token(bt)
local tok = M.trim(bt.tok or "")
if tok == "" then return end
local ident, after = M.read_ident(tok, 1)
if not ident then ident = "?" end
local _, args = token_ident_and_args(tok)
local tok_line = line_of(body_off + bt.rel) or 0
if M.DELAY_MARKERS[ident] then
emit_marker("delay", ident, nil, tok_line)
local rest = M.trim(tok:sub(after or (#tok + 1)))
if rest ~= "" then
process_token({ tok = rest, rel = bt.rel })
end
return
end
-- embedded markers live only in non-marker tokens.
-- Pass `ident` as the consuming instruction so `emit_embedded_markers` can compute each marker's arg position + record the consuming_encoder for the offsets pass.
-- Canonicalize `jump_rel` to `branch_equal` (its preprocessor-expanded form) so the `consuming_encoder` metadata in marker records is canonical.
-- `jump_rel`: unconditional jump alias from `code/duffle/mips.h`.
local consuming_encoder_for_markers = (ident == "jump_rel") and "branch_equal" or ident
if ident ~= "atom_label" and ident ~= "atom_offset" then
emit_embedded_markers(tok, tok_line, consuming_encoder_for_markers)
end
-- atom_label / atom_offset: terminal markers, no further descent.
-- Top-level markers (the marker IS the entire token) have no consuming instruction;
-- nil for both `consuming_encoder` and `consuming_arg_pos`.
-- The offsets pass treats these as branch-equivalent for backward compatibility.
-- TODO(Ed): Review this don't want legacy cruft here..
if ident == "atom_label" then emit_marker("label", args[1] or "", nil, tok_line); return
elseif ident == "atom_offset" then emit_marker("offset", args[1] or "", args[2] or "", tok_line); return
end
if ident:sub(1, 4) == "mac_" then
local bare = ident:sub(5)
local comp = ctx_table.component_index[bare]
if comp then
local invocation_root_call_text = walk_root_call_text or tok
if ctx_table.visiting[bare] then
-- Cycle: still allocate inv_id, emit zero-width begin/end, record the cycle error; do NOT recurse.
local inv = emit_invoke_begin(comp.kind or "comp_bare", bare, tok, invocation_root_call_text, def_source, tok_line)
inv.parent_id = walk_parent_inv_id
inv.call_text = tok
local err = {
kind = "cycle",
msg = string.format("project_emission: component cycle detected: %q", bare),
source = def_source,
line = tok_line,
}
inv.errors[#inv.errors + 1] = err
errors [#errors + 1] = err
emit_invoke_end(inv)
return
end
-- First visit: descend + count + count_mismatch-check below.
ctx_table.visiting[bare] = true
local inv = emit_invoke_begin(comp.kind or "comp_bare", bare, tok, invocation_root_call_text, def_source, tok_line)
inv.parent_id = walk_parent_inv_id
inv.call_text = tok
inv.def_path = comp.source
inv.def_line = comp.declaration
-- Propagate trackers into the recursive walk:
-- immediate_call_text = this call's tok (the IMMEDIATE outer call for words emitted in this body)
-- root_call_text = the OUTERMOST call (immutable across the recursion)
local formal_names = ctx_table.component_index[bare]
and ctx_table.component_index[bare].arg_names
local child_map = nil
if formal_names then
child_map = {}
for i, fname in ipairs(formal_names) do
child_map[fname] = apply_sub(sub_map, args[i])
end
end
walk_body_entry({
body_tokens = comp.body_tokens or {},
body_off = comp.body_off or 0,
line_of = comp.line_of,
source = comp.source,
declaration = comp.declaration,
sub_map = child_map,
},
inv.id,
invocation_root_call_text,
tok)
ctx_table.visiting[bare] = nil
emit_invoke_end(inv)
-- Count `word` items inside [start_word, end_word].
local wc_inside = 0
for i = inv.start_word, inv.end_word do
local it = items[i]
if it and it.kind == "word" then
wc_inside = wc_inside + 1
end
end
inv.word_count = wc_inside
-- count_mismatch is a construction error: word_counts["mac_X"] is the declared count populated by the components pass;
-- We compare against the measured word count.
local declared = ctx_table.word_counts["mac_" .. bare]
if declared and wc_inside ~= declared then
local err = {
kind = "count_mismatch",
msg = string.format("project_emission: mac_%s declared=%d measured=%d", bare, declared, wc_inside),
source = def_source,
line = tok_line,
}
inv.errors[#inv.errors + 1] = err
errors [#errors + 1] = err
end
return
end
-- mac_X NOT in component_index: fall through to opaque emit.
end
-- Direct encoder, or mac_X-without-component: resolve count + emit n words.
-- Resolve_count may emit a warning if the count is unresolved.
local n = resolve_count(ident, tok_line)
local out_ident = (ident == "nop2") and "nop" or ident
for _ = 1, n do
emit_word(out_ident, args, tok_line, tok, def_source, def_line, walk_immediate_call_text, walk_root_call_text, sub_map)
end
end
for _, bt in ipairs(tokens) do
process_token(bt)
end
end
-- Initialize the per-walk mutable context.
-- `visiting` is the active DFS component stack; `root_call_path` / `root_call_line` are preserved across recursion so nested words always point at the
-- ORIGINAL root atom call site.
ctx_table.visiting = ctx_table.visiting or {}
ctx_table.root_call_path = ctx_table.root_call_path or ""
ctx_table.root_call_line = ctx_table.root_call_line or 0
-- Walk first; the pass caller stamps the root call site for direct words after the projection returns.
-- For nested words the def_path / def_line already point at the component source and MUST be preserved (the stamping helper checks for that).
walk_body_entry(root_body_entry, 0, nil, nil)
-- Boundary check: every invoke_begin must have a matching invoke_end.
-- If anything is still open, surface a hard error.
if #invocation_stack > 0 then
errors[#errors + 1] = {
kind = "unbalanced",
msg = string.format("project_emission: invocation boundaries not balanced (%d unclosed invocation(s) at end of walk)", #invocation_stack),
}
end
return {
items = items,
word_events = word_events,
markers = markers,
invocations = invocations,
errors = errors,
warnings = warnings,
}
end
--- Project a body string into the per-atom emission projection.
---
--- Semantics:
--- * Direct one-word tokens (`nop`, `add_ui`, ...): one `word` item, encoder = ident, word_count = 1.
--- * Metadata-backed N-word tokens (`nop2`, `mask_upper`, ...): N `word` items, all sharing the same encoder + word_count = 1.
--- `nop2` is normalized to encoder `nop` (per the spec).
--- * `atom_label(F)` markers: one `label` item with `name = "F"`, `word_index = current word_idx`; zero-width (does NOT advance word_idx).
--- * `atom_offset(B, T)` markers: one `offset` item with `name = "B"`, `target = "T"`, `word_index = current word_idx`; zero-width.
--- * Delay markers (`GteDelay_` / `LdSlot_` / `BdSlot_` / `DmaSlot_`): one `delay` item; zero-width. The following encoder is the next token.
--- * `mac_X(...)` calls: emit `invoke_begin` (zero-width), recurse into the component body, emit `invoke_end` (zero-width).
--- The component body's words land between the begin/end pair; one invocation record is allocated per call (monotonic ID per atom).
--- * Unknown uncounted macros emit 1 opaque word + one warning per occurrence.
--- * Tokens whose count cannot be resolved (e.g. `mac_unknown` not in word_counts and not in component_index) surface one
--- warning; cycle + count-mismatch + boundary violations are construction errors on `pass.errors`.
---
--- Every emitted `word` carries: `i` (0-based word index), `encoder`, `args` (top-level args), `def_path`, `def_line`,
--- `call_text` (the immediate token spelling), `root_call_text` (outermost `mac_X(...)` text), `word_count` (always 1),
--- `invocation_ids` (innermost last), `outermost_invocation_id`.
--- Markers carry: `kind`, `name`, `line`, `word_index`, `target` (only for offset kind), plus `invocation_ids` / `outermost_invocation_id`
--- for the open invocation stack at that word.
---
--- @param body_text string -- the raw atom body string
--- @param component_index table -- bare-name → component record (corpus.component_body_index)
--- @param word_counts table -- macro name → emitted word count
--- @param components table -- bare-name → component definition (corpus.components); REQUIRED — consumed at the invocation-construction site to stamp
--- `invocation.debug_skip`. A missing or non-table `components` raises a fail-loud error rather than silently falling back.
--- @return EmissionProjection
function M.project_emission(body_text, component_index, word_counts, components, reg_use_ctx)
-- The recursive walk delegates to `_project_emission_inner` so component bodies (which arrive as
-- `{body_tokens, body_off, line_of, source, declaration}` records from `corpus.component_body_index`)
-- re-enter the same walker with the same shared output state.
--
-- The walker is body-relative: it builds `line_of` from `body_text` and stamps body-relative line numbers (1..N)
-- into `item.line` and `invocation.call_line`. `passes/emission_model.lua::stamp_root_provenance` performs the single
-- conversion from body-relative to physical source line at the close site, using the source's `line_of` closure that
-- the pass forwarded. One owner of the line state.
if type(components) ~= "table" then
error("duffle.project_emission: `components` is required "
.. "(bare-name -> component definition, e.g. corpus.components); "
.. "got " .. type(components) .. ". "
.. "The emission pass MUST forward the corpus registry "
.. "so the invocation-construction site can stamp `debug_skip` "
.. "without a second pass, source parse, or parallel lookup.",
0)
end
if type(body_text) ~= "string" or body_text == "" then
-- Empty body: still return a valid (empty) projection.
return {
items = {},
word_events = {},
markers = {},
invocations = {},
errors = {},
warnings = {},
}
end
local tokens = M.tokenize_body(body_text)
return _project_emission_inner({
body_tokens = tokens,
body_off = 0,
line_of = M.LineIndex(body_text),
source = "",
declaration = 0,
},
{
component_index = component_index or {},
word_counts = word_counts or {},
components = components,
reg_use_schema = reg_use_ctx and reg_use_ctx.reg_use_schema,
reg_use_param = reg_use_ctx and reg_use_ctx.reg_use_param,
atom_name = reg_use_ctx and reg_use_ctx.atom_name,
schema_name = reg_use_ctx and reg_use_ctx.schema_name,
})
end
-------------------------------------------------------------------------------
-- find_function_decl_for — backward walk for MipsAtomComp_Proc_ name extraction.
--
-- After the `sym` arg was dropped from MipsAtomComp_Proc_, the component name is derived from the preceding
-- `FI_ Slice_MipsCode ac_X(args)` function declaration. This function walks backward from `before_pos` to find it.
--
-- Returns (raw_name, args_inner) or (nil, nil).
-- raw_name — e.g. "ac_load_word_imm"
-- args_inner — e.g. "AtomBuilder_R ab, Reg dst, U4 imm"
--
-- The walk finds the LAST "Slice_MipsCode" before before_pos, then skips whitespace + qualifiers
-- (FI_, atom_dbg_skip, comments) until it finds an ident followed by "(".
-- That ident is the function name; the parens contents are the args.
-------------------------------------------------------------------------------
function M.find_function_decl_for(source, before_pos, slice_mips_code_len)
local search_pos = 1
local last_match = nil
while true do
local found = source:find("Slice_MipsCode", search_pos, true)
if not found or found >= before_pos then break end
last_match = found
search_pos = found + slice_mips_code_len
end
if not last_match then return nil, nil end
local pos = last_match + slice_mips_code_len
while pos < before_pos do
-- skip whitespace
while pos <= #source do
local c = source:sub(pos, pos)
if c == " " or c == "\t" or c == "\n" or c == "\r" then
pos = pos + 1
else
break
end
end
if pos > #source then break end
-- skip line comments
if source:sub(pos, pos + 1) == "//" then
while pos <= #source and source:sub(pos, pos) ~= "\n" do pos = pos + 1 end
pos = pos + 1
goto continue
end
-- skip block comments
if source:sub(pos, pos + 1) == "/*" then
local close = source:find("*/", pos + 2, true)
if not close then break end
pos = close + 2
goto continue
end
-- try to read an ident
local ident, ident_end = M.read_ident(source, pos)
if not ident then break end
-- check if the next non-ws char after ident is "("
local next_pos = M.skip_ws_and_cmt(source, ident_end)
if source:sub(next_pos, next_pos) == "(" then
local inner = M.read_parens(source, next_pos)
if inner then
return ident, inner
end
end
-- ident not followed by "(" — it's a qualifier (FI_, atom_dbg_skip, etc); skip it
pos = ident_end
::continue::
end
return nil, nil
end
-------------------------------------------------------------------------------
-- find_atom_proc_decl_for — backward walk for MipsAtom_Proc_ name extraction.
--
-- The atom name is the preceding `MipsAtom* ident(args)` function ident.
-- This function walks backward from `before_pos` to find it.
--
-- Returns (raw_name, args_inner, func_ident, after_paren) or (nil, nil).
-- raw_name — the function ident as written
-- args_inner — e.g. "AtomArena_R aa, U4 r_scratch, ..."
-- after_paren — source position after the function `)`
--
-- The walk finds the LAST "MipsAtom*" before before_pos, then skips whitespace + qualifiers (internal, I_, FI_, comments)
-- until it finds an ident followed by "(".
-------------------------------------------------------------------------------
function M.find_atom_proc_decl_for(source, before_pos, mips_atom_ptr_len)
local search_pos = 1
local last_match = nil
while true do
-- plain=true: "*" is literal, no escaping needed
local found = source:find("MipsAtom*", search_pos, true)
if not found or found >= before_pos then break end
last_match = found
search_pos = found + mips_atom_ptr_len
end
if not last_match then return nil, nil end
local pos = last_match + mips_atom_ptr_len
while pos < before_pos do
-- skip whitespace
while pos <= #source do
local c = source:sub(pos, pos)
if c == " " or c == "\t" or c == "\n" or c == "\r" then
pos = pos + 1
else
break
end
end
if pos > #source then break end
-- skip line comments
if source:sub(pos, pos + 1) == "//" then
while pos <= #source and source:sub(pos, pos) ~= "\n" do pos = pos + 1 end
pos = pos + 1
goto continue
end
-- skip block comments
if source:sub(pos, pos + 1) == "/*" then
local close = source:find("*/", pos + 2, true)
if not close then break end
pos = close + 2
goto continue
end
-- try to read an ident
local ident, ident_end = M.read_ident(source, pos)
if not ident then break end
-- check if the next non-ws char after ident is "("
local next_pos = M.skip_ws_and_cmt(source, ident_end)
if source:sub(next_pos, next_pos) == "(" then
local inner, after_paren = M.read_parens(source, next_pos)
if inner then
return ident, inner, ident, after_paren
end
end
-- ident not followed by "(" — it's a qualifier; skip it
pos = ident_end
::continue::
end
return nil, nil
end
return M
+725
View File
@@ -0,0 +1,725 @@
--- duffle_isa.lua — encoder / GTE / hardware tables.
local M = {}
-- Section 7: domain tables
-- ════════════════════════════════════════════════════════════════════════════
-- atom_info sub-calls: atom_bind, atom_reads, atom_writes, atom_view, atom_reg_types, atom_ctx, atom_phase.
M.TAPE_ATOM_MACROS = {
["atom_info"] = { kind = "info", binds = false },
}
-- Empty C macros that prefix the next encoder. Zero words.
-- BdSlot_ nop is one nop word. The marker is not the BD instruction.
M.DELAY_MARKERS = {
["GteDelay_"] = true,
["LdSlot_"] = true,
["BdSlot_"] = true,
["DmaSlot_"] = true,
}
-- One row per encoder. Old table names are load-time views (build_isa_views).
M.INSTRUCTION = {
["BdSlot_"] = { cycles = 0, kind = "marker", },
["LdSlot_"] = { cycles = 0, kind = "marker", },
["add_s"] = { cycles = 1, kind = "alu", },
["add_si"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, },}, },
["add_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
["add_u_self"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, value = { dest = 1, op = "add_u", sources = { 1, 2 }, }, },
["add_ui"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, value = { dest = 1, immediate = 3, op = "add_ui", source = 2, }, },
["add_ui_self"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = { 1 }, imm = { { arg = 2, signed = true, width = 16, }, }, value = { dest = 1, immediate = 2, op = "add_ui", source = 1, }, },
["and"] = { cycles = 1, kind = "alu", },
["and_i"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, width = 16, }, }, value = { dest = 1, immediate = 3, op = "and_i", source = 2, }, },
["and_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
["atom_bind"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
["atom_info"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
["atom_label"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
["atom_offset"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
["atom_reads"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
["atom_writes"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
["branch_equal"] = { cycles = 2, kind = "branch", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
["branch_ge_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
["branch_gt_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
["branch_le_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
["branch_lt_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
["branch_ne"] = { cycles = 2, kind = "branch", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
["call_addr"] = { cycles = 2, kind = "call", reads = {}, writes = { 1 }, },
["call_reg"] = { cycles = 2, kind = "call", reads = { 1 }, writes = { 2 }, },
["div_s"] = { cycles = 35, kind = "alu", reads = { 1, 2 }, writes = {}, },
["div_u"] = { cycles = 35, kind = "alu", reads = { 1, 2 }, writes = {}, },
["gte_load_v0"] = { cycles = 2, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
["gte_load_v0v1v2"] = { cycles = 6, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
["gte_load_v1"] = { cycles = 2, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
["gte_load_v2"] = { cycles = 2, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
["gte_lw"] = { cycles = 1, kind = "load", reads = { 2 }, writes = {}, },
["gte_lwc2"] = { cycles = 1, kind = "load", },
["gte_mv_from_ctrl_r"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = { 1 }, },
["gte_mv_from_data_r"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = { 1 }, },
["gte_mv_to_ctrl_r"] = { cycles = 1, kind = "cop2_xfer", reads = { 1 }, writes = {}, },
["gte_mv_to_data_r"] = { cycles = 1, kind = "cop2_xfer", reads = { 1 }, writes = {}, },
["gte_stotz"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = {}, },
["gte_stsxy3"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = {}, },
["gte_sw"] = { cycles = 1, kind = "store", reads = { 2 }, writes = {}, },
["gte_swc2"] = { cycles = 1, kind = "store", },
["jump"] = { cycles = 2, kind = "jump", reads = {}, writes = {}, },
["jump_link"] = { cycles = 2, kind = "call", reads = { 1 }, writes = { 2 }, },
["jump_reg"] = { cycles = 2, kind = "jump", reads = { 1 }, writes = {}, suppress_arg1 = { R_AtomJmp = "fixed mac_yield handshake", }, },
["jump_rel"] = { cycles = 2, kind = "branch", delay_slot = true, },
["li_s"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, value = { dest = 1, immediate = 3, op = "add_ui", source = 2, }, },
["load_byte"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
["load_byte_u"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
["load_half"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
["load_half_u"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
["load_imm"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, },
["load_ui"] = { cycles = 1, kind = "alu", reads = {}, writes = { 1 }, },
["load_upper_i"] = { cycles = 1, kind = "alu", reads = {}, writes = { 1 }, imm = { { arg = 2, width = 16, }, }, value = { dest = 1, immediate = 2, op = "load_upper_i", }, },
["load_word"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
["mac_yield"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
["mask_upper"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, },
["mov_from_high"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, },
["mov_from_low"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, },
["mov_to_high"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = {}, },
["mov_to_low"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = {}, },
["mult_s"] = { cycles = 12, kind = "alu", reads = { 1, 2 }, writes = {}, },
["mult_u"] = { cycles = 12, kind = "alu", reads = { 1, 2 }, writes = {}, },
["nop"] = { cycles = 1, kind = "nop", reads = {}, writes = {}, },
["nop2"] = { cycles = 2, kind = "nop", reads = {}, writes = {}, },
["nor_u"] = { cycles = 1, kind = "alu", },
["or_i"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, width = 16, }, }, value = { dest = 1, immediate = 3, op = "or_i", source = 2, }, },
["or_i_self"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = { 1 }, imm = { { arg = 2, width = 16, }, }, value = { dest = 1, immediate = 2, op = "or_i", source = 1, }, },
["or_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
["or_u_self"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, value = { dest = 1, op = "or", sources = { 1, 2 }, }, },
["set_lt_s"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
["set_lt_si"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, },
["set_lt_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
["set_lt_ui"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, },
["shift_aright"] = { cycles = 1, kind = "alu", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
["shift_aright_var"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
["shift_lleft"] = { cycles = 1, kind = "alu", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
["shift_lleft_self"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = { 1 }, imm = { { arg = 2, width = 5, }, }, value = { dest = 1, immediate = 2, op = "shift_lleft", source = 1, }, },
["shift_lleft_var"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
["shift_lright"] = { cycles = 1, kind = "alu", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
["slt_s"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
["slt_si"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
["slt_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
["slt_ui"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16,}, }, },
["store_byte"] = { cycles = 1, kind = "store", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
["store_half"] = { cycles = 1, kind = "store", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
["store_word"] = { cycles = 1, kind = "store", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
["sub_s"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
["sub_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
["sys_mov_from_cop0"] = { cycles = 1, kind = "cop0_xfer", reads = {}, writes = { 1 }, },
["sys_mov_to_cop0"] = { cycles = 1, kind = "cop0_xfer", reads = { 1 }, writes = {}, },
["xor_i"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, width = 16, }, }, value = { dest = 1, immediate = 3, op = "xor_i", source = 2, }, },
["xor_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
}
-- One row per GTE command. Alias cycle numbers live here, not on INSTRUCTION.
M.GTE_COMMAND = {
["gte_cmdw_avsz3"] = {
aliases = { "gte_avg_sort_z3", "gte_avsz3", "gte_cmdw_avg_sort_z3" },
cycles = 5,
inputs = { "C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3", "gte_cr_ZSF3" },
outputs = {
{ register = "C2_OTZ", role = "otz", },
},
latch = {
{ register = "C2_OTZ", required = 4, },
},
},
["gte_cmdw_avsz4"] = {
aliases = { "gte_avg_sort_z4", "gte_avsz4", "gte_cmdw_avg_sort_z4" },
cycles = 6,
inputs = { "C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3", "gte_cr_ZSF4" },
outputs = {
{ register = "C2_OTZ", role = "otz", },
},
latch = {
{ register = "C2_OTZ", required = 4, },
},
},
["gte_cmdw_gpf"] = {
aliases = {},
cycles = 5,
inputs = { "C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3" },
outputs = {
{ register = "C2_MAC1", role = "mac_result", },
{ register = "C2_MAC2", role = "mac_result", },
{ register = "C2_MAC3", role = "mac_result", },
{ register = "C2_IR1", role = "latest_color", },
{ register = "C2_IR2", role = "latest_color", },
{ register = "C2_IR3", role = "latest_color", },
},
latch = {
{ register = "C2_MAC1", required = 4, },
{ register = "C2_MAC2", required = 4, },
{ register = "C2_MAC3", required = 4, },
{ register = "C2_IR1", required = 4, },
{ register = "C2_IR2", required = 4, },
{ register = "C2_IR3", required = 4, },
},
},
["gte_cmdw_mvmva"] = {
aliases = {},
cycles = 8,
inputs = {
"C2_VXY0", "C2_VZ0",
"C2_VXY1", "C2_VZ1",
"C2_VXY2", "C2_VZ2",
"C2_IR1", "C2_IR2", "C2_IR3",
"gte_cr_RT11", "gte_cr_RT12", "gte_cr_RT13",
"gte_cr_RT21", "gte_cr_RT22", "gte_cr_RT23",
"gte_cr_RT31", "gte_cr_RT32", "gte_cr_RT33",
"gte_cr_TRX", "gte_cr_TRY", "gte_cr_TRZ"
},
outputs = {
{ register = "C2_IR1", role = "latest_color", },
{ register = "C2_IR2", role = "latest_color", },
{ register = "C2_IR3", role = "latest_color", },
},
latch = {
{ register = "C2_IR1", required = 4, },
{ register = "C2_IR2", required = 4, },
{ register = "C2_IR3", required = 4, },
},
},
["gte_cmdw_nclip"] = {
aliases = { "gte_nclip" },
cycles = 8,
inputs = { "C2_SXY0", "C2_SXY1", "C2_SXY2" },
outputs = {
{ register = "C2_SZ3", role = "mac_result", },
},
latch = {
{ register = "C2_SZ3", required = 4, },
},
},
["gte_cmdw_op"] = {
aliases = { "gte_cmdw_outer_product", "gte_cmdw_wedge" },
cycles = 6,
inputs = {},
outputs = {
{ register = "C2_IR1", role = "latest_color", },
{ register = "C2_IR2", role = "latest_color", },
{ register = "C2_IR3", role = "latest_color", },
},
latch = {
{ register = "C2_IR1", required = 4, },
{ register = "C2_IR2", required = 4, },
{ register = "C2_IR3", required = 4, },
},
},
["gte_cmdw_rtps"] = {
aliases = { "gte_cmdw_rotate_translate_perspective_single", "gte_rtps" },
cycles = 15,
inputs = {
"C2_VXY0", "C2_VZ0",
"C2_VXY1", "C2_VZ1",
"C2_VXY2", "C2_VZ2",
"C2_RGB", "C2_OTZ",
"C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3",
"C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3",
"gte_cr_RT11", "gte_cr_RT12", "gte_cr_RT13",
"gte_cr_RT21", "gte_cr_RT22", "gte_cr_RT23",
"gte_cr_RT31", "gte_cr_RT32", "gte_cr_RT33",
"gte_cr_TRX", "gte_cr_TRY", "gte_cr_TRZ",
"gte_cr_OFX", "gte_cr_OFY",
"gte_cr_H",
"gte_cr_DQA", "gte_cr_DQB"
},
outputs = {
{ register = "C2_SXY2", role = "latest_screen_xy", },
{ register = "C2_SZ2", role = "latest_screen_z", },
{ register = "C2_OTZ", role = "otz", },
{ register = "C2_IR0", role = "latest_color", },
},
latch = {
{ register = "C2_SXY2", required = 4, },
{ register = "C2_SZ2", required = 4, },
{ register = "C2_OTZ", required = 4, },
{ register = "C2_IR0", required = 4, },
},
},
["gte_cmdw_rtpt"] = {
aliases = { "gte_cmdw_rotate_translate_perspective_triple", "gte_rtpt" },
cycles = 23,
inputs = {
"C2_VXY0", "C2_VZ0",
"C2_VXY1", "C2_VZ1",
"C2_VXY2", "C2_VZ2",
"C2_RGB", "C2_OTZ",
"C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3",
"C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3",
"gte_cr_RT11", "gte_cr_RT12", "gte_cr_RT13",
"gte_cr_RT21", "gte_cr_RT22", "gte_cr_RT23",
"gte_cr_RT31", "gte_cr_RT32", "gte_cr_RT33",
"gte_cr_TRX", "gte_cr_TRY", "gte_cr_TRZ",
"gte_cr_OFX", "gte_cr_OFY",
"gte_cr_H",
"gte_cr_DQA", "gte_cr_DQB"
},
outputs = {
{ register = "C2_SXY0", role = "screen_xy[0]", },
{ register = "C2_SXY1", role = "screen_xy[1]", },
{ register = "C2_SXY2", role = "latest_screen_xy", },
{ register = "C2_SZ3", role = "latest_screen_z", },
{ register = "C2_OTZ", role = "otz", },
},
latch = {
{ register = "C2_SXY0", required = 4, },
{ register = "C2_SXY1", required = 4, },
{ register = "C2_SXY2", required = 4, },
{ register = "C2_SZ3", required = 4, },
{ register = "C2_OTZ", required = 4, },
},
},
["gte_cmdw_sqr"] = {
aliases = {},
cycles = 5,
inputs = { "C2_IR1", "C2_IR2", "C2_IR3" },
outputs = {
{ register = "C2_MAC1", role = "mac_result", },
{ register = "C2_MAC2", role = "mac_result", },
{ register = "C2_MAC3", role = "mac_result", },
{ register = "C2_IR1", role = "latest_color", },
{ register = "C2_IR2", role = "latest_color", },
{ register = "C2_IR3", role = "latest_color", },
},
latch = {
{ register = "C2_MAC1", required = 4, },
{ register = "C2_MAC2", required = 4, },
{ register = "C2_MAC3", required = 4, },
{ register = "C2_IR1", required = 4, },
{ register = "C2_IR2", required = 4, },
{ register = "C2_IR3", required = 4, },
},
},
}
function M.instr (ident) return M.INSTRUCTION [ident] end
function M.gte_canon(ident) return M.ALIAS_TO_CANONICAL [ident] or ident end
function M.gte (ident) return M.GTE_COMMAND[M.gte_canon(ident)] end
local function build_isa_views()
M.ALIAS_TO_CANONICAL = {}
for canon, row in pairs(M.GTE_COMMAND) do
M.ALIAS_TO_CANONICAL[canon] = canon
for _, alias in ipairs(row.aliases or {}) do
M.ALIAS_TO_CANONICAL[alias] = canon
end
end
M.INSTRUCTION_LATENCY = {}
M.INSTRUCTION_GPR_EFFECTS = {}
M.IMMEDIATE_FIELD_WIDTHS = {}
M.GPR_VALUE_RULES = {}
M.CONTROL_TRANSFER_DELAY_SLOT_POLICIES = {}
for name, row in pairs(M.INSTRUCTION) do
M.INSTRUCTION_LATENCY[name] = row.cycles
if row.reads or row.writes then
M.INSTRUCTION_GPR_EFFECTS[name] = {
reads = row.reads or {},
writes = row.writes or {},
}
end
if row.imm then M.IMMEDIATE_FIELD_WIDTHS[name] = row.imm end
if row.value then M.GPR_VALUE_RULES [name] = row.value end
if (row.kind == "branch" or row.kind == "jump" or row.kind == "call")
and row.delay_slot ~= false then
M.CONTROL_TRANSFER_DELAY_SLOT_POLICIES[name] = {
family = row.kind,
suppress_arg1 = row.suppress_arg1,
}
end
end
M.GTE_COMMAND_ALIASES = {}
M.GTE_COMMAND_INPUTS = {}
M.GTE_COMMAND_OUTPUTS = {}
M.GTE_COMMAND_LATCH_WINDOWS = {}
for canon, row in pairs(M.GTE_COMMAND) do
M.GTE_COMMAND_ALIASES [canon] = canon
M.INSTRUCTION_LATENCY [canon] = row.cycles
M.INSTRUCTION_GPR_EFFECTS[canon] = { reads = {}, writes = {} }
for _, alias in ipairs(row.aliases or {}) do
M.GTE_COMMAND_ALIASES [alias] = canon
M.INSTRUCTION_LATENCY [alias] = row.cycles
M.INSTRUCTION_GPR_EFFECTS[alias] = { reads = {}, writes = {} }
end
M.GTE_COMMAND_INPUTS [canon] = row.inputs
M.GTE_COMMAND_OUTPUTS [canon] = row.outputs
M.GTE_COMMAND_LATCH_WINDOWS[canon] = row.latch
end
end
build_isa_views()
--- GTE control-register alias groups.
--- Aliases within a group write to the same C2 control-register slot (the HW double-maps some C2 slots across multiple PSX SDK / libgte conventions).
--- Aliases across groups write to distinct C2 slots.
---
--- Cross-alias writes inside one atom body, or across the wave-context boundary, silently clobber each other.
--- The `check_gte_cr_alias_writes` check warns about each pair per source. See `docs/gte_reference.md` §"Control-register alias table"
--- for the HW rationale and the libgte outer-product convention.
M.GTE_CR_ALIAS_GROUPS = {
{ 24, { "gte_cr_RBK", "gte_cr_OFX" } }, -- background R vs screen offset X
{ 25, { "gte_cr_GBK", "gte_cr_OFY" } }, -- background G vs screen offset Y
{ 26, { "gte_cr_BBK", "gte_cr_H" } }, -- background B vs projection plane distance H
}
-- Packed RT slots named by the gte.h packed-slot comment. First must be written before second.
M.GTE_PACKED_SLOT_RELATIONS = {
{ slot = 2, first = "gte_cr_RT13", second = "gte_cr_RT22" },
}
-- Operand-class table for the COP2->GPR load-delay check.
-- Maps each emitting-token ident to the set of GPR operand positions it reads.
-- Covers the current encoder vocabulary (`code/duffle/mips.h` + `code/duffle/gte.h`); add rows here as new encoders land.
--
-- Semantics:
-- * A "GPR operand position" is the textual slot in the macro's argument list, 1-based; e.g. `load_word(rt, base, off)` has positional operands 1 (rt), 2 (base), 3 (off).
-- The table reads operands 1 + 2 + 3 to find what GPRs the macro touches.
-- * The check tracks one entry per destination GPR per MFC2 / CFC2 event.
-- A subsequent event counts as a "use" iff any of its read operand positions reference that destination GPR's ident (e.g. `R_T0`).
-- * Branch delay slots are out of scope (MIPS control-flow; tracked separately).
M.OPERAND_READ_POSITIONS = {
-- CPU ALU with one or two GPR operands. Reads every GPR operand.
["add_ui"] = {1, 2},
["li_s"] = {1, 2}, -- rt (write), imm16 (immediate)
["add_ui_self"] = {1},
["add_si"] = {1, 2},
["add_u"] = {1, 2, 3},
["add_u_self"] = {1, 2},
["sub_s"] = {1, 2, 3},
["sub_u"] = {1, 2, 3},
["and_i"] = {1, 2},
["and"] = {1, 2, 3},
["or_i"] = {1, 2},
["or_i_self"] = {1},
["or"] = {1, 2, 3},
["or_self"] = {1, 2},
["xor_i"] = {1, 2},
["xor"] = {1, 2, 3},
["slt_s"] = {1, 2, 3},
["slt_u"] = {1, 2, 3},
["slt_si"] = {1, 2},
["slt_ui"] = {1, 2},
["mult_s"] = {1, 2},
["mult_u"] = {1, 2},
["div_s"] = {1, 2},
["div_u"] = {1, 2},
-- Shifts: shift_lleft(rd, rt, shamt); the rt operand is the value, rd is dest.
["shift_lleft"] = {1, 2},
["shift_lright"] = {1, 2},
["shift_aright"] = {1, 2},
["shift_lleft_self"] = {1},
-- Loads: load_word(rt, base, off); the rt operand is the destination (it's written, not read) and base + off are non-GPR operands.
-- The check treats the rt operand as a write, so the read-positions table for `load_*` is empty.
["load_word"] = {},
["load_half_u"] = {},
["load_byte_u"] = {},
["load_half"] = {},
["load_byte"] = {},
["load_upper_i"] = {},
["load_ui"] = {},
-- Stores write to memory; base + rt operands are non-read for load-delay purposes.
["store_word"] = {},
["store_half"] = {},
["store_byte"] = {},
-- Branches read rs (+ rt for beq/bne). The branch delay slot is out of scope.
["branch_equal"] = {1, 2},
["branch_ne"] = {1, 2},
["branch_le_zero"] = {1},
["branch_lt_zero"] = {1},
["branch_ge_zero"] = {1},
["branch_gt_zero"] = {1},
-- Jumps / link: jr / jalr read rs only (the target). RD is the destination link.
["jump_reg"] = {1},
["jump_link"] = {1},
["call_reg"] = {1},
["call_addr"] = {},
["jump"] = {},
-- mask_upper is a 2-word macro: shift_lleft then shift_lright. The first reads rt.
["mask_upper"] = {1, 2},
-- move from/to HI/LO.
["mov_from_high"] = {},
["mov_from_low"] = {},
["mov_to_high"] = {1},
["mov_to_low"] = {1},
-- GTE transfers / loads / stores / commands: the relevant table values live in the check itself.
-- `gte_mv_to_*` writes its rt operand; `gte_mv_from_*` writes its rt operand; `gte_*` commands are atomic-from-the-CPU-POV
-- once they issue (the CPU holds until the command completes, so load-delay violations don't surface here).
["gte_mv_from_data_r"] = {},
["gte_mv_from_ctrl_r"] = {},
["gte_mv_to_data_r"] = {},
["gte_mv_to_ctrl_r"] = {},
["gte_lw"] = {},
["gte_sw"] = {},
["shift_lleft_var"] = {1, 2, 3}, -- rd, rt, rs (variable shift amount)
["shift_aright_var"] = {1, 2, 3},
}
-- GP0 packet sizes (total words including the 1-word tag) per GP0 cmd byte.
-- Per PSX-SPX `docs/psx-spx/docs/graphicsprocessingunitgpu.md` §"GPU Render Polygon Commands":
-- Each polygon command's word count = 1 (tag/cmd) + per-vertex (vertex + optional color + optional UV).
-- F3: cmd + 3 vertices = 4 words; +1 tag = 5
-- F4: cmd + 4 vertices = 5 words; +1 tag = 6
-- G3: cmd + 3×(color + vertex) = 6 words; +1 tag = 7
-- G4: cmd + 4×(color + vertex) = 8 words; +1 tag = 9
-- FT3: cmd + tpage + clut + 3×(vertex + UV) = 7 words; +1 tag = 8
-- FT4: cmd + tpage + clut + 4×(vertex + UV) = 9 words; +1 tag = 10
-- GT3: cmd + tpage + clut + 3×(color + vertex + UV) = 9 words; +1 tag = 10
-- GT4: cmd + tpage + clut + 4×(color + vertex + UV) = 12 words; +1 tag = 13
--
-- Cross-checked against code/duffle/gp.h struct sizes + the set_poly_* macros
-- (which encode "len" = "words after tag"):
-- set_poly_f3(p) -> set_len(p, 4) -> 5 total GP0 0x20
-- set_poly_ft3(p) -> set_len(p, 7) -> 8 total GP0 0x24
-- set_poly_f4(p) -> set_len(p, 5) -> 6 total GP0 0x28
-- set_poly_ft4(p) -> set_len(p, 9) -> 10 total GP0 0x2C
-- set_poly_g3(p) -> set_len(p, 6) -> 7 total GP0 0x30
-- set_poly_gt3(p) -> set_len(p, 9) -> 10 total GP0 0x34
-- set_poly_g4(p) -> set_len(p, 8) -> 9 total GP0 0x38
-- set_poly_gt4(p) -> set_len(p, 12) -> 13 total GP0 0x3C
M.GP0_CMD_SIZE = {
[0x20] = 5, -- Poly_F3
[0x24] = 8, -- Poly_FT3
[0x28] = 6, -- Poly_F4
[0x2C] = 10, -- Poly_FT4
[0x30] = 7, -- Poly_G3
[0x34] = 10, -- Poly_GT3
[0x38] = 9, -- Poly_G4
[0x3C] = 13, -- Poly_GT4
}
-- Shape suffix (after `ac_format_` / `mac_format_` prefix) -> GP0 cmd byte.
-- Lets the static-analysis check derive the cmd byte from a macro name like `mac_format_g4_color` -> `g4` -> 0x38 -> 9 expected words.
M.GP0_CMD_BY_SHAPE = {
["f3"] = 0x20, ["ft3"] = 0x24,
["f4"] = 0x28, ["ft4"] = 0x2C,
["g3"] = 0x30, ["gt3"] = 0x34,
["g4"] = 0x38, ["gt4"] = 0x3C,
}
M.UNKNOWN_INSTRUCTION_CYCLES = 1
-- Hardware-relation policy table.
--
-- The forward walker in `passes/static_analysis.lua::analyze_hardware_relations` reads every emitted word_event, matches its `encoder` against `row.token`, and:
-- * stages the event as a producer in `atom.paths.forward_state`; or
-- * matches it as a consumer against pending producers and records a hazard on `atom.paths.hazards` when the gap is below `visibility.required`.
--
-- Each row is the contract for one CPU-to-coprocessor transfer semantic (the coprocessor-to-CPU path mirrors the same shape).
-- The `reads` / `writes` sub-tables carry the argument positions the analyzer inspects:
-- * `writes.arg` is the destination operand (the producer's effect); the analyzer stages this register as a pending producer.
-- * `reads` (when present) lists the operand positions the same token reads back from hardware; for MTC2 / CTC2 the producer reads the GPR source it is loading from.
-- The `fanout_to` field (MTC2-IRGB row only) tells the consumer-match logic which downstream COP2 registers are transitively updated by the write.
--
-- Visibility semantics:
-- * `kind = "post_producer_words"` means the consumer observes the producer's effect after `required` independent emitted words that are
-- strictly between the producer and the consumer. The producer's own emitted slot is implicit (it counts as the slot of issue, not toward `required`)
-- per the PSX-SPX rule: "Store delays are counted in numbers of clock cycles (not in numbers of opcodes).
-- For 3 cycle delay, one must usually insert 3 cached opcodes (or one uncached opcode)."
-- * `required` is the minimum count of intervening emitted words between producer and consumer.
-- `required = 0` permits the consumer on the very next slot; `required < 0` would place the consumer on the same slot as the producer
-- and is reserved for future "self-retires" relations.
--
-- Evidence:
-- * `evidence.confidence` is one of `"exact"`, `"conservative"`, `"unknown"`. The severity comes from `violation_kind`;
-- A hardware measurement that the vendor caveats may still classify as `"conservative"` even when the underlying timing is numerically known.
-- * `evidence.source` is the upstream reference (file + line range) the row is sourced from. New rows must carry this citation.
--
-- Consumers:
-- * passes/static_analysis.lua::analyze_hardware_relations (forward walker).
-- * passes/static_analysis.lua::transfer_hazards CHECK_RULES reader (renders hazards onto `findings`).
-- This table is consumed by the hardware-relation analyzer and hazard renderer.
M.HARDWARE_RELATIONS = {
-- CPU → COP2 data register (MTC2). The ordinary default is 2 cached words between producer and consumer (cpuspecifications.md:407-419).
{
id = "mtc2_gpr_visibility",
semantic = "MTC2",
token = "gte_mv_to_data_r",
direction = "gpr_to_cop2_data",
reads = { domain = "gpr", arg = 1 },
writes = { domain = "cop2.data", arg = 2 },
visibility = { kind = "post_producer_words", required = 2 },
evidence = {
confidence = "exact",
source = "cpuspecifications.md:407-419",
},
violation_kind = "error",
},
-- CPU → COP2 data register when the destination is C2_IRGB (data 28).
-- C2_IRGB drives the IR1/IR2/IR3 color-conversion fan-out, which extends the propagation delay to 3 cached words.
-- `destination_match = "C2_IRGB"` is the row's filter; the analyzer consults this when the producer's destination operand equals "C2_IRGB".
-- C2_ORGB (data 29) is read-only and is never classified as a writable fan-out destination.
{
id = "mtc2_irgb_visibility",
semantic = "MTC2",
token = "gte_mv_to_data_r",
direction = "gpr_to_cop2_data",
reads = { domain = "gpr", arg = 1 },
writes = { domain = "cop2.data", arg = 2 },
destination_match = "C2_IRGB",
fanout_to = { "C2_IR1", "C2_IR2", "C2_IR3" },
visibility = { kind = "post_producer_words", required = 3 },
evidence = {
confidence = "exact",
source = "cpuspecifications.md:407-419",
},
violation_kind = "error",
},
-- CPU → COP2 control register (CTC2). Ordinary minimum 2;
-- no IRGB-style fan-out exists for control registers (per spec §3.6: only C2_IRGB has the 3-cycle fan-out on the data side).
{
id = "ctc2_gpr_visibility",
semantic = "CTC2",
token = "gte_mv_to_ctrl_r",
direction = "gpr_to_cop2_control",
reads = { domain = "gpr", arg = 1 },
writes = { domain = "cop2.ctrl", arg = 2 },
visibility = { kind = "post_producer_words", required = 2 },
evidence = {
confidence = "exact",
source = "cpuspecifications.md:407-419",
},
violation_kind = "error",
},
-- COP2 data → GPR (MFC2). One cached slot between the transfer and the first GPR consumer;
-- the GPR is not updated until the instruction AFTER the MFC2 completes (geometrytransformationenginegte.md:29-32).
{
id = "mfc2_gpr_visibility",
semantic = "MFC2",
token = "gte_mv_from_data_r",
direction = "cop2_data_to_gpr",
reads = { domain = "cop2.data", arg = 2 },
writes = { domain = "gpr", arg = 1 },
visibility = { kind = "post_producer_words", required = 1 },
evidence = {
confidence = "exact",
source = "geometrytransformationenginegte.md:29-32",
},
violation_kind = "error",
},
-- COP2 control → GPR (CFC2). Same delay as MFC2 (cpuspecifications.md treats the two load-from-COP2 paths symmetrically).
{
id = "cfc2_gpr_visibility",
semantic = "CFC2",
token = "gte_mv_from_ctrl_r",
direction = "cop2_control_to_gpr",
reads = { domain = "cop2.ctrl", arg = 2 },
writes = { domain = "gpr", arg = 1 },
visibility = { kind = "post_producer_words", required = 1 },
evidence = {
confidence = "exact",
source = "cpuspecifications.md:382-419",
},
violation_kind = "error",
},
-- COP0 control → GPR (MFC0).
-- One cached slot; the analyzer treats `sys_mov_from_cop0(rt, 12)` (the SR/CU2 transfer) as the same shape as the COP2 load-delay path.
-- The semantic-level SR/CU2 transition models the load delay;
-- SR.CU2 bounded-value propagation is modeled separately).
{
id = "mfc0_gpr_visibility",
semantic = "MFC0",
token = "sys_mov_from_cop0",
direction = "cop0_control_to_gpr",
reads = { domain = "cop0.ctrl", arg = 2 },
writes = { domain = "gpr", arg = 1 },
visibility = { kind = "post_producer_words", required = 1 },
evidence = {
confidence = "exact",
source = "cpuspecifications.md:171-178",
},
violation_kind = "error",
},
-- Memory -> COP2 data register (LWC2).
-- The memory-side timing is not measured by the vendored GTE latch experiment, so this relation has no numeric retirement threshold.
-- The LWC2 destination has TWO retirement regimes (per PSX-SPX):
-- * GTE-command consumer (`gte_cmdw_*`): the GTE pipeline LATCHES the LWC2 result, so a `gte_cmdw_*`
-- in the very next slot uses the latched value. Gap = 0 is allowed. (Per `docs/psx-spx/docs/gtepipelinetimings.md:271-274`.)
-- * Any other consumer: standard MIPS load delay applies. Gap = 1 required. (Per `docs/psx-spx/docs/cpuspecifications.md:407-419`.)
-- Two separate relations so the walker can dispatch by consumer type and emit different severities
-- (the GTE-command path is `info` because the latch is intentional; the non-GTE-consumer path is `error` because the missing nop is a real bug).
{
id = "lwc2_to_gte_command",
semantic = "LWC2_to_GTE",
token = "gte_lw",
direction = "memory_to_cop2_data",
reads = { domain = "memory", arg = 2 },
writes = { domain = "cop2.data", arg = 1 },
required = 0, -- GTE-command consumer: gap = 0 OK (latched).
evidence = {
confidence = "measured",
source = "gtepipelinetimings.md:271-274",
},
violation_kind = "info",
clear_on_consumer = true,
},
{
id = "lwc2_to_other_consumer",
semantic = "LWC2_to_other",
token = "gte_lw",
direction = "memory_to_cop2_data",
reads = { domain = "memory", arg = 2 },
writes = { domain = "cop2.data", arg = 1 },
required = 1, -- Non-GTE-consumer: standard MIPS load delay.
evidence = {
confidence = "inferred",
source = "cpuspecifications.md:407-419",
},
violation_kind = "error",
clear_on_consumer = true,
},
-- COP2 data register -> memory (SWC2). A read of C2 state, not a CPU-to-COP2 write.
-- The policy row stays in for direction/provenance; staging it as a later command-input producer is suppressed.
{
id = "swc2_memory_write",
semantic = "SWC2",
token = "gte_sw",
direction = "cop2_data_to_memory",
reads = { domain = "cop2.data", arg = 1 },
writes = { domain = "memory", arg = 2 },
visibility = { kind = "none", required = 0 },
evidence = {
confidence = "exact",
source = "cpuspecifications.md:79",
},
violation_kind = "info",
stage = false,
},
-- MTC0 Status/SR.CU2. The ordinary COP0 store has no general store-delay relation;
-- this row feeds the dedicated CU2 transition logic in the same forward walk and is therefore not staged in `pending`.
{
id = "mtc0_cu2_visibility",
semantic = "MTC0",
token = "sys_mov_to_cop0",
direction = "gpr_to_cop0_status",
reads = { domain = "gpr", arg = 1 },
writes = { domain = "cop0.status", arg = 2 },
status_register = 12,
visibility = { kind = "post_producer_words", required = 2 },
evidence = {
confidence = "conservative",
source = "cpuspecifications.md:543,625-628",
},
violation_kind = "warning",
stage = false,
cu2_transition = true,
},
}
-- Bounded Status/SR.CU2 transition policy.
-- The value lattice and the transition consumer both read this immutable row; no second value pass is permitted.
-- The source says the enable/disable transition takes "2 clock cycles or so", so the boundary is conservative rather than exact.
M.CU2_TRANSITION_POLICY = {
status_register = 12,
enable_bit = 0x40000000,
required = 2,
visibility_kind = "post_producer_words",
evidence = {
confidence = "conservative",
source = "cpuspecifications.md:543,625-628",
},
}
return M
+9 -15
View File
@@ -22,12 +22,10 @@ local M = {}
local CACHE_KEY = "__duffle_repo_root__" local CACHE_KEY = "__duffle_repo_root__"
--- Resolve the repo root from this script's own path. Zero shell spawn. --- Resolve the repo root from this script's own path. Zero shell spawn.
--- `duffle_paths.lua` always lives at `<repo>/scripts/duffle_paths.lua`, so the repo root is the --- `duffle_paths.lua` always lives at `<repo>/scripts/duffle_paths.lua`, so the repo root is the parent of the directory containing this script.
--- parent of the directory containing this script. We derive it directly from `debug.getinfo(1, "S").source` --- We derive it directly from `debug.getinfo(1, "S").source` (returns `@<path>` for the currently-running chunk).
--- (returns `@<path>` for the currently-running chunk).
--- ---
--- If `debug.getinfo` can't parse this script's path (shouldn't happen — dofile always populates source), --- If `debug.getinfo` can't parse this script's path (shouldn't happen — dofile always populates source), return nil and let `M.setup()` fail loud.
--- return nil and let `M.setup()` fail loud.
--- @return string|nil --- @return string|nil
local function find_repo_root() local function find_repo_root()
if package.loaded[CACHE_KEY] then return package.loaded[CACHE_KEY] end if package.loaded[CACHE_KEY] then return package.loaded[CACHE_KEY] end
@@ -51,17 +49,13 @@ end
--- ---
--- This script does NOT touch the OS environment: no `os.setenv`, no `os.putenv`, no `$PATH` mods. --- This script does NOT touch the OS environment: no `os.setenv`, no `os.putenv`, no `$PATH` mods.
--- It just sets `package.path` and `package.cpath` (the standard Lua way to register module search dirs). --- It just sets `package.path` and `package.cpath` (the standard Lua way to register module search dirs).
--- lpeg is built by `update_deps.ps1` to `toolchain/lpeg/`, --- lpeg is built by `update_deps.ps1` to `toolchain/lpeg/`, which we wire into `package.cpath` here (so `require("lpeg")` from `duffle.lua` resolves without any global state).
--- which we wire into `package.cpath` here (so `require("lpeg")` from `duffle.lua` resolves without any global state).
function M.setup() function M.setup()
local repo_root = find_repo_root() local repo_root = find_repo_root()
if not repo_root then if not repo_root then
-- Unreachable in practice: find_repo_root() derives the repo root from this script's -- Unreachable in practice: find_repo_root() derives the repo root from this script's own source path via debug.getinfo(1, "S").source (no subprocess, no git CLI, <1ms).
-- own source path via debug.getinfo(1, "S").source (no subprocess, no git CLI, <1ms). -- A nil return means the source path did not match the expected <repo>/scripts/duffle_paths.lua layout — a packaging bug, not a "missing git repo" condition.
-- A nil return means the source path did not match the expected -- os.exit(2) is retained so a real failure surfaces loud rather than silently producing an unconfigured module table.
-- <repo>/scripts/duffle_paths.lua layout — a packaging bug, not a "missing git repo"
-- condition. os.exit(2) is retained so a real failure surfaces loud rather than
-- silently producing an unconfigured module table.
os.exit(2) os.exit(2)
end end
@@ -86,6 +80,6 @@ end
-- Run the setup as a side effect. -- Run the setup as a side effect.
M.setup() M.setup()
-- Now that package.path includes scripts/, `require("duffle")` resolves. Return the duffle module -- Now that package.path includes scripts/, `require("duffle")` resolves.
-- so callers can do `local duffle = dofile(...duffle_paths.lua)` in one line. -- Return the duffle module so callers can do `local duffle = dofile(...duffle_paths.lua)` in one line.
return require("duffle") return require("duffle")
File diff suppressed because it is too large Load Diff
+3 -4
View File
@@ -468,8 +468,7 @@ function M.find_type_unit_by_signature(info, target_sig_lo, target_sig_hi)
-- 20: <children> -- 20: <children>
if body_end - body_start >= 20 then if body_end - body_start >= 20 then
-- read_ref_sig8 / write_u32_le / etc. are 1-indexed (string:byte); -- read_ref_sig8 / write_u32_le / etc. are 1-indexed (string:byte);
-- pos / body_start / body_end are 0-based wire offsets, so the -- pos / body_start / body_end are 0-based wire offsets, so the 1-indexed byte at 0-based wire offset X is string:byte(X + 1).
-- 1-indexed byte at 0-based wire offset X is string:byte(X + 1).
-- Per DWARF5 §7.5.6, the type_unit body is laid out as: -- Per DWARF5 §7.5.6, the type_unit body is laid out as:
-- byte 0-1: version (2) -- byte 0-1: version (2)
-- byte 2: unit_type (1) -- DW_UT_type = 0x02 -- byte 2: unit_type (1) -- DW_UT_type = 0x02
@@ -612,13 +611,13 @@ function M.read_elf_sections(elf_path, section_names)
end end
--- Read ELF symbol addresses by walking the `.symtab` + `.strtab` sections directly (no `nm` subprocess). --- Read ELF symbol addresses by walking the `.symtab` + `.strtab` sections directly (no `nm` subprocess).
--- Returns a map `{name -> {addr, size_bytes}}` for every `code_<name>` symbol. --- Returns a map `{name -> {addr, size_bytes}}` for every defined symbol.
--- ---
--- **Conventions:** --- **Conventions:**
--- - ELF32 symtab entry = 16 bytes (`st_name:4 + st_value:4 + st_size:4 + st_info:1 + st_other:1 + st_shndx:2`); offsets within each entry are zero-based wire offsets. --- - ELF32 symtab entry = 16 bytes (`st_name:4 + st_value:4 + st_size:4 + st_info:1 + st_other:1 + st_shndx:2`); offsets within each entry are zero-based wire offsets.
--- - Direct Lua `string.byte`/`string.sub`/`string.find` boundaries receive `+ 1`. --- - Direct Lua `string.byte`/`string.sub`/`string.find` boundaries receive `+ 1`.
--- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded. --- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded.
--- - The `code_` prefix is stripped (MipsAtom_ macros emit bare atom names, no `code_` prefix). --- - Keys are the ELF symbol names as written (the C ident).
--- - `st_size > 0` filter excludes undefined/imported symbols. --- - `st_size > 0` filter excludes undefined/imported symbols.
--- ---
--- @param elf_path Path --- @param elf_path Path
+1 -1
View File
@@ -16,7 +16,7 @@ define tape_atoms
echo "[gdb_tape_atoms] STUB: run .\\build_psyq.ps1 to regenerate, then re-source this file." echo "[gdb_tape_atoms] STUB: run .\\build_psyq.ps1 to regenerate, then re-source this file."
end end
document tape_atoms document tape_atoms
List every tape atom symbol in the loaded ELF (code_<name>) with its .rodata address and word count. List every tape atom symbol in the loaded ELF with its .rodata address and word count.
STUB state: runtime file not sourced. Run build_psyq.ps1 to regenerate. STUB state: runtime file not sourced. Run build_psyq.ps1 to regenerate.
end end
+6 -6
View File
@@ -1,10 +1,9 @@
# scripts/launch_pcsx_debug.ps1 # scripts/launch_pcsx_debug.ps1
# #
# One-shot launcher for debug sessions: starts pcsx-redux with the .ps-exe # One-shot launcher for debug sessions:
# loaded, the gdb stub enabled, AND the pcsx_debug_helper Lua plugin loaded # Starts pcsx-redux with the .ps-exe loaded, the gdb stub enabled,
# so external CLI tools (gdb's `shell` command, etc.) # AND the pcsx_debug_helper Lua plugin loaded so external CLI tools (gdb's `shell` command, etc.)
# can read GTE state via http://localhost:8080/api/v1/lua/gte # can read GTE state via http://localhost:8080/api/v1/lua/gte (the gdb stub doesn't expose COP2 at all).
# (the gdb stub doesn't expose COP2 at all).
# #
# usage: # usage:
# .\scripts\launch_pcsx_debug.ps1 # .\scripts\launch_pcsx_debug.ps1
@@ -84,7 +83,8 @@ try {
$r = Invoke-WebRequest -Uri "http://localhost:$WebPort/api/v1/lua/gte" -UseBasicParsing -TimeoutSec 5 $r = Invoke-WebRequest -Uri "http://localhost:$WebPort/api/v1/lua/gte" -UseBasicParsing -TimeoutSec 5
$firstLine = ([System.Text.Encoding]::UTF8.GetString($r.Content) -split "`n")[0] $firstLine = ([System.Text.Encoding]::UTF8.GetString($r.Content) -split "`n")[0]
Write-Host "GTE handler OK: $firstLine" -ForegroundColor Green Write-Host "GTE handler OK: $firstLine" -ForegroundColor Green
} catch { }
catch {
Write-Warning "GTE handler NOT responding: $_" Write-Warning "GTE handler NOT responding: $_"
Write-Host "Check the pcsx-redux Lua Console for debug cli messages." -ForegroundColor Yellow Write-Host "Check the pcsx-redux Lua Console for debug cli messages." -ForegroundColor Yellow
} }
+17 -49
View File
@@ -115,7 +115,7 @@ end
--- Post-loop: Needs full-corpus `annot_counts` from pipe_ctx. --- Post-loop: Needs full-corpus `annot_counts` from pipe_ctx.
--- @param pipe_ctx PipeCtx --- @param pipe_ctx PipeCtx
--- @param findings Findings --- @param findings Findings
local function check_unique_annotation(pipe_ctx, findings) local function check_unique_annotation(_item, pipe_ctx, findings)
for name, n in pairs(pipe_ctx.annot_counts) do for name, n in pairs(pipe_ctx.annot_counts) do
if n > 1 then if n > 1 then
findings.errors[#findings.errors + 1] = { findings.errors[#findings.errors + 1] = {
@@ -147,7 +147,8 @@ end
--- @param m MacroEntry --- @param m MacroEntry
--- @param wc table<string, integer> -- Shared word-count table (from ctx.shared.word_counts) --- @param wc table<string, integer> -- Shared word-count table (from ctx.shared.word_counts)
--- @param findings Findings --- @param findings Findings
local function check_macro_word_drift(m, wc, findings) local function check_macro_word_drift(m, pipe_ctx, findings)
local wc = (pipe_ctx and pipe_ctx.word_counts) or {}
local declared = wc[m.name] local declared = wc[m.name]
if not declared then if not declared then
findings.errors[#findings.errors + 1] = { findings.errors[#findings.errors + 1] = {
@@ -429,40 +430,17 @@ local CHECK_RULES = {
--- @param ctx PassCtx --- @param ctx PassCtx
--- @return PipeCtx --- @return PipeCtx
local function build_corpus_pipe_ctx(ctx) local function build_corpus_pipe_ctx(ctx)
local corpus = ctx.shared and ctx.shared.corpus local view = duffle.corpus_view(ctx)
if not corpus then
error("annotation requires ctx.shared.corpus "
.. "(the canonical corpus is the source of truth; "
.. "no per-source fallback is supported)", 0)
end
-- `corpus.atom_infos` preserves source order and duplicates; I precompute counts here for `check_unique_annotation` and the per-source checks.
local annot_counts = {} local annot_counts = {}
for _, info in ipairs(corpus.atom_infos or {}) do for _, info in ipairs(view.atom_infos) do
if info and info.atom_name then if info and info.atom_name then
annot_counts[info.atom_name] = (annot_counts[info.atom_name] or 0) + 1 annot_counts[info.atom_name] = (annot_counts[info.atom_name] or 0) + 1
end end
end end
view.annot_counts = annot_counts
-- Every consumer of these fields observes mutations via the canonical corpus without independently mutable registry construction. view.atom_infos_list = view.atom_infos
return { view.word_counts = ctx.shared.corpus.word_counts or {}
-- Cross-source lookup tables from corpus. return view
register_alias_registry = corpus.register_alias_registry or {},
type_name_registry = corpus.type_name_registry or {},
atom_views = corpus.atom_views or {},
atom_ctxs = corpus.atom_ctxs or {},
atom_phases = corpus.atom_phases or {},
binds_by_name = corpus.binds_by_name or {},
atoms_by_name = corpus.atoms_by_name or {},
-- Corpus-wide ordered list of atom_info records (source-order + duplicates).
atom_infos_list = corpus.atom_infos or {},
-- Corpus-wide annotation count aggregation (post-rule consumes this).
annot_counts = annot_counts,
-- Corpus-wide collisions (recorded by scan_source.merge_corpus_registries).
collisions = corpus.collisions or {},
-- `check_macro_word_drift` reads `corpus.word_counts`, populated by word_count_eval.run.
word_counts = corpus.word_counts or {},
}
end end
--- Validate one source against its pre-scanned SourceScan payload + the corpus-wide pipe_ctx. --- Validate one source against its pre-scanned SourceScan payload + the corpus-wide pipe_ctx.
@@ -477,8 +455,8 @@ local function validate(ctx, src, corpus_pipe_ctx)
-- Project the pre-scanned atoms to the AtomEntry shape this pass needs. -- Project the pre-scanned atoms to the AtomEntry shape this pass needs.
local atoms = {} local atoms = {}
for _, a in ipairs(scan.atoms) do for _, a in ipairs(scan.atoms) do
if a.kind == "atom" then if a.kind == "atom" or a.kind == "atom_proc" then
atoms[#atoms + 1] = { line = a.line, name = a.raw_name } atoms[#atoms + 1] = { line = a.line, name = a.raw_name or a.name }
end end
end end
@@ -536,38 +514,28 @@ local function validate(ctx, src, corpus_pipe_ctx)
-- THE per-annotation pipeline. ONE loop. CHECK_RULES dispatches per_annot rules. -- THE per-annotation pipeline. ONE loop. CHECK_RULES dispatches per_annot rules.
for _, a in ipairs(annots) do for _, a in ipairs(annots) do
for _, rule in ipairs(CHECK_RULES) do duffle.run_check_rules(CHECK_RULES, "per_annot", a, pipe_ctx, findings)
if rule.per_annot then rule.per_annot(a, pipe_ctx, findings) end
end
end end
-- Post-loop rules (one-shot checks that need full-corpus aggregation in pipe_ctx). -- Post-loop rules (one-shot checks that need full-corpus aggregation in pipe_ctx).
for _, rule in ipairs(CHECK_RULES) do duffle.run_check_rules(CHECK_RULES, "post", nil, pipe_ctx, findings)
if rule.post then rule.post(pipe_ctx, findings) end
end
-- scan_source records each marker in scan.debug_skip_markers; this loop validates each record independently and emits at most one error per marker. -- scan_source records each marker in scan.debug_skip_markers; this loop validates each record independently and emits at most one error per marker.
-- Valid markers stamp `debug_skip = true` on the following atom or component declaration, which downstream consumers read directly. -- Valid markers stamp `debug_skip = true` on the following atom or component declaration, which downstream consumers read directly.
local skip_markers = scan.debug_skip_markers or {} local skip_markers = scan.debug_skip_markers or {}
for _, marker in ipairs(skip_markers) do for _, marker in ipairs(skip_markers) do
for _, rule in ipairs(CHECK_RULES) do duffle.run_check_rules(CHECK_RULES, "per_skip_marker", marker, pipe_ctx, findings)
if rule.per_skip_marker then rule.per_skip_marker(marker, pipe_ctx, findings) end
end
end end
-- Per-macro rules (TAPE_WORDS vs WORD_COUNT drift). -- Per-macro rules (TAPE_WORDS vs WORD_COUNT drift).
local wc = corpus_pipe_ctx.word_counts pipe_ctx.word_counts = corpus_pipe_ctx.word_counts
for _, m in ipairs(scan.macros) do for _, m in ipairs(scan.macros) do
for _, rule in ipairs(CHECK_RULES) do duffle.run_check_rules(CHECK_RULES, "per_macro", m, pipe_ctx, findings)
if rule.per_macro then rule.per_macro(m, wc, findings) end
end
end end
-- Per-source rules (reg defaults, atom_view layout, compute-register type overrides, Binds_* field uniqueness). -- Per-source rules (reg defaults, atom_view layout, compute-register type overrides, Binds_* field uniqueness).
-- Each per_source rule sees the full scan payload via pipe_ctx. -- Each per_source rule sees the full scan payload via pipe_ctx.
for _, rule in ipairs(CHECK_RULES) do duffle.run_check_rules(CHECK_RULES, "per_source", src, pipe_ctx, findings)
if rule.per_source then rule.per_source(src, pipe_ctx, findings) end
end
-- Information summary (always emitted). -- Information summary (always emitted).
findings.info[#findings.info + 1] = { findings.info[#findings.info + 1] = {
+19 -7
View File
@@ -83,6 +83,7 @@ local function canonical_word_entries(atom)
line = event.call_line or item.line or 0, line = event.call_line or item.line or 0,
text = event.call_text or item.call_text or "", text = event.call_text or item.call_text or "",
body_line = event.body_line or item.body_line or item.line or 0, body_line = event.body_line or item.body_line or item.line or 0,
gpr_keys = event.gpr_keys,
invocation = (event.outermost_invocation_id invocation = (event.outermost_invocation_id
and paths.invocations and paths.invocations
and paths.invocations[event.outermost_invocation_id]) or nil, and paths.invocations[event.outermost_invocation_id]) or nil,
@@ -260,12 +261,12 @@ local function append_gdb_commands(lines, matched)
for _, a in ipairs(matched) do for _, a in ipairs(matched) do
-- gdb 12.1 quirk: literals in printf args require an attached target. -- gdb 12.1 quirk: literals in printf args require an attached target.
-- Use the per-atom convenience vars set above as printf args. -- Use the per-atom convenience vars set above as printf args.
lines[#lines + 1] = string.format(' printf " code_%%-32s @ 0x%%08x %%4d words\\n", $__atom_name_%d, $__atom_addr_%d, $__atom_words_%d', lines[#lines + 1] = string.format(' printf " %%-32s @ 0x%%08x %%4d words\\n", $__atom_name_%d, $__atom_addr_%d, $__atom_words_%d',
a.idx, a.idx, a.idx) a.idx, a.idx, a.idx)
end end
lines[#lines + 1] = "end" lines[#lines + 1] = "end"
lines[#lines + 1] = "document tape_atoms" lines[#lines + 1] = "document tape_atoms"
lines[#lines + 1] = " List every tape atom symbol in the loaded ELF (code_<name>) with .rodata addr + word count." lines[#lines + 1] = " List every tape atom symbol in the loaded ELF with .rodata addr + word count."
lines[#lines + 1] = "end" lines[#lines + 1] = "end"
lines[#lines + 1] = "" lines[#lines + 1] = ""
@@ -284,10 +285,10 @@ local function append_gdb_commands(lines, matched)
for _, a in ipairs(matched) do for _, a in ipairs(matched) do
lines[#lines + 1] = string.format("define break_atom_%s", a.name) lines[#lines + 1] = string.format("define break_atom_%s", a.name)
lines[#lines + 1] = string.format(" break *$__atom_addr_%d", a.idx) lines[#lines + 1] = string.format(" break *$__atom_addr_%d", a.idx)
lines[#lines + 1] = string.format(' printf " Breakpoint set at code_%s (0x%%08x)\\n", $__atom_addr_%d', a.name, a.idx) lines[#lines + 1] = string.format(' printf " Breakpoint set at %s (0x%%08x)\\n", $__atom_addr_%d', a.name, a.idx)
lines[#lines + 1] = "end" lines[#lines + 1] = "end"
lines[#lines + 1] = string.format("document break_atom_%s", a.name) lines[#lines + 1] = string.format("document break_atom_%s", a.name)
lines[#lines + 1] = string.format(" Set a breakpoint at code_%s.", a.name) lines[#lines + 1] = string.format(" Set a breakpoint at %s.", a.name)
lines[#lines + 1] = "end" lines[#lines + 1] = "end"
lines[#lines + 1] = "" lines[#lines + 1] = ""
end end
@@ -322,7 +323,7 @@ local function append_gdb_commands(lines, matched)
-- Precompute end_addr (gdb 12.1's expression evaluator chokes on `addr + words*4`). -- Precompute end_addr (gdb 12.1's expression evaluator chokes on `addr + words*4`).
lines[#lines + 1] = string.format(" set $__end_%d = $__atom_addr_%d + $__atom_words_%d * 4", a.idx, a.idx, a.idx) lines[#lines + 1] = string.format(" set $__end_%d = $__atom_addr_%d + $__atom_words_%d * 4", a.idx, a.idx, a.idx)
lines[#lines + 1] = string.format(" if $__pc >= $__atom_addr_%d && $__pc < $__end_%d", a.idx, a.idx) lines[#lines + 1] = string.format(" if $__pc >= $__atom_addr_%d && $__pc < $__end_%d", a.idx, a.idx)
lines[#lines + 1] = string.format(' printf "atom: code_%%s\\n", $__atom_name_%d', a.idx) lines[#lines + 1] = string.format(' printf "atom: %%s\\n", $__atom_name_%d', a.idx)
lines[#lines + 1] = ' printf "addr: 0x%08x\\n", $__pc' lines[#lines + 1] = ' printf "addr: 0x%08x\\n", $__pc'
lines[#lines + 1] = string.format(" set $__word = ($__pc - $__atom_addr_%d) / 4", a.idx) lines[#lines + 1] = string.format(" set $__word = ($__pc - $__atom_addr_%d) / 4", a.idx)
lines[#lines + 1] = string.format(' printf "word: %%d/%%d\\n", $__word, $__atom_words_%d', a.idx) lines[#lines + 1] = string.format(' printf "word: %%d/%%d\\n", $__word, $__atom_words_%d', a.idx)
@@ -491,8 +492,19 @@ function M.render_atom_source_map(atom)
local lines = {} local lines = {}
lines[#lines + 1] = string.format("ATOM %s %d", (atom.raw_name or atom.name), total) lines[#lines + 1] = string.format("ATOM %s %d", (atom.raw_name or atom.name), total)
for _, entry in ipairs(entries) do for _, entry in ipairs(entries) do
lines[#lines + 1] = string.format("WORD %d LINE %d TEXT %s", local word_line = string.format("WORD %d LINE %d TEXT %s",
entry.pos, entry.line, entry.text) entry.pos, entry.line, entry.text)
local keys = {}
for pos = 1, 16 do
local k = entry.gpr_keys and entry.gpr_keys[pos]
if type(k) == "string" and k:sub(1, 7) == "reguse:" then
keys[#keys + 1] = k
end
end
if #keys > 0 then
word_line = word_line .. " KEYS " .. table.concat(keys, ",")
end
lines[#lines + 1] = word_line
end end
lines[#lines + 1] = "ENDATOM" lines[#lines + 1] = "ENDATOM"
return table.concat(lines, "\n") .. "\n" return table.concat(lines, "\n") .. "\n"
@@ -529,7 +541,7 @@ function M.render_atom_provenance(atom, wc, rel_path)
return table.concat(lines, "\n") .. "\n" return table.concat(lines, "\n") .. "\n"
end end
--- Pass entry. For each source that declares at least one `MipsAtom_(name)` / `MipsCode code_<name>`, --- Pass entry. For each source that declares at least one tape atom,
--- emit two files in `<out_root>/`: `<basename>.atoms.sourcemap.txt` (per-word call-site map) and `<basename>.atoms.provenance.txt` --- emit two files in `<out_root>/`: `<basename>.atoms.sourcemap.txt` (per-word call-site map) and `<basename>.atoms.provenance.txt`
--- (per-word definition + body line, resolved via the outermost `mac_X(...)` invocation). --- (per-word definition + body line, resolved via the outermost `mac_X(...)` invocation).
--- When `ctx.flags.gdb_runtime` is true and `ctx.flags.elf_path` exists, also emit the post-link gdb script `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`. --- When `ctx.flags.gdb_runtime` is true and `ctx.flags.elf_path` exists, also emit the post-link gdb script `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`.
+28 -40
View File
@@ -27,50 +27,34 @@
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
--- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- THE GPR ALLOCATION POOL — what is allocatable, and (more importantly) WHY -- THE GPR ALLOCATION POOL — what's allocatable, and (more importantly) WHY
--- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- --
--- The auto-reg pass picks physical GPRs for `atom_auto_reg(...)` / `phase_auto_reg(...)` markers. -- The auto-reg pass picks physical GPRs for `atom_auto_reg(...)` / `phase_auto_reg(...)` markers.
--- It allocates from a FIXED 10-register pool. -- The 24-register pool covers R2-R25 (the user/atom allocatable surface):
--- This comment block makes the inclusion AND exclusion criteria obvious so a reader doesn't have -- R_T0..R_T7, R_V0..R_V1, R_A0..A3, R_S0..S7, R_T8..T9.
--- to grep lottes_tape.h + mips.h to understand the design. -- Excluded (and never added to the pool):
--- -- R_0 (code 0) — hardwired zero. Cannot be written.
--- ── WHAT'S IN THE POOL (10 GPRs, all caller-trash per the O32 ABI) ──────── -- R_AT (code 1) — assembler temporary. Reserved by the MIPS O32 ABI.
--- R_T0..R_T7 (GPR codes 8..15), R_V0..R_V1 (GPR codes 2..3) -- R_A0..A3 — explicitly omitted above even though their integer codes
--- The workhorse of every atom body. The uesr should be aware of atom allocation across atoms they chain. -- map to POOL entries; the pool-construction loop below
--- If they have a collision it means either they didn't saturate the register file optimally for a phase, -- only references the POOL string literals, never the
--- or the may have made the workload to large for the run. -- integer codes, so they are NOT auto-allocated by default.
--- -- (A0-A3 become available when the user adds them to
--- ── WHAT'S NOT IN THE POOL — and WHY (the "obvious exclusions") ──────────── -- POOL or hardcodes an R_A0 reference in the atom body.)
--- R_T9 (GPR code 25) — R_TapePtr, the tape instruction stream pointer. -- R_K0/K1 (codes 26-27) — kernel / interrupt handler reserves. Never touched by user code.
--- Owned by the tape runtime (in tape_run / tape_run_a02_s07). -- R_GP/SP/FP/RA (codes 28-31) — R_SP/R_FP/R_RA are tape-runtime carriers between
--- `rgcc(R_TapePtr)` register-variable ties the C compiler's view to $t9 across the whole tape_run. -- tape_enter and tape_exit; R_GP stays the host global pointer.
--- The auto-reg pass MUST NOT clobber this; doing so would desync the C-side tape pointer from the
--- hardware pointer and crash on the next tape_run.
---
--- R_T8 (GPR code 24) — R_AtomJmp, the atom-jump register used by the 4-word yield handshake.
--- Every `mac_yield()` / `mac_yield_tail` does `load_word R_AtomJmp, R_TapePtr, 0` then
--- `jump_reg R_AtomJmp`. The auto-reg pass MUST NOT clobber this either, or the atom dispatcher breaks.
--- Owned by the tape runtime, same family as R_TapePtr.
---
--- R_AT (GPR code 1) — Assembler temporary. Reserved by the MIPS O32 ABI for pseudoinstruction expansion
--- (lottes_tape.h:86, mips.h:93). The ISA's psuedo instructions use it as a scratch temporary.
---
--- R_A0..A3 (codes 4..7) — Function arguments. Used in tape_run_a02_s07, see below.
--- R_S0..S7 (codes 16..23) — Callee-saved. Preserved across C-ABI calls by convention.
--- The `tape_run_a02_s07` variant clobbers them deliberately, but the default `tape_run` does NOT.
--- Kept out of POOL to preserve the conservative default.
--- Add them in a separate "big clobber" pool if/when needed.
---
--- R_K0/K1 (codes 26..27) — Kernel / interrupt handler reserves. Never touched by user code; OS-internal.
--- R_GP/SP/FP/RA (codes 28..31) — Stack frame + return-address. Owned by the C compiler; never allocatable.
--- R_0 (code 0) — Hardwired zero. Cannot be written.
--- ---
local POOL = { local POOL = {
"R_T0", "R_T1", "R_T2", "R_T3", "R_T0", "R_T1", "R_T2", "R_T3",
"R_T4", "R_T5", "R_T6", "R_T7", "R_T4", "R_T5", "R_T6", "R_T7",
"R_V0", "R_V1", "R_V0", "R_V1",
"R_A0", "R_A1", "R_A2", "R_A3",
"R_S0", "R_S1", "R_S2", "R_S3",
"R_S4", "R_S5", "R_S6", "R_S7",
"R_T8", "R_T9",
} }
-- Map from integer MIPS GPR code (the `code` field on AliasEntry) to the physical GPR ident in POOL. -- Map from integer MIPS GPR code (the `code` field on AliasEntry) to the physical GPR ident in POOL.
@@ -80,8 +64,12 @@ local POOL = {
-- are deliberately omitted — see the comment block above for the WHY of each exclusion. -- are deliberately omitted — see the comment block above for the WHY of each exclusion.
local INT_CODE_TO_POOL_GPR = { local INT_CODE_TO_POOL_GPR = {
[2] = "R_V0", [3] = "R_V1", [2] = "R_V0", [3] = "R_V1",
[4] = "R_A0", [5] = "R_A1", [6] = "R_A2", [7] = "R_A3",
[8] = "R_T0", [9] = "R_T1", [10] = "R_T2", [11] = "R_T3", [8] = "R_T0", [9] = "R_T1", [10] = "R_T2", [11] = "R_T3",
[12] = "R_T4", [13] = "R_T5", [14] = "R_T6", [15] = "R_T7", [12] = "R_T4", [13] = "R_T5", [14] = "R_T6", [15] = "R_T7",
[16] = "R_S0", [17] = "R_S1", [18] = "R_S2", [19] = "R_S3",
[20] = "R_S4", [21] = "R_S5", [22] = "R_S6", [23] = "R_S7",
[24] = "R_T8", [25] = "R_T9",
} }
-- Stable sort for deterministic allocation order. -- Stable sort for deterministic allocation order.
@@ -109,7 +97,7 @@ local function allocate_phase(phase_label, decls)
line = 0, line = 0,
msg = string.format("phase_register_pool_exhausted: " msg = string.format("phase_register_pool_exhausted: "
.. "phase '%s' requested symbol '%s' but the pool has no remaining registers " .. "phase '%s' requested symbol '%s' but the pool has no remaining registers "
.. "(max 10 per phase: R_T0..R_T7 + R_V0..R_V1). Split the phase or use hardcoded GPRs." .. "(max 24 per phase: R_T0..R_T7 + R_V0..R_V1 + R_A0..R_A3 + R_S0..R_S7 + R_T8..R_T9). Split the phase or use hardcoded GPRs."
, phase_label, sym), , phase_label, sym),
} }
return result, errors return result, errors
+182 -53
View File
@@ -7,7 +7,7 @@
--- then resolves the function-args string from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration via a backward walk. --- then resolves the function-args string from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration via a backward walk.
--- ---
--- `MipsAtom_Proc_(X, ab, { body })` declarations (kind="atom_proc") are ATOMS, not components, and are deliberately excluded — --- `MipsAtom_Proc_(X, ab, { body })` declarations (kind="atom_proc") are ATOMS, not components, and are deliberately excluded —
--- atoms get emitted via `tb_emit(tb, code_<name>)` linker symbols, not inlined as `mac_*` macros. --- the ELF symbol is the C ident. Raw `MipsCode code_*` is leftover, not the atom rule.
--- ---
--- Emits one `gen/macs.h` per *immediate source directory* with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation. --- Emits one `gen/macs.h` per *immediate source directory* with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
--- All sources inside the same directory contribute to the same file (per-directory aggregation). --- All sources inside the same directory contribute to the same file (per-directory aggregation).
@@ -96,48 +96,21 @@ local M = {}
-- so this file reads it forward rather than re-walking the source. -- so this file reads it forward rather than re-walking the source.
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- Find the args of the function declaration that immediately precedes a `MipsAtomComp_Proc_` invocation of the given name. --- Find the args of the function declaration that immediately precedes a `MipsAtomComp_Proc_` invocation.
--- Returns the args string (e.g., `"U4 off, U4 code, U1 r, U1 g, U1 b"`) or nil if no function declaration is found. --- Returns the args string (e.g., `"U4 off, U4 code, U1 r, U1 g, U1 b"`) or nil if no function declaration is found.
--- ---
--- Convention: function form is --- After the `sym` arg was dropped from MipsAtomComp_Proc_, the component name
--- `FI_ Slice_MipsCode ac_X(args) MipsAtomComp_Proc_(ac_X, { body })` --- and the args both come from the preceding `FI_ Slice_MipsCode ac_X(args)`
--- We find the LAST occurrence of `"ac_X("` before `before_pos` and extract the args from inside the parens. --- declaration. The shared `duffle.find_function_decl_for` helper does the
--- We then verify the preceding context ends with `Slice_MipsCode` --- backward walk; this function returns just the args.
--- (the function-decl keyword with possible qualifiers between).
--- ---
--- @param source string --- @param source string
--- @param name string --- @param name string (retained for signature stability; unused — the walk derives the name)
--- @param before_pos integer --- @param before_pos integer
--- @return string|nil --- @return string|nil
local function find_function_args_for(source, name, before_pos) local function find_function_args_for(source, name, before_pos)
-- Find the LAST occurrence of `name + "("` in `source[1..before_pos]`. local _, args_inner = duffle.find_function_decl_for(source, before_pos, #MIPS_ATOM)
local name_open = name .. "(" return args_inner
local last_idx = nil
local scan_pos = 1
while true do
-- Pass `before_pos + 1` so string.find only returns positions < before_pos + 1
-- (string.find's 4th arg `plain` is true; we use the 3rd arg `init` for the upper bound).
local found = source:find(name_open, scan_pos, true)
if not found or found >= before_pos then break end
last_idx = found
scan_pos = found + #name_open
end
if not last_idx then return nil end
-- Verify the preceding context ends with "MipsAtom" (with possible qualifiers between).
local before = source:sub(1, last_idx - 1)
local trimmed = duffle.trim(before)
if trimmed:sub(-#MIPS_ATOM) ~= MIPS_ATOM then
-- Preceding context is not a function declaration.
return nil
end
local open_paren = last_idx + #name -- position of "("
-- scan: MipsAtom ac_X(
local inner = duffle.read_parens(source, open_paren)
-- scan: MipsAtom ac_X(<args>)
if not inner then return nil end
return inner
end end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -157,6 +130,60 @@ local function extract_arg_names(args_str)
for _, tok in ipairs(tokens) do for _, tok in ipairs(tokens) do
local trimmed = duffle.trim(tok) local trimmed = duffle.trim(tok)
if trimmed ~= "" then if trimmed ~= "" then
-- Strip trailing block comment (/* ... */) from the token, if present.
-- split_top_level_commas only skips block comments at TOP LEVEL (between commas),
-- not block comments embedded WITHIN a token between a parameter and a trailing comma.
-- Without this strip, the identifier-walk below stops at the `/` of `*/` and returns
-- the wrong name (or nothing). See `test_extract_arg_names_handles_trailing_block_comments`.
local trimmed_end = #trimmed
if trimmed_end >= 2 and trimmed:sub(trimmed_end - 1, trimmed_end) == "*/" then
-- Find the matching `/*` that opens the trailing comment.
-- Walk back from the `*/` looking for `/*` (whitespace + `/*`).
local close_pos = trimmed_end - 1 -- position of the second-to-last char
-- Walk back: skip trailing whitespace, then look for the `/*` opener.
while close_pos > 1 do
local ch = trimmed:sub(close_pos, close_pos)
if ch == " " or ch == "\t" or ch == "\n" or ch == "\r" then
close_pos = close_pos - 1
else
break
end
end
-- Now scan back from close_pos for the `/*` opener (slashes are at close_pos-1 and close_pos-2).
local opener_pos = nil
local scan = close_pos - 3
while scan >= 1 do
if trimmed:sub(scan, scan + 1) == "/*" then
opener_pos = scan
break
end
scan = scan - 1
end
if opener_pos then
-- Truncate everything from opener_pos onwards.
trimmed = duffle.trim(trimmed:sub(1, opener_pos - 1))
end
end
if trimmed == "" then goto continue end
-- Strip trailing array suffix `[N]` if present.
-- Example: `Reg r_data[4]` → identifier is `r_data`, not `4`.
trimmed_end = #trimmed
if trimmed_end >= 4 and trimmed:sub(trimmed_end, trimmed_end) == "]" then
-- Walk back: skip digits, expect `[`.
local bracket_pos = trimmed_end - 1
while bracket_pos > 1 do
local ch = trimmed:sub(bracket_pos, bracket_pos)
if ch >= "0" and ch <= "9" then
bracket_pos = bracket_pos - 1
else
break
end
end
if bracket_pos >= 1 and trimmed:sub(bracket_pos, bracket_pos) == "[" then
trimmed = duffle.trim(trimmed:sub(1, bracket_pos - 1))
end
end
if trimmed == "" then goto continue end
-- Find the identifier at the end: walk back over trailers (whitespace + `*` + `[]`), -- Find the identifier at the end: walk back over trailers (whitespace + `*` + `[]`),
-- then walk back over the identifier chars (alnum + `_`). -- then walk back over the identifier chars (alnum + `_`).
local ident_end = #trimmed local ident_end = #trimmed
@@ -180,12 +207,21 @@ local function extract_arg_names(args_str)
ident_start = ident_start + 1 ident_start = ident_start + 1
local name = trimmed:sub(ident_start, ident_end) local name = trimmed:sub(ident_start, ident_end)
if name ~= "" then names[#names + 1] = name end if name ~= "" then names[#names + 1] = name end
::continue::
end end
end end
if #names == 0 then return nil end if #names == 0 then return nil end
return names return names
end end
local function formal_arg_names(args_str)
local names = extract_arg_names(args_str)
if not names then return nil end
if names[1] == "ab" then table.remove(names, 1) end
if #names == 0 then return nil end
return names
end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Component projection (read from pre-scanned SourceScan) -- Component projection (read from pre-scanned SourceScan)
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -206,7 +242,7 @@ local function project_components(source, scan)
-- Only `MipsAtomComp_(ac_X)` (kind="comp_bare") and `MipsAtomComp_Proc_(ac_X, ...)` (kind="comp_proc") -- Only `MipsAtomComp_(ac_X)` (kind="comp_bare") and `MipsAtomComp_Proc_(ac_X, ...)` (kind="comp_proc")
-- are COMPONENTS — they get inlined via `mac_<name>` aliases inside atom bodies. -- are COMPONENTS — they get inlined via `mac_<name>` aliases inside atom bodies.
-- `MipsAtom_Proc_` (kind="atom_proc") is an ATOM (ends with `mac_yield()`); it gets emitted via -- `MipsAtom_Proc_` (kind="atom_proc") is an ATOM (ends with `mac_yield()`); it gets emitted via
-- `tb_emit(tb, code_<name>)` (linker symbol), NOT inlined as a macro. Including `atom_proc` here -- `tb_emit` of the C ident, NOT inlined as a macro. Including `atom_proc` here
-- would incorrectly emit `mac_<name>` aliases for atoms, polluting `gen/macs.h`. -- would incorrectly emit `mac_<name>` aliases for atoms, polluting `gen/macs.h`.
-- See `docs/duffle_dsl_primer.md` §"mac_* aliases" for the contract. -- See `docs/duffle_dsl_primer.md` §"mac_* aliases" for the contract.
if a.kind == "comp_bare" or a.kind == "comp_proc" then if a.kind == "comp_bare" or a.kind == "comp_proc" then
@@ -224,6 +260,7 @@ local function project_components(source, scan)
body_off = a.body_off, body_off = a.body_off,
body_tokens = a.body_tokens, body_tokens = a.body_tokens,
args = args, args = args,
arg_names = formal_arg_names(args),
comment = comment, comment = comment,
kind = a.kind, -- "comp_bare" | "comp_proc"; provenance emitter reads this. kind = a.kind, -- "comp_bare" | "comp_proc"; provenance emitter reads this.
debug_skip = a.debug_skip == true, debug_skip = a.debug_skip == true,
@@ -390,8 +427,10 @@ local function cycle_cost_rec(name, comp_by_name, latency, cache)
local nested = ident:sub(MAC_PREFIX_LEN + 1) local nested = ident:sub(MAC_PREFIX_LEN + 1)
n = n + cycle_cost_rec(nested, comp_by_name, latency, cache) n = n + cycle_cost_rec(nested, comp_by_name, latency, cache)
else else
-- Leaf instruction or pseudo-macro. Look up in INSTRUCTION_LATENCY; default 1. -- Leaf instruction or pseudo-macro.
n = n + (latency[ident] or 1) local isa = duffle.instr(ident)
local gte = duffle.gte(ident)
n = n + ((isa and isa.cycles) or (gte and gte.cycles) or latency[ident] or 1)
end end
end end
end end
@@ -411,6 +450,10 @@ end
--- @param cache table<string, integer> --- @param cache table<string, integer>
--- @return integer --- @return integer
local function gp0_contrib_rec(name, comp_by_name, cache) local function gp0_contrib_rec(name, comp_by_name, cache)
if name:match("^insert_ot_tag") then
cache[name] = 0
return 0
end
if cache[name] ~= nil then return cache[name] end if cache[name] ~= nil then return cache[name] end
cache[name] = -1 cache[name] = -1
local cc = comp_by_name[name] local cc = comp_by_name[name]
@@ -426,8 +469,15 @@ local function gp0_contrib_rec(name, comp_by_name, cache)
-- Nested `mac_X(...)` call: recurse. -- Nested `mac_X(...)` call: recurse.
local nested = ident:sub(MAC_PREFIX_LEN + 1) local nested = ident:sub(MAC_PREFIX_LEN + 1)
n = n + gp0_contrib_rec(nested, comp_by_name, cache) n = n + gp0_contrib_rec(nested, comp_by_name, cache)
elseif ident == "gte_sw" then
n = n + 1
elseif ident == "store_word" or ident == "store_half" or ident == "store_byte" then elseif ident == "store_word" or ident == "store_half" or ident == "store_byte" then
if trimmed:find("R_PrimCursor", 1, true) then if trimmed:find("R_PrimCursor", 1, true)
or trimmed:find("O_(Poly_", 1, true)
or trimmed:find("r_prim_cursor", 1, true)
or trimmed:find("r_primitive_cursor", 1, true)
or trimmed:find("r_base", 1, true)
then
n = n + 1 n = n + 1
end end
end end
@@ -493,18 +543,9 @@ end
--- @param args_str string|nil --- @param args_str string|nil
--- @return string --- @return string
local function signature_from_args(args_str) local function signature_from_args(args_str)
local arg_names = extract_arg_names(args_str) local names = formal_arg_names(args_str)
if arg_names and #arg_names > 0 then if names then
-- Drop the leading `ab` (atom-builder) first arg if present. return table.concat(names, ", ")
-- Convention: `MipsAtomComp_Proc_` components always declare `ab` as the first function-arg
-- (type `MipsAtomBuilder_R`), mirroring the macro signature in `lottes_tape.h`.
if arg_names[1] == "ab" then
table.remove(arg_names, 1)
end
if #arg_names > 0 then
return table.concat(arg_names, ", ")
end
return "..." -- `ab` was the only arg; fall through to variadic
end end
return "..." return "..."
end end
@@ -518,16 +559,103 @@ local function strip_trailing_continuation(lines)
end end
end end
--- Classify a token as a "pure delay marker token" (a delay-marker identifier
--- with no following instruction — only whitespace and/or block comments).
--- Examples that match:
--- * `GteDelay_` → marker alone
--- * `GteDelay_ /* RT diagonal: D1 = a.x... */` → marker + block comment
--- * `GteDelay_ /* RT diagonal: ... */\n\t` → marker + comment + trailing whitespace
--- Examples that DO NOT match (these contain a real instruction after the marker
--- and must be preserved verbatim so the instruction still gets emitted):
--- * `GteDelay_ nop2`
--- * `GteDelay_ add_si(r.dst_ptr, r.scratch, dst_offset)`
---
--- Why this classification matters: the metaprogram emits tokens separated by `,`
--- and joins them with `\<newline>` line continuations. After C preprocessor
--- phase 2 (line splicing), the macro body collapses to a single logical line.
--- Each delay-marker identifier expands to empty (its definition
--- `#define GteDelay_ // ...` consumes the `//` line comment during preprocessing
--- of the definition itself, leaving an empty replacement list). When a token
--- is purely a delay marker with only a trailing comment, the `,` the metaprogram
--- normally adds before each token-after-the-first brackets empty content and
--- produces the syntax error `,,` (`expected expression before ',' token`) at
--- C compile. The metaprogram therefore emits such tokens WITHOUT the leading
--- `,` (see `token_skips_leading_comma`) — but the marker + trailing comment
--- are still emitted verbatim so the annotation is preserved in `gen/macs.h`.
--- @param tok string -- a single token from split_top_level_commas (already trimmed at the start, may contain trailing whitespace + block comment)
--- @return boolean
local function is_pure_delay_marker_token(tok)
local markers = duffle.DELAY_MARKERS
if type(markers) ~= "table" then return false end
-- Identify a leading delay-marker identifier (e.g. `GteDelay_`).
local ident_end = 1
while ident_end <= #tok do
local ch = tok:sub(ident_end, ident_end)
if ch:match("[%w_]") then
ident_end = ident_end + 1
else
break
end
end
local ident = tok:sub(1, ident_end - 1)
if not markers[ident] then return false end
-- Walk the remainder: only whitespace and block comments are allowed.
local scan = ident_end
while scan <= #tok do
local ch = tok:sub(scan, scan)
if ch:match("%s") then
scan = scan + 1
elseif ch == "/" and tok:sub(scan + 1, scan + 1) == "*" then
local close = tok:find("*/", scan + 2, true)
if not close then return false end
scan = close + 2
else
-- Non-whitespace, non-block-comment content: a real instruction
-- follows the marker (e.g. `GteDelay_ nop2`); keep this token intact.
return false
end
end
return true
end
--- Classify a token's "leading comma requirement".
--- Pure delay-marker tokens (`GteDelay_` / `LdSlot_` / `BdSlot_` / `DmaSlot_`
--- followed by whitespace + optional block comment and NOTHING ELSE) expand
--- to empty at C preprocessor time. Emitting them WITHOUT the leading `,`
--- separator that the metaprogram normally adds before each token after the
--- first keeps exactly one `,` between the surrounding real expressions in
--- the spliced macro body:
---
--- * before this rule: `<tok1> ,\t<gdelay> ,\t<tok3>` → after expansion
--- `<tok1> , /* comment */ , <tok3>` → `,,` syntax error.
--- * after this rule: `<tok1> \t<gdelay> ,\t<tok3>` → after expansion
--- `<tok1> /* comment */ , <tok3>` → `<tok1>, <tok3>` — valid.
---
--- Tokens like `GteDelay_ nop2` keep the leading `,` (the marker is followed
--- by a real instruction, so the marker + instruction together need the
--- separator on the LEFT to land between two real expressions).
--- @param tok string
--- @return boolean -- true if the token needs NO leading `,` separator.
local function token_skips_leading_comma(tok)
return is_pure_delay_marker_token(tok)
end
--- Emit the `#define mac_X(sig) \<newline>\t<tok1> \<newline>,\t<tok2> ...` block. --- Emit the `#define mac_X(sig) \<newline>\t<tok1> \<newline>,\t<tok2> ...` block.
--- Converts `//` line comments to `/* */` block comments in each token so they don't break the C macro `\` line continuations. --- Converts `//` line comments to `/* */` block comments in each token so they don't break the C macro `\` line continuations.
---
--- Pure delay-marker tokens (`GteDelay_` / `LdSlot_` / `BdSlot_` / `DmaSlot_` with only a trailing block comment, no real instruction) are emitted WITHOUT a leading `,` separator; the annotation IS preserved in the generated header (so the comment + marker remain visible to anyone reading `gen/macs.h`), but the C preprocessor expands the marker to empty, so leaving the `,` separator out is what stops the `,,` syntax error. See `token_skips_leading_comma` for the contract.
local function emit_macro_body(lines, c, sig, tokens) local function emit_macro_body(lines, c, sig, tokens)
for tok_idx = 1, #tokens do for tok_idx = 1, #tokens do
tokens[tok_idx] = convert_line_comments_to_block(tokens[tok_idx]) tokens[tok_idx] = convert_line_comments_to_block(tokens[tok_idx])
end end
if #tokens == 0 then return end
lines[#lines + 1] = "#define mac_" .. c.name .. "(" .. sig .. ") \\" lines[#lines + 1] = "#define mac_" .. c.name .. "(" .. sig .. ") \\"
lines[#lines + 1] = "\t" .. tokens[1] .. " \\" lines[#lines + 1] = "\t" .. tokens[1] .. " \\"
for tok_idx = 2, #tokens do for tok_idx = 2, #tokens do
lines[#lines + 1] = ",\t" .. tokens[tok_idx] .. " \\" local sep = token_skips_leading_comma(tokens[tok_idx]) and "\t" or ",\t"
lines[#lines + 1] = sep .. tokens[tok_idx] .. " \\"
end end
strip_trailing_continuation(lines) strip_trailing_continuation(lines)
end end
@@ -737,6 +865,7 @@ local function update_canonical_component_body_index(corpus, src, components, sc
source = src.path, source = src.path,
declaration = c.line, declaration = c.line,
kind = c.kind, kind = c.kind,
arg_names = c.arg_names,
} }
end end
end end
+4 -3
View File
@@ -2,7 +2,7 @@
--- ---
--- Reads the post-link ELF directly (io.open; walks the ELF32 section header table to find --- Reads the post-link ELF directly (io.open; walks the ELF32 section header table to find
--- `.debug_info` + `.debug_abbrev` + `.debug_str` + `.debug_line` + `.debug_aranges` + `.debug_rnglists`), --- `.debug_info` + `.debug_abbrev` + `.debug_str` + `.debug_line` + `.debug_aranges` + `.debug_rnglists`),
--- APPENDS synthetic DWARF line-program sequences for every `code_<name>` atom, EXTENDS the `.debug_aranges` --- APPENDS synthetic DWARF line-program sequences for every tape atom, EXTENDS the `.debug_aranges`
--- and main-CU range tables with the atom ranges, and INSERTS synthetic atom/component DIE children into the --- and main-CU range tables with the atom ranges, and INSERTS synthetic atom/component DIE children into the
--- existing main compilation unit in `.debug_info` (no second compilation unit). --- existing main compilation unit in `.debug_info` (no second compilation unit).
--- Per-atom `DW_TAG_subprogram` + per-register `DW_TAG_variable` entries make --- Per-atom `DW_TAG_subprogram` + per-register `DW_TAG_variable` entries make
@@ -1344,7 +1344,7 @@ local function build_new_abbrev()
attr( DW_AT_name, DW_FORM_string) attr( DW_AT_name, DW_FORM_string)
.. attr(DW_AT_low_pc, DW_FORM_addr) .. attr(DW_AT_low_pc, DW_FORM_addr)
.. attr(DW_AT_high_pc, DW_FORM_addr) .. attr(DW_AT_high_pc, DW_FORM_addr)
.. attr(DW_AT_linkage_name, DW_FORM_string)) -- equals DW_AT_name; lets gdb's symbol-table lookup resolve to our subprogram (not the gcc global `code_<name>` const U4 array) .. attr(DW_AT_linkage_name, DW_FORM_string)) -- equals DW_AT_name; gdb resolves the subprogram, not the gcc global array
local abbrev_variable = abbrev(ABBREV_VARIABLE, DW_TAG_variable, false, -- DW_CHILDREN_no local abbrev_variable = abbrev(ABBREV_VARIABLE, DW_TAG_variable, false, -- DW_CHILDREN_no
attr( DW_AT_name, DW_FORM_string) attr( DW_AT_name, DW_FORM_string)
@@ -1857,7 +1857,7 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
end end
-- 4) Emit per-atom DW_TAG_subprograms (children of main CU). -- 4) Emit per-atom DW_TAG_subprograms (children of main CU).
-- Subprogram names match nm symbols without a `code_` prefix. -- Subprogram names match the written C ident (the ELF symbol).
-- The gcc global `<name>[]` is a DW_TAG_variable without children; our subprogram has the wave-context var children. -- The gcc global `<name>[]` is a DW_TAG_variable without children; our subprogram has the wave-context var children.
-- gdb's symbol resolution picks our subprogram (it has low_pc/high_pc + children) over the gcc global for function-context lookups. -- gdb's symbol resolution picks our subprogram (it has low_pc/high_pc + children) over the gcc global for function-context lookups.
for _, atom in ipairs(atom_table) do for _, atom in ipairs(atom_table) do
@@ -2313,5 +2313,6 @@ end
M.compute_loclists_offsets_for_test = compute_loclists_offsets M.compute_loclists_offsets_for_test = compute_loclists_offsets
M.build_debug_loclists_section_for_test = build_debug_loclists_section M.build_debug_loclists_section_for_test = build_debug_loclists_section
M.tape_piece_size_for_test = tape_piece_size M.tape_piece_size_for_test = tape_piece_size
M.build_atom_table_for_test = build_atom_table
return M return M
+21 -1
View File
@@ -155,8 +155,28 @@ local function project_atom(atom_record, src, corpus)
local body = atom_record.body or "" local body = atom_record.body or ""
local wc = corpus.word_counts or {} local wc = corpus.word_counts or {}
local cbi = corpus.component_body_index or {} local cbi = corpus.component_body_index or {}
local schema = nil
if atom_record.reg_use_schema_name then
schema = corpus.reg_use_schemas and corpus.reg_use_schemas[atom_record.reg_use_schema_name]
end
-- That construction site stamps `invocation.debug_skip` while appending each record to `proj.invocations`. -- That construction site stamps `invocation.debug_skip` while appending each record to `proj.invocations`.
local proj = duffle.project_emission(body, cbi, wc, corpus.components) local proj = duffle.project_emission(body, cbi, wc, corpus.components, {
reg_use_schema = schema,
reg_use_param = atom_record.reg_use_param_name,
atom_name = atom_record.name,
schema_name = atom_record.reg_use_schema_name,
})
if atom_record.reg_use_schema_name and not schema then
proj.errors[#proj.errors + 1] = {
kind = "reguse_missing_schema",
msg = string.format("RegUse schema %q is missing", atom_record.reg_use_schema_name),
}
end
for _, err in ipairs(corpus.reg_use_errors or {}) do
if err.schema_name == atom_record.reg_use_schema_name then
proj.errors[#proj.errors + 1] = err
end
end
local paths = { local paths = {
tokens = atom_record.body_tokens or {}, tokens = atom_record.body_tokens or {},
line_in_body = duffle.build_body_line_index(body), line_in_body = duffle.build_body_line_index(body),
+2 -1
View File
@@ -1,7 +1,8 @@
--- passes/offsets.lua — Branch-offset generator. --- passes/offsets.lua — Branch-offset generator.
--- ---
--- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`) --- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`)
--- for `MipsAtom_(name)` and `MipsCode code_<name>` declarations, computes the word offset --- for `MipsAtom_(name)` and leftover `MipsCode code_*` declarations, computes the word offset
--- (ELF symbol is the C ident; raw `code_*` is leftover, not the atom rule)
--- from each `atom_offset(F, T)` marker to its target `atom_label(T)` declaration, and emits --- from each `atom_offset(F, T)` marker to its target `atom_label(T)` declaration, and emits
--- `gen/offsets.h` with one `#define _atom_offset_F_T = N` per branch. --- `gen/offsets.h` with one `#define _atom_offset_F_T = N` per branch.
--- ---
+578 -227
View File
@@ -4,8 +4,8 @@
--- - `build/gen/<dir_basename>.annotations.txt` — one per source-directory containing atoms; aggregates across all sources in the directory. --- - `build/gen/<dir_basename>.annotations.txt` — one per source-directory containing atoms; aggregates across all sources in the directory.
--- - `build/gen/annotation_validation.txt` — the project summary. --- - `build/gen/annotation_validation.txt` — the project summary.
--- ---
--- The annotation pass emits `errors.h` files per module and the canonical `corpus.sources_by_dir` projection groups sources by directory. --- The canonical `corpus.sources_by_dir` projection groups sources by directory.
--- This pass iterates the dir projection directly and re-validates each source via `annotation.validate()` to get the detailed per-source results. --- This pass builds one ModuleView per directory and walks SECTION_RENDERERS.
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Module-scope requires + package.path setup -- Module-scope requires + package.path setup
@@ -20,11 +20,6 @@
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
-- Load the annotation pass so we can re-validate each source against the canonical corpus projection.
-- The annotation pass exposes `M.validate`, which returns the per-source AnnotationResult (atoms / annots / macros / binds / errors / warnings)
-- that the report pass renders into the per-module `<dir_basename>.annotations.txt` output.
local annotation = dofile(_bootstrap_dir .. "annotation.lua")
-- Load atoms_source_map for the `render_source_map` / `render_provenance` module functions (used by `render_module_atoms_md` to produce `<module>.atoms.md` without re-walking source tokens). -- Load atoms_source_map for the `render_source_map` / `render_provenance` module functions (used by `render_module_atoms_md` to produce `<module>.atoms.md` without re-walking source tokens).
-- The pass itself emits no per-source files anymore; we only consume the two pure renderers here. -- The pass itself emits no per-source files anymore; we only consume the two pure renderers here.
-- Defined BEFORE the renderer functions below so their upvalues resolve to this local (not the global `atoms_source_map`, which is nil). -- Defined BEFORE the renderer functions below so their upvalues resolve to this local (not the global `atoms_source_map`, which is nil).
@@ -225,7 +220,7 @@ local function render_module_atoms_md(dir, dir_sources, wc)
for _, atom in ipairs(atoms_list) do for _, atom in ipairs(atoms_list) do
lines[#lines + 1] = string.format( lines[#lines + 1] = string.format(
"### atom: %s (line %d, %d words)", "### atom: %s (line %d, %d words)",
atom.name, atom.line or 0, #(atom.paths.items or {})) atom.name, atom.line or 0, #((atom.paths or {}).word_events or {}))
lines[#lines + 1] = "" lines[#lines + 1] = ""
lines[#lines + 1] = "**Sourcemap** — per-word call site:" lines[#lines + 1] = "**Sourcemap** — per-word call site:"
lines[#lines + 1] = "```" lines[#lines + 1] = "```"
@@ -246,17 +241,544 @@ local function render_module_atoms_md(dir, dir_sources, wc)
return table.concat(lines, "\n") .. "\n" return table.concat(lines, "\n") .. "\n"
end end
--- Render the consolidated per-module markdown (`build/<module>.atom_meta_report.md`). local function decl_words(atom)
--- Aggregates annotation + static-analysis content across all sources in `dir`. local p = atom.paths or {}
--- Annotations come from re-running `annotation.validate()` per source (the existing pattern); return #(p.word_events or {})
--- static-analysis comes from `corpus.static_analysis_results[dir_basename]` (populated by `static_analysis.lua` — no second corpus_pipe_ctx build). end
--- @param dir string
--- @param dir_sources SourceFile[] local function count_kinds(decls)
--- @param annot_results AnnotationResult[] local n = { atom = 0, atom_proc = 0, comp_bare = 0, comp_proc = 0 }
--- @param sa_results table -- corpus.static_analysis_results[dir_basename] for _, a in ipairs(decls or {}) do
--- @return string if n[a.kind] ~= nil then n[a.kind] = n[a.kind] + 1 end
local function render_module_meta_report(dir, dir_sources, annot_results, sa_results) end
return n
end
local function slot_suffix(key)
if type(key) ~= "string" or key:sub(1, 7) ~= "reguse:" then return nil end
return key:match("([^:]+)$")
end
local function decl_names(view)
local names = {}
for _, a in ipairs(view.decls or {}) do
if a.name then names[a.name] = true end
end
return names
end
local function path_in_module(path, view)
if type(path) ~= "string" or path == "" then return false end
local norm = path:gsub("\\", "/")
local dir = (view.dir or ""):gsub("\\", "/")
if dir ~= "" and (norm == dir or norm:sub(1, #dir + 1) == dir .. "/") then
return true
end
for _, src in ipairs(view.sources or {}) do
if (src.path or ""):gsub("\\", "/") == norm then return true end
end
return false
end
local function build_module_view(dir, dir_sources, corpus)
local decls = {}
for _, src in ipairs(dir_sources or {}) do
for _, a in ipairs((src.scan and src.scan.atoms) or {}) do
if not a.source_path then a.source_path = src.path end
decls[#decls + 1] = a
end
end
local dir_basename = source_basename(dir) local dir_basename = source_basename(dir)
local sa = (corpus.static_analysis_results or {})[dir_basename] or {}
local schemas = {}
for name, schema in pairs(corpus.reg_use_schemas or {}) do
for _, a in ipairs(decls) do
if a.reg_use_schema_name == name then
schemas[#schemas + 1] = schema
break
end
end
end
return {
dir = dir,
sources = dir_sources or {},
decls = decls,
schemas = schemas,
findings = sa.findings or {},
sa = sa,
corpus = corpus,
}
end
local function render_section_declarations(add, view)
if #view.decls == 0 then add("_(none)_"); add(""); return end
add("| kind | name | source | line | words | min | max | branches | paths |")
add("|------|------|--------|------|-------|-----|-----|----------|-------|")
for _, a in ipairs(view.decls) do
local p = a.paths or {}
add(string.format("| %s | %s | %s | %d | %d | %s | %s | %s | %s |",
a.kind or "?",
a.name or "?",
source_basename(a.source_path or ""),
a.line or 0,
decl_words(a),
tostring(p.cycles_min or ""),
tostring(p.cycles_max or ""),
tostring(p.branches or ""),
tostring(p.paths or "")))
end
add("")
end
local function render_section_components(add, view)
local rows = {}
local index = (view.corpus and view.corpus.component_body_index) or {}
for _, a in ipairs(view.decls) do
if a.kind == "comp_bare" or a.kind == "comp_proc" then
local idx = index[a.name] or {}
local args = idx.arg_names or {}
rows[#rows + 1] = {
name = a.name,
kind = a.kind,
args = table.concat(args, ", "),
words = decl_words(a),
map = a.map_command or "",
}
end
end
if #rows == 0 then add("_(none)_"); add(""); return end
add("| name | kind | arg_names | words | map |")
add("|------|------|-----------|-------|-----|")
for _, r in ipairs(rows) do
add(string.format("| %s | %s | %s | %d | %s |",
r.name, r.kind, r.args ~= "" and r.args or "", r.words, r.map))
end
add("")
end
local function render_section_reguse(add, view)
local wrote = false
for _, schema in ipairs(view.schemas or {}) do
wrote = true
add(string.format("### %s", schema.name or "?"))
for _, slot in ipairs(schema.slots or {}) do
local aliases = table.concat(slot.aliases or { slot.name }, ", ")
local ro = slot.readonly and " readonly" or ""
add(string.format("- slot `%s` aliases %s%s", slot.name, aliases, ro))
end
for _, a in ipairs(view.decls) do
if a.reg_use_schema_name == schema.name then
add(string.format("- bound `%s` param `%s`", a.name, a.reg_use_param_name or "?"))
end
end
add("")
end
local bound = {}
for _, schema in ipairs(view.schemas or {}) do
if schema.name then bound[schema.name] = true end
end
local errors = {}
for _, err in ipairs((view.corpus and view.corpus.reg_use_errors) or {}) do
if bound[err.schema_name] or path_in_module(err.source_file, view) then
errors[#errors + 1] = err
end
end
if #errors > 0 then
wrote = true
add("### parse errors")
for _, err in ipairs(errors) do
add(string.format("- `%s` %s", err.kind or "?", err.schema_name or ""))
end
add("")
end
if not wrote then add("_(none)_"); add("") end
end
local function render_section_annotations(add, view)
local rows = {}
for _, src in ipairs(view.sources) do
for _, info in ipairs((src.scan and src.scan.atom_infos) or {}) do
rows[#rows + 1] = {
source = source_basename(src.path),
line = info.info_line or 0,
name = info.atom_name or "?",
binds = info.binds or "",
reads = (#(info.reads or {}) > 0 and table.concat(info.reads, ",")) or "",
writes = (#(info.writes or {}) > 0 and table.concat(info.writes, ",")) or "",
phase = info.phase or "",
}
end
end
if #rows == 0 then add("_(none)_"); add(""); return end
add("| source | line | name | binds | reads | writes | phase |")
add("|--------|------|------|-------|-------|--------|-------|")
for _, r in ipairs(rows) do
add(string.format("| %s | %d | %s | %s | %s | %s | %s |",
r.source, r.line, r.name, r.binds, r.reads, r.writes, r.phase))
end
add("")
end
local function render_section_component_annotations(add, view)
local rows = {}
for _, src in ipairs(view.sources) do
for _, info in ipairs((src.scan and src.scan.component_atom_infos) or {}) do
rows[#rows + 1] = {
source = source_basename(src.path),
line = info.info_line or 0,
name = info.atom_name or "?",
reads = (#(info.reads or {}) > 0 and table.concat(info.reads, ",")) or "",
writes = (#(info.writes or {}) > 0 and table.concat(info.writes, ",")) or "",
}
end
end
if #rows == 0 then add("_(none)_"); add(""); return end
add("| source | line | name | reads | writes |")
add("|--------|------|------|-------|--------|")
for _, r in ipairs(rows) do
add(string.format("| %s | %d | %s | %s | %s |",
r.source, r.line, r.name, r.reads, r.writes))
end
add("")
end
local function render_section_binds(add, view)
local wrote = false
for _, src in ipairs(view.sources) do
for _, b in ipairs((src.scan and src.scan.binds) or {}) do
wrote = true
add(string.format("### %s (%s:%s, %s bytes)",
b.name, source_basename(src.path), tostring(b.line or 0), tostring(b.bytes or "")))
for _, f in ipairs(b.fields or {}) do
add(string.format("- `+%s %s`", tostring(f.offset or "?"), f.name or "?"))
end
add("")
end
end
if not wrote then add("_(none)_"); add("") end
end
local function render_section_phases(add, view)
local corpus = view.corpus or {}
local names = decl_names(view)
local wrote = false
for phase, entry in pairs(corpus.atom_phases or {}) do
local here = {}
for _, atom_name in ipairs(entry.atoms or {}) do
if names[atom_name] then here[#here + 1] = atom_name end
end
if #here > 0 then
wrote = true
add(string.format("- phase `%s`: %s", phase, table.concat(here, ", ")))
end
end
for name, entry in pairs(corpus.atom_views or {}) do
if names[name] then
wrote = true
add(string.format("- view `%s` binds `%s`", name, entry.binds_name or ""))
end
end
for name, entry in pairs(corpus.atom_ctxs or {}) do
if names[name] then
wrote = true
add(string.format("- ctx `%s` rbind `%s`", name, entry.rbind_atom or ""))
end
end
if not wrote then add("_(none)_") end
add("")
end
local function render_section_aliases(add, view)
local names = {}
local seen = {}
for _, src in ipairs(view.sources or {}) do
for name, entry in pairs((src.scan and src.scan.register_alias_registry) or {}) do
if not seen[name] then
seen[name] = entry
names[#names + 1] = name
end
end
end
table.sort(names)
if #names == 0 then add("_(none)_"); add(""); return end
add("| alias | type |")
add("|-------|------|")
for _, name in ipairs(names) do
local e = seen[name]
add(string.format("| %s | %s |", name, (e and e.default_type) or ""))
end
add("")
end
local function render_section_autoreg(add, view)
local allowed = decl_names(view)
for phase, entry in pairs((view.corpus and view.corpus.atom_phases) or {}) do
for _, atom_name in ipairs(entry.atoms or {}) do
if allowed[atom_name] then allowed[phase] = true end
end
end
local wrote = false
local seen = {}
local function dump(label, table_map)
local scopes = {}
for scope in pairs(table_map or {}) do
if allowed[scope] and not seen[label .. "\0" .. scope] then
scopes[#scopes + 1] = scope
end
end
table.sort(scopes)
for _, scope in ipairs(scopes) do
seen[label .. "\0" .. scope] = true
wrote = true
local syms = {}
for sym, gpr in pairs(table_map[scope] or {}) do
if type(gpr) == "string" and gpr ~= sym then
syms[#syms + 1] = string.format("%s → %s", sym, gpr)
else
syms[#syms + 1] = tostring(sym)
end
end
table.sort(syms)
add(string.format("- %s `%s`: %s", label, scope, table.concat(syms, ", ")))
end
end
local corpus = view.corpus or {}
dump("atom", corpus.atom_auto_regs)
dump("phase", corpus.phase_auto_regs)
for _, src in ipairs(view.sources or {}) do
dump("atom", src.scan and src.scan.atom_auto_regs)
dump("phase", src.scan and src.scan.phase_auto_regs)
end
if not wrote then add("_(none)_") end
add("")
end
local function render_section_collisions(add, view)
local rows = {}
for _, c in ipairs((view.corpus and view.corpus.collisions) or {}) do
local first = c.first_site or {}
local other = c.conflicting_site or {}
if path_in_module(first.path, view) or path_in_module(other.path, view) then
rows[#rows + 1] = c
end
end
if #rows == 0 then add("_(none)_"); add(""); return end
for _, c in ipairs(rows) do
local first = c.first_site or {}
local other = c.conflicting_site or {}
add(string.format("- `%s` `%s` first %s:%s conflict %s:%s",
c.kind or "?", c.name or "?",
tostring(first.path or "?"), tostring(first.line or "?"),
tostring(other.path or "?"), tostring(other.line or "?")))
end
add("")
end
local function render_section_findings(add, view)
local by_atom = {}
for _, f in ipairs(view.findings or {}) do
local key = f.atom or "?"
by_atom[key] = by_atom[key] or {}
by_atom[key][#by_atom[key] + 1] = f
end
if next(by_atom) == nil then add("_(none)_"); add(""); return end
local seen = {}
local function emit(name, fs)
add("### " .. name)
for _, f in ipairs(fs) do
local msg = f.msg or ""
local slot = slot_suffix(f.gpr_key or f.producer_destination)
if slot and not msg:find("(slot ", 1, true) then
msg = msg .. " (slot " .. slot .. ")"
end
add(string.format("- `[%s/%s] %s`", f.kind or "info", f.check or "?", msg))
end
add("")
end
for _, a in ipairs(view.decls) do
if by_atom[a.name] then
seen[a.name] = true
emit(a.name, by_atom[a.name])
end
end
local leftovers = {}
for name in pairs(by_atom) do
if not seen[name] then leftovers[#leftovers + 1] = name end
end
table.sort(leftovers)
for _, name in ipairs(leftovers) do emit(name, by_atom[name]) end
end
local function render_section_relations(add, view)
local wrote = false
for _, a in ipairs(view.decls) do
local rels = (a.paths and a.paths.relations) or {}
if #rels > 0 then
wrote = true
add("### " .. a.name)
for _, rel in ipairs(rels) do
local dest = rel.destination or rel.producer_destination or ""
local slot = slot_suffix(dest)
local dest_s = tostring(dest)
if slot then dest_s = dest_s .. " (slot " .. slot .. ")" end
add(string.format("- `%s` words %s → %s dest %s",
rel.semantic or "?",
tostring(rel.producer_word or "?"),
tostring(rel.consumer_word or "?"),
dest_s))
end
add("")
end
end
if not wrote then add("_(none)_"); add("") end
end
local HIDDEN_UNLESS_WRITTEN = {
R_AT = true, R_TapePtr = true, R_AtomJmp = true,
}
local PHYSICAL_GPR = {
R_T0 = true, R_T1 = true, R_T2 = true, R_T3 = true,
R_T4 = true, R_T5 = true, R_T6 = true, R_T7 = true,
R_V0 = true, R_V1 = true,
}
local function encoder_wrote_key(atom, key)
for _, ev in ipairs((atom.paths and atom.paths.word_events) or {}) do
for _, dest in pairs(ev.gpr_keys or {}) do
if dest == key then return true end
end
end
return false
end
local function written_name_for(key, atom)
local slot = key:match("^reguse:.+:(.+)$")
if slot then
local param = atom.reg_use_param_name
if param and param ~= "" then return param .. "." .. slot end
return slot
end
return key
end
local function aliases_for_key(key, atom, view)
local slot = key:match("^reguse:.+:(.+)$")
if not slot then return "" end
local schema_name = atom.reg_use_schema_name
local schema = view.corpus and view.corpus.reg_use_schemas and view.corpus.reg_use_schemas[schema_name]
if not schema then return "" end
for _, s in ipairs(schema.slots or {}) do
if s.name == slot then
local names = {}
for _, alias in ipairs(s.aliases or {}) do
if alias ~= slot then names[#names + 1] = alias end
end
if #names == 0 then
if s.aliases and #s.aliases > 0 then return table.concat(s.aliases, ", ") end
return ""
end
return table.concat(names, ", ")
end
end
return ""
end
local function physical_for_key(key, atom, view)
if PHYSICAL_GPR[key] then return key end
local corpus = view.corpus or {}
local alias = (corpus.register_alias_registry or {})[key]
if type(alias) == "table" then
local phys = alias.physical or alias.gpr or alias.code_name
if type(phys) == "string" and PHYSICAL_GPR[phys] then return phys end
if type(alias.name) == "string" and PHYSICAL_GPR[alias.name] then return alias.name end
elseif type(alias) == "string" and PHYSICAL_GPR[alias] then
return alias
end
local atom_map = (corpus.atom_auto_regs or {})[atom.name]
if type(atom_map) == "table" then
local slot = key:match("^reguse:.+:(.+)$") or key
local bound = atom_map[slot] or atom_map["R_" .. slot]
if type(bound) == "string" and PHYSICAL_GPR[bound] then return bound end
end
return ""
end
local function last_relation_for(key, atom)
local last = nil
for _, rel in ipairs((atom.paths and atom.paths.relations) or {}) do
local dest = rel.destination or rel.producer_destination
if dest == key then last = rel end
end
if not last then return "" end
local sem = last.semantic or "?"
local a = last.producer_word
local b = last.consumer_word
if a and b then return string.format("%s w%s→%s", sem, tostring(a), tostring(b)) end
return sem
end
local function render_section_forward(add, view)
local wrote = false
for _, a in ipairs(view.decls) do
local gpr = a.paths and a.paths.forward_state and a.paths.forward_state.gpr_values
local keys = {}
for k in pairs(gpr or {}) do
if k == "R_0" then
-- hidden
elseif HIDDEN_UNLESS_WRITTEN[k] and not encoder_wrote_key(a, k) then
-- hidden
else
keys[#keys + 1] = k
end
end
if #keys > 0 then
wrote = true
add("### " .. a.name)
add("| written | aliases | physical | lattice | last relation |")
add("|---|---|---|---|---|")
table.sort(keys)
for _, k in ipairs(keys) do
local slot = gpr[k]
local lattice = ""
if slot and slot.kind == "constant" then
lattice = tostring(slot.value)
end
add(string.format("| `%s` | %s | %s | %s | %s |",
written_name_for(k, a),
aliases_for_key(k, a, view),
physical_for_key(k, a, view),
lattice,
last_relation_for(k, a)))
end
add("")
end
end
if not wrote then add("_(none)_"); add("") end
end
local SECTION_RENDERERS = {
{ header = "## Declarations", render = render_section_declarations },
{ header = "## Components", render = render_section_components },
{ header = "## RegUse schemas", render = render_section_reguse },
{ header = "## Annotations", render = render_section_annotations },
{ header = "## Component annotations", render = render_section_component_annotations },
{ header = "## Binds_* structs", render = render_section_binds },
{ header = "## Phases / views / ctx", render = render_section_phases },
{ header = "## Register aliases", render = render_section_aliases },
{ header = "## Auto-reg", render = render_section_autoreg },
{ header = "## Collisions", render = render_section_collisions },
{ header = "## Findings", render = render_section_findings },
{ header = "## Relations", render = render_section_relations },
{ header = "## GPR model", render = render_section_forward },
}
--- Render the consolidated per-module markdown (`build/<module>.atom_meta_report.md`).
--- One ModuleView from the corpus; SECTION_RENDERERS walks it.
--- @param view table
--- @return string
local function render_module_meta_report(view)
local dir_basename = source_basename(view.dir)
local lines = { local lines = {
"# " .. dir_basename .. " — atom meta report", "# " .. dir_basename .. " — atom meta report",
"> Auto-generated by ps1_meta.lua (passes/report.lua). Do not edit.", "> Auto-generated by ps1_meta.lua (passes/report.lua). Do not edit.",
@@ -264,199 +786,41 @@ local function render_module_meta_report(dir, dir_sources, annot_results, sa_res
} }
local function add(s) lines[#lines + 1] = s end local function add(s) lines[#lines + 1] = s end
-- Module summary table. local kinds = count_kinds(view.decls)
local n_atoms = 0 local n_annot, n_binds, n_macros = 0, 0, 0
local n_annot = 0 for _, src in ipairs(view.sources) do
local n_binds = 0 n_annot = n_annot + #((src.scan and src.scan.atom_infos) or {})
local n_macros = 0 n_binds = n_binds + #((src.scan and src.scan.binds) or {})
local n_bare, n_proc = 0, 0 n_macros = n_macros + #((src.scan and src.scan.macros) or {})
for _, r in ipairs(annot_results) do
n_atoms = n_atoms + #r.atoms
n_annot = n_annot + #r.annots
n_binds = n_binds + #r.binds
n_macros = n_macros + #r.macros
end end
for _, a in ipairs(sa_results.atoms or {}) do local n_err, n_warn, n_info = 0, 0, 0
if a.kind == "comp_bare" then n_bare = n_bare + 1 for _, f in ipairs(view.findings or {}) do
elseif a.kind == "comp_proc" then n_proc = n_proc + 1 if f.kind == "error" then n_err = n_err + 1
elseif f.kind == "warning" then n_warn = n_warn + 1
else n_info = n_info + 1
end end
end end
add("## Module summary"); add("") add("## Module summary"); add("")
add("| metric | value |"); add("|--------|-------|") add("| metric | value |"); add("|--------|-------|")
add(string.format("| sources | %d |", #dir_sources)) add(string.format("| sources | %d |", #view.sources))
add(string.format("| atoms | %d (atoms: %d, comp_bare: %d, comp_proc: %d) |", add(string.format("| decls | %d (atom: %d, atom_proc: %d, comp_bare: %d, comp_proc: %d) |",
#(sa_results.atoms or {}), #view.decls, kinds.atom, kinds.atom_proc, kinds.comp_bare, kinds.comp_proc))
#(sa_results.atoms or {}) - n_bare - n_proc, n_bare, n_proc))
add(string.format("| annotations | %d |", n_annot)) add(string.format("| annotations | %d |", n_annot))
add(string.format("| binds structs | %d |", n_binds)) add(string.format("| binds structs | %d |", n_binds))
add(string.format("| macro decls | %d |", n_macros)) add(string.format("| macro decls | %d |", n_macros))
add(string.format("| findings | %d (errors: %d, warnings: %d, info: %d) |", add(string.format("| findings | %d (errors: %d, warnings: %d, info: %d) |",
#(sa_results.findings or {}), #(view.findings or {}), n_err, n_warn, n_info))
#(sa_results.errors or {}),
#(sa_results.warnings or {}),
#(sa_results.info or {})))
add("") add("")
-- Sources
add("## Sources"); add("") add("## Sources"); add("")
for _, s in ipairs(dir_sources) do add("- `" .. s.path .. "`") end for _, s in ipairs(view.sources) do add("- `" .. s.path .. "`") end
add("") add("")
-- Atoms (annotation) for _, row in ipairs(SECTION_RENDERERS) do
add("## Atoms"); add("") add(row.header); add("")
add("| kind | name | source | line |"); add("|------|------|--------|------|") row.render(add, view)
for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source)
for _, a in ipairs(r.atoms) do
add(string.format("| atom | %s | %s | %d |", a.name, src_name, a.line))
end end
end
add("")
-- Annotations
add("## Annotations"); add("")
if #annot_results == 0 then
add("_(none)_")
else
add("| source | line | name | binds | reads | writes |")
add("|--------|------|------|-------|-------|--------|")
for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source)
for _, a in ipairs(r.annots) do
local binds = a.binds or ""
local reads = (#a.reads > 0 and table.concat(a.reads, ",")) or ""
local writes = (#a.writes > 0 and table.concat(a.writes, ",")) or ""
add(string.format("| %s | %d | %s | %s | %s | %s |"
, src_name, a.line, a.name, binds, reads, writes))
end
end
end
add("")
-- Binds_* structs
add("## Binds_* structs"); add("")
if #annot_results == 0 then
add("_(none)_")
else
for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source)
for _, b in ipairs(r.binds) do
add(string.format("### %s (%s:%d, %d bytes)",
b.name, src_name, b.line, b.bytes))
for _, f in ipairs(b.fields) do
add(string.format("- `+%d %s`", f.offset, f.name))
end
add("")
end
end
end
-- Macro decls
add("## Macro word-count declarations"); add("")
if #annot_results == 0 then
add("_(none)_")
else
add("| source | line | macro declaration |")
add("|--------|------|-------------------|")
for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source)
for _, m in ipairs(r.macros) do
add(string.format("| %s | %d | %s |",
src_name, m.line, m.name))
end
end
end
add("")
-- Findings by atom (static-analysis)
add("## Static analysis — findings by atom"); add("")
local by_atom = {}
for _, f in ipairs(sa_results.findings or {}) do
by_atom[f.atom] = by_atom[f.atom] or {}
by_atom[f.atom][#by_atom[f.atom] + 1] = f
end
if next(by_atom) == nil then
add("_(no findings)_")
else
for _, a in ipairs(sa_results.atoms or {}) do
local fs = by_atom[a.name]
if fs then
add(string.format("### %s", a.name))
for _, f in ipairs(fs) do
add(string.format("- `[%s] %s`", f.check, f.msg))
end
add("")
end
end
end
-- Errors / Warnings / Info
local function add_findings(label, entries)
add(string.format("## %s", label))
if #entries == 0 then
add("_(none)_")
else
for _, e in ipairs(entries) do
add(string.format("- line %d %s", e.line, e.msg))
end
end
add("")
end
add_findings("Errors", sa_results.errors or {})
add_findings("Warnings", sa_results.warnings or {})
add_findings("Info", sa_results.info or {})
-- Per-atom cycle counts (path-aware)
add("## Per-atom cycle counts (path-aware, best case, no stalls)"); add("")
add("| atom | source | min | max | branches | paths | notes |")
add("|------|--------|-----|-----|----------|-------|-------|")
local sorted = {}
for _, a in ipairs(sa_results.atoms or {}) do sorted[#sorted + 1] = a end
table.sort(sorted, function(x, y)
return ((x.paths or {}).cycles_max or 0) > ((y.paths or {}).cycles_max or 0)
end)
for _, a in ipairs(sorted) do
local p = a.paths or {}
local src_name = a.source_path and source_basename(a.source_path) or ""
local notes = ""
if p.has_loops then notes = notes .. " [loop!]" end
if p.unknown_macros and #p.unknown_macros > 0 then
notes = notes .. " [unknown: " .. table.concat(p.unknown_macros, ", ") .. "]"
end
add(string.format("| %s | %s | %d | %d | %d | %d | %s |",
a.name, src_name,
p.cycles_min or 0, p.cycles_max or 0,
p.branches or 0, p.paths or 0, notes))
end
add("")
-- Per-source scan summary
add("## Per-source scan summary"); add("")
for _, src in ipairs(dir_sources) do
local src_atoms = {}
for _, a in ipairs(sa_results.atoms or {}) do
if a.source_path == src.path then src_atoms[#src_atoms + 1] = a end
end
if #src_atoms > 0 then
local mn, mx = math.huge, -1
for _, a in ipairs(src_atoms) do
local p = a.paths or {}
if (p.cycles_min or 0) < mn then mn = p.cycles_min or 0 end
if (p.cycles_max or 0) > mx then mx = p.cycles_max or 0 end
end
local path_str
if mx > 0 then
path_str = string.format(" cycles=%d..%d", mn, mx)
else
path_str = string.format(" %d cycles", mn)
end
add(string.format("- `%s` — %d atom%s%s",
src.basename, #src_atoms,
#src_atoms == 1 and "" or "s", path_str))
end
end
add("")
return table.concat(lines, "\n") .. "\n" return table.concat(lines, "\n") .. "\n"
end end
@@ -474,19 +838,8 @@ local REPORT_RENDERERS = {
basename = function(dir_basename) return dir_basename .. ".atom_meta_report" end, basename = function(dir_basename) return dir_basename .. ".atom_meta_report" end,
once = false, once = false,
gather = function(ctx, dir, dir_sources) gather = function(ctx, dir, dir_sources)
-- Annotations: re-run `annotation.validate()` per source (the existing pattern). local corpus = ctx.shared.corpus
local annot_results = {} return render_module_meta_report(build_module_view(dir, dir_sources, corpus))
for _, src in ipairs(dir_sources) do
if src.scan then
local r = annotation.validate(ctx, src, nil)
r.source = src.path
annot_results[#annot_results + 1] = r
end
end
-- Static-analysis: read stashed projection (no re-validate).
local dir_basename = dir:match("([^/\\]+)$") or dir
local sa_results = (ctx.shared.corpus.static_analysis_results or {})[dir_basename] or {}
return render_module_meta_report(dir, dir_sources, annot_results, sa_results)
end, end,
}, },
{ {
@@ -554,32 +907,30 @@ function M.run(ctx)
end end
end end
-- For the summary, compute per-module totals once (re-validating annotations per source — same pattern as the meta_report renderer). local view = build_module_view(dir, dir_sources, corpus)
local annot_results = {}
for _, src in ipairs(dir_sources) do
if src.scan then
local r = annotation.validate(ctx, src, nil)
r.source = src.path
annot_results[#annot_results + 1] = r
end
end
local n_annot, n_binds, n_macros = 0, 0, 0 local n_annot, n_binds, n_macros = 0, 0, 0
for _, r in ipairs(annot_results) do for _, src in ipairs(dir_sources) do
n_annot = n_annot + #r.annots n_annot = n_annot + #((src.scan and src.scan.atom_infos) or {})
n_binds = n_binds + #r.binds n_binds = n_binds + #((src.scan and src.scan.binds) or {})
n_macros = n_macros + #r.macros n_macros = n_macros + #((src.scan and src.scan.macros) or {})
end
local n_err, n_warn, n_info = 0, 0, 0
for _, f in ipairs(view.findings or {}) do
if f.kind == "error" then n_err = n_err + 1
elseif f.kind == "warning" then n_warn = n_warn + 1
else n_info = n_info + 1
end
end end
local sa_results = (corpus.static_analysis_results or {})[dir_basename] or {}
all_modules[#all_modules + 1] = { all_modules[#all_modules + 1] = {
module = dir_basename, module = dir_basename,
atoms = #(sa_results.atoms or {}), atoms = #view.decls,
annots = n_annot, annots = n_annot,
binds = n_binds, binds = n_binds,
macros = n_macros, macros = n_macros,
findings = #(sa_results.findings or {}), findings = #(view.findings or {}),
errors = #(sa_results.errors or {}), errors = n_err,
warnings = #(sa_results.warnings or {}), warnings = n_warn,
info = #(sa_results.info or {}), info = n_info,
} }
end end
+701 -134
View File
@@ -6,6 +6,7 @@
--- MipsAtom_Proc_ (kind = "atom_proc", body inside last {}) --- MipsAtom_Proc_ (kind = "atom_proc", body inside last {})
--- MipsAtomComp_ (kind = "comp_bare") --- MipsAtomComp_ (kind = "comp_bare")
--- MipsAtomComp_Proc_ (kind = "comp_proc", body inside last {}) --- MipsAtomComp_Proc_ (kind = "comp_proc", body inside last {})
--- MipsAtomComp_ProcMap_ (kind = "comp_proc", body is the one command)
--- atom_dbg_skip — bare whole-atom/component debug-step marker; following declaration disambiguates --- atom_dbg_skip — bare whole-atom/component debug-step marker; following declaration disambiguates
--- MipsCode code_<name> (kind = "raw_atom", offsets pass only) --- MipsCode code_<name> (kind = "raw_atom", offsets pass only)
--- typedef Struct_(Binds_X) { fields } --- typedef Struct_(Binds_X) { fields }
@@ -137,6 +138,16 @@ local QUALIFIER_KEYWORDS = {
local AC_PREFIX = "ac_" local AC_PREFIX = "ac_"
local AC_PREFIX_LEN = 3 local AC_PREFIX_LEN = 3
-- The function-decl keyword that precedes a MipsAtomComp_Proc_ call.
-- Used by the backward walk in duffle.find_function_decl_for.
local SLICE_MIPS_CODE = "Slice_MipsCode"
local SLICE_MIPS_CODE_LEN = #SLICE_MIPS_CODE
-- The return type that precedes a MipsAtom_Proc_ function declaration.
-- Used by the backward walk in duffle.find_atom_proc_decl_for.
local MIPS_ATOM_PTR = "MipsAtom*"
local MIPS_ATOM_PTR_LEN = #MIPS_ATOM_PTR
--- Strip the "ac_" prefix from a component name. --- Strip the "ac_" prefix from a component name.
--- Returns the input unchanged if it doesn't start with the prefix. --- Returns the input unchanged if it doesn't start with the prefix.
--- @param raw_name string --- @param raw_name string
@@ -405,31 +416,51 @@ local function walk_body_fields(body, build_field)
return fields return fields
end end
-- Parse the `<type> <field>;` declarations from a Struct_ body. -- Parse the `<type> <field>[, <field>...];` declarations from a Struct_ body.
-- After the type and `*` chain, keep reading `, ident` until `;`.
-- Same type, same pointer depth for every name on that list.
-- Returns the raw fields array with `{name, type_name, pointer_depth}` only (NO offset / byte_size). -- Returns the raw fields array with `{name, type_name, pointer_depth}` only (NO offset / byte_size).
-- The propagation pass `resolve_struct_field_sizes` walks each struct's fields AFTER type resolution and populates offset + byte_size in place. -- The propagation pass `resolve_struct_field_sizes` walks each struct's fields AFTER type resolution and populates offset + byte_size in place.
-- Returns (fields). The aggregate byte_count is computed in the propagation pass (it depends on whether every field's type resolved).
local function parse_struct_body_fields(body) local function parse_struct_body_fields(body)
return walk_body_fields(body, function(type_name, type_end, after_type) local fields = {}
-- Parse the trailing `*` chain to derive pointer_depth. local body_pos = 1
local depth, cursor = 0, after_type local body_len = #body
while cursor <= #body and body:sub(cursor, cursor) == "*" do while body_pos <= body_len do
body_pos = duffle.skip_ws_and_cmt(body, body_pos)
if body_pos > body_len then break end
local type_name, type_end = duffle.read_ident(body, body_pos)
if not type_name then
body_pos = body_pos + 1
else
local depth, cursor = 0, duffle.skip_ws_and_cmt(body, type_end)
while cursor <= body_len and body:sub(cursor, cursor) == "*" do
depth = depth + 1 depth = depth + 1
cursor = cursor + 1 cursor = duffle.skip_ws_and_cmt(body, cursor + 1)
cursor = duffle.skip_ws_and_cmt(body, cursor)
end end
-- Read the field ident immediately after the type chain. while cursor <= body_len do
local field_ident, field_end = duffle.read_ident(body, cursor) local field_ident, field_end = duffle.read_ident(body, cursor)
if not field_ident then return nil, type_end + 1 end if not field_ident then break end
return { fields[#fields + 1] = {
name = field_ident, name = field_ident,
type_name = type_name, type_name = type_name,
pointer_depth = depth, pointer_depth = depth,
-- offset + byte_size filled by resolve_struct_field_sizes
offset = nil, offset = nil,
byte_size = nil, byte_size = nil,
}, field_end }
end) cursor = duffle.skip_ws_and_cmt(body, field_end)
if body:sub(cursor, cursor) == "," then
cursor = duffle.skip_ws_and_cmt(body, cursor + 1)
else
break
end
end
if cursor <= body_len and body:sub(cursor, cursor) == ";" then
cursor = cursor + 1
end
body_pos = cursor
end
end
return fields
end end
-- Parse the `Enum_(<underlying>, <name>) { <body> }` body for entries. -- Parse the `Enum_(<underlying>, <name>) { <body> }` body for entries.
@@ -1241,31 +1272,20 @@ local function parse_atom_dbg_reg_default(source, pos, ident_end, line_of, out)
return after_paren return after_paren
end end
--- Parse: `MipsAtom_(<name>) [atom_info(<binds>, <reads>, <writes>)] { <body> }` --- Lookahead for `atom_info(...)` after a declaration's closing paren.
--- @param source string --- Records into `dest` (atom_infos or component_atom_infos). Returns the position after the info, or after_paren if none.
--- @param pos integer local function parse_atom_info_after_decl(source, after_paren, raw_name, line_of, out, dest)
--- @param ident_end integer
--- @param line_of fun(pos: integer): integer
--- @param out SourceScan
--- @return integer
local function parse_mips_atom(source, pos, ident_end, line_of, out)
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
if not inner then return after_paren end
local raw_name = duffle.read_ident(inner, 1)
-- Lookahead for atom_info(...) between `)` and `{`. Captures sub-calls; updates brace search start.
local brace_search_pos = after_paren
local lookahead = duffle.skip_ws_and_cmt(source, after_paren) local lookahead = duffle.skip_ws_and_cmt(source, after_paren)
local look_ident, look_end = duffle.read_ident(source, lookahead) local look_ident, look_end = duffle.read_ident(source, lookahead)
if look_ident == "atom_info" then if look_ident ~= "atom_info" then return after_paren end
local info_open = duffle.skip_ws_and_cmt(source, look_end) local info_open = duffle.skip_ws_and_cmt(source, look_end)
if source:sub(info_open, info_open) == "(" then if source:sub(info_open, info_open) ~= "(" then return after_paren end
local info_inner, info_after = duffle.read_parens(source, info_open) local info_inner, info_after = duffle.read_parens(source, info_open)
-- info_line feeds the per-atom reg_type_overrides table. if not info_inner then return after_paren end
local info_line = line_of(info_open) local info_line = line_of(info_open)
local ai_binds, ai_reads, ai_writes, ai_view, ai_overrides, ai_ctx, ai_phase = scan_atom_info_subcalls(info_inner, info_line) local ai_binds, ai_reads, ai_writes, ai_view, ai_overrides, ai_ctx, ai_phase = scan_atom_info_subcalls(info_inner, info_line)
out.atom_infos[#out.atom_infos + 1] = { dest = dest or out.atom_infos
dest[#dest + 1] = {
atom_name = raw_name or "?", binds = ai_binds, atom_name = raw_name or "?", binds = ai_binds,
reads = ai_reads or {}, writes = ai_writes or {}, reads = ai_reads or {}, writes = ai_writes or {},
view = ai_view, view = ai_view,
@@ -1282,11 +1302,9 @@ local function parse_mips_atom(source, pos, ident_end, line_of, out)
info_line = line_of(lookahead), info_line = line_of(lookahead),
} }
elseif raw_name and ai_overrides then elseif raw_name and ai_overrides then
-- Record per-atom overrides even without atom_view.
out.atom_views[raw_name] = out.atom_views[raw_name] or { atom_name = raw_name, binds_name = nil, reg_type_overrides = nil, info_line = line_of(lookahead) } out.atom_views[raw_name] = out.atom_views[raw_name] or { atom_name = raw_name, binds_name = nil, reg_type_overrides = nil, info_line = line_of(lookahead) }
out.atom_views[raw_name].reg_type_overrides = ai_overrides out.atom_views[raw_name].reg_type_overrides = ai_overrides
end end
-- Project the per-atom atom_ctx / atom_phase declarations onto the global phase index.
if raw_name then if raw_name then
if ai_ctx then if ai_ctx then
out.atom_ctxs = out.atom_ctxs or {} out.atom_ctxs = out.atom_ctxs or {}
@@ -1298,115 +1316,163 @@ local function parse_mips_atom(source, pos, ident_end, line_of, out)
out.atom_phases[ai_phase].atoms[#out.atom_phases[ai_phase].atoms + 1] = raw_name out.atom_phases[ai_phase].atoms[#out.atom_phases[ai_phase].atoms + 1] = raw_name
end end
end end
brace_search_pos = info_after return info_after
end
end
local body, after_brace, body_off = find_body_braces(source, brace_search_pos, open_paren + 1)
if not body then return after_brace end
if raw_name and raw_name ~= "" then
register_atom(out, "atom", line_of(pos), raw_name, body, body_off, raw_name, pos, after_paren, source)
end
return after_brace
end end
--- Parse: `MipsAtomComp_(<name>) { <body> }` local DECL_FORMS = {
--- @param source string MipsAtom_ = {
--- @param pos integer kind = "atom", name = "paren_ident", body = "braces_after",
--- @param ident_end integer info_dest = "atom_infos", strip = false,
--- @param line_of fun(pos: integer): integer },
--- @param out SourceScan MipsAtom_Proc_ = {
--- @return integer kind = "atom_proc", name = "backward_atom_proc", body = "last_brace_in_args",
local function parse_mips_atom_comp(source, pos, ident_end, line_of, out) info_dest = "atom_infos", strip = false, after = "reguse_hook",
local inner, after_paren, open_paren = read_parens_after(source, ident_end) },
if not inner then return after_paren end MipsAtomComp_ = {
kind = "comp_bare", name = "paren_ident", body = "braces_after",
info_dest = "component_atom_infos", strip = "ac_",
},
MipsAtomComp_Proc_ = {
kind = "comp_proc", name = "backward_fi", body = "last_brace_in_args",
info_dest = nil, strip = "ac_",
},
MipsAtomComp_ProcMap_ = {
kind = "comp_proc", name = "backward_fi", body = "comma_arg_2",
info_dest = nil, strip = "ac_", after = "map_command_hook",
},
}
local raw_name = duffle.read_ident(inner, 1) local function last_brace_body(inner, open_paren)
if not raw_name then return open_paren + 1 end
local body, after_brace, body_off = find_body_braces(source, after_paren, open_paren + 1)
if not body then return after_brace end
local name = strip_ac_prefix(raw_name)
register_atom(out, "comp_bare", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
return after_brace
end
--- Parse: `MipsAtomComp_Proc_(<name>, { <body> })` — body is inside the LAST `{` in args.
--- @param source string
--- @param pos integer
--- @param ident_end integer
--- @param line_of fun(pos: integer): integer
--- @param out SourceScan
--- @return integer
local function parse_mips_atom_comp_proc(source, pos, ident_end, line_of, out)
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
if not inner then return after_paren end
-- Find the LAST `{` in inner (the body brace, not any potential embedded braces in expressions).
local last_brace_pos = nil local last_brace_pos = nil
for search_pos = #inner, 1, -1 do for search_pos = #inner, 1, -1 do
if inner:sub(search_pos, search_pos) == "{" then last_brace_pos = search_pos; break end if inner:sub(search_pos, search_pos) == "{" then
last_brace_pos = search_pos
break
end end
if not last_brace_pos then return after_paren end end
if not last_brace_pos then return nil end
-- Use duffle.read_braces to find the matching close brace.
-- Uses `read_balanced` for delimiter-depth tracking.
-- If close_pos is past the end of inner, the brace didn't match (malformed input); skip.
local body, close_pos = duffle.read_braces(inner, last_brace_pos) local body, close_pos = duffle.read_braces(inner, last_brace_pos)
if close_pos > #inner + 1 then return after_paren end if close_pos > #inner + 1 then return nil end
return body, open_paren + 2 + last_brace_pos
local raw_name = inner:match("^%s*([%w_]+)") or "?"
local name = strip_ac_prefix(raw_name)
-- Position of body[1] in source = open_paren + 1 (start of inner) + last_brace_pos + 1 (past '{').
local body_off = open_paren + 2 + last_brace_pos
register_atom(out, "comp_proc", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
return after_paren
end end
--- Parse: `MipsAtom_Proc_(<name>, <abuilder>, { <body> })` — body is inside the LAST `{` in args. local function reguse_hook(source, pos, line_of, out, extras)
--- Per Task 12.10: full support for the runtime-proc atom form. Registers the atom local entry = out.atoms[#out.atoms]
--- with kind `"atom_proc"` so offsets.lua / components.lua can emit if not entry then return end
--- * `mac_<name>` aliases in `gen/macs.h` (the components pass) local reg_use_schema_name, reg_use_param_name
--- * `atom_offset__X__Y` defs in `gen/offsets.h` (the offsets pass) if extras.args_inner then
--- The atom name is the FIRST ident of the args (the second arg `ab` is the local arg_tokens = duffle.split_top_level_commas(extras.args_inner)
--- atom-builder, not the name). Unlike `MipsAtomComp_Proc_`, there is no `ac_` for _, tok in ipairs(arg_tokens) do
--- prefix on the symbol — `MipsAtom_Proc_` is the runtime-proc wrapper, so the local trimmed = duffle.trim(tok)
--- symbol IS the bare atom name (e.g. `normalize_v3s4`, not `ac_normalize_v3s4`). local schema_suffix, param = trimmed:match("RegUse_([%w_]+)%s+([%w_]+)$")
--- @param source string if schema_suffix then
--- @param pos integer if reg_use_schema_name then
--- @param ident_end integer out.reg_use_errors[#out.reg_use_errors + 1] = {
--- @param line_of fun(pos: integer): integer kind = "reguse_multiple_params",
--- @param out SourceScan schema_name = "RegUse_" .. schema_suffix,
--- @return integer source_line = line_of(pos),
local function parse_mips_atom_proc(source, pos, ident_end, line_of, out) }
else
reg_use_schema_name = "RegUse_" .. schema_suffix
reg_use_param_name = param
end
end
end
end
entry.reg_use_schema_name = reg_use_schema_name
entry.reg_use_param_name = reg_use_param_name
if reg_use_schema_name and extras.func_ident then
local expected = "RegUse_" .. extras.func_ident
if reg_use_schema_name ~= expected then
out.reg_use_errors[#out.reg_use_errors + 1] = {
kind = "reguse_name_mismatch",
schema_name = reg_use_schema_name,
func_ident = extras.func_ident,
source_line = line_of(pos),
}
end
end
end
local function parse_decl_form(source, pos, ident_end, line_of, out)
local ident = duffle.read_ident(source, pos)
local form = ident and DECL_FORMS[ident]
if not form then return ident_end end
local inner, after_paren, open_paren = read_parens_after(source, ident_end) local inner, after_paren, open_paren = read_parens_after(source, ident_end)
if not inner then return after_paren end if not inner then return after_paren end
-- Find the LAST `{` in inner (the body brace, not any potential embedded braces in expressions). local extras = {}
local last_brace_pos = nil local raw_name
for search_pos = #inner, 1, -1 do if form.name == "paren_ident" then
if inner:sub(search_pos, search_pos) == "{" then last_brace_pos = search_pos; break end raw_name = duffle.read_ident(inner, 1)
if form.strip and not raw_name then return open_paren + 1 end
elseif form.name == "backward_fi" then
raw_name = duffle.find_function_decl_for(source, open_paren, SLICE_MIPS_CODE_LEN)
elseif form.name == "backward_atom_proc" then
raw_name, extras.args_inner, extras.func_ident, extras.after_func_paren =
duffle.find_atom_proc_decl_for(source, open_paren, MIPS_ATOM_PTR_LEN)
end end
if not last_brace_pos then return after_paren end if form.name == "paren_ident" and form.strip and not raw_name then
return open_paren + 1
end
if not raw_name then raw_name = "?" end
local name = form.strip and strip_ac_prefix(raw_name) or raw_name
-- Use duffle.read_braces to find the matching close brace. if form.info_dest == "component_atom_infos" then
-- Uses `read_balanced` for delimiter-depth tracking. out.component_atom_infos = out.component_atom_infos or {}
-- If close_pos is past the end of inner, the brace didn't match (malformed input); skip. end
local body, close_pos = duffle.read_braces(inner, last_brace_pos) local info_dest = form.info_dest and out[form.info_dest]
if close_pos > #inner + 1 then return after_paren end
-- The atom name is the FIRST ident of the args (matches MipsAtomComp_Proc_'s "first ident" rule).
-- MipsAtom_Proc_ has no `ac_` prefix; `strip_ac_prefix` is a no-op for unprefixed names.
local raw_name = inner:match("^%s*([%w_]+)") or "?"
local name = strip_ac_prefix(raw_name)
-- Position of body[1] in source = open_paren + 1 (start of inner) + last_brace_pos + 1 (past '{').
local body_off = open_paren + 2 + last_brace_pos
register_atom(out, "atom_proc", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
local body, body_off, resume
if form.body == "braces_after" then
local brace_search = after_paren
if info_dest then
brace_search = parse_atom_info_after_decl(
source, after_paren, name, line_of, out, info_dest)
end
local after_brace
body, after_brace, body_off = find_body_braces(source, brace_search, open_paren + 1)
if not body then return after_brace end
resume = after_brace
elseif form.body == "last_brace_in_args" then
body, body_off = last_brace_body(inner, open_paren)
if not body then return after_paren end
resume = after_paren
if form.info_dest then
if extras.after_func_paren then
parse_atom_info_after_decl(
source, extras.after_func_paren, name, line_of, out, out.atom_infos)
else
parse_atom_info_after_decl(source, pos, name, line_of, out, out.atom_infos)
end
end
elseif form.body == "comma_arg_2" then
local args = duffle.split_top_level_commas(inner)
if #args < 2 then return after_paren end
body = duffle.trim(args[2])
if body == "" then return after_paren end
body_off = open_paren + 1 + (inner:find(body, 1, true) or 1) - 1
resume = after_paren
else
return after_paren return after_paren
end
local skip_register = form.name == "paren_ident" and not form.strip
and (raw_name == "?" or raw_name == "")
if not skip_register then
register_atom(out, form.kind, line_of(pos), name, body, body_off,
raw_name, pos, after_paren, source)
end
if form.after == "reguse_hook" then
reguse_hook(source, pos, line_of, out, extras)
elseif form.after == "map_command_hook" then
local entry = out.atoms[#out.atoms]
if entry then entry.map_command = body end
end
return resume
end end
--- Parse: `MipsCode code_<name> { <body> }` (raw atom form — offsets pass only). --- Parse: `MipsCode code_<name> { <body> }` (raw atom form — offsets pass only).
@@ -1513,6 +1579,236 @@ local function register_typedef_alias(underlying, name, pos, line_of, out)
} }
end end
local parse_reg_use_schema_body
local function fields_for_reg_type(type_name, type_registry)
local reg_name = "Reg_" .. type_name
local entry = type_registry and type_registry[reg_name]
if entry and entry.fields and #entry.fields > 0 then
local names = {}
for _, field in ipairs(entry.fields) do
if field.name then names[#names + 1] = field.name end
end
if #names > 0 then return names end
end
if entry and entry.body and parse_reg_use_schema_body then
local schema = parse_reg_use_schema_body(entry.body, type_registry)
if schema and schema.slots then
local names = {}
for _, slot in ipairs(schema.slots) do
if slot.name then names[#names + 1] = slot.name end
end
if #names > 0 then return names end
end
end
return nil
end
parse_reg_use_schema_body = function(body, type_registry, opts)
opts = opts or {}
local require_types = opts.require_types == true
local pending = false
local slots = {}
local alias_to_slot = {}
local slot_names = {}
local errors = {}
local function add_alias(path, slot)
if alias_to_slot[path] then
errors[#errors + 1] = { kind = "reguse_duplicate_alias", path = path }
return false
end
alias_to_slot[path] = slot
return true
end
local function add_slot(name, aliases, readonly)
if slot_names[name] then
errors[#errors + 1] = { kind = "reguse_duplicate_slot", name = name }
return nil
end
slot_names[name] = true
local slot = { name = name, aliases = aliases, readonly = readonly == true }
slots[#slots + 1] = slot
return slot
end
local function parse_reg_names(text, pos)
local names = {}
while pos <= #text do
pos = duffle.skip_ws_and_cmt(text, pos)
local name, name_end = duffle.read_ident(text, pos)
if not name then return nil, pos end
names[#names + 1] = name
pos = duffle.skip_ws_and_cmt(text, name_end)
if text:sub(pos, pos) == "," then
pos = pos + 1
else
break
end
end
if text:sub(pos, pos) == ";" then pos = pos + 1 end
return names, pos
end
local pos = 1
while pos <= #body do
pos = duffle.skip_ws_and_cmt(body, pos)
if pos > #body then break end
local first, first_end = duffle.read_ident(body, pos)
if not first then
pos = pos + 1
goto continue
end
local after = duffle.skip_ws_and_cmt(body, first_end)
if first == "const" then
errors[#errors + 1] = { kind = "reguse_const_reg_spelling" }
return nil, errors
elseif first == "union" then
if body:sub(after, after) ~= "{" then
errors[#errors + 1] = { kind = "reguse_malformed" }
return nil, errors
end
local inner, after_braces = duffle.read_braces(body, after)
if not inner then
errors[#errors + 1] = { kind = "reguse_malformed" }
return nil, errors
end
local members = {}
local union_readonly = nil
local inner_pos = 1
while inner_pos <= #inner do
inner_pos = duffle.skip_ws_and_cmt(inner, inner_pos)
if inner_pos > #inner then break end
local m_type, m_type_end = duffle.read_ident(inner, inner_pos)
if not m_type then
inner_pos = inner_pos + 1
goto continue_inner
end
if m_type == "const" then
errors[#errors + 1] = { kind = "reguse_const_reg_spelling" }
return nil, errors
end
if m_type ~= "Reg" then
errors[#errors + 1] = { kind = "reguse_malformed" }
return nil, errors
end
local m_after = duffle.skip_ws_and_cmt(inner, m_type_end)
local m_readonly = false
local maybe_const, maybe_end = duffle.read_ident(inner, m_after)
if maybe_const == "const" then
m_readonly = true
m_after = duffle.skip_ws_and_cmt(inner, maybe_end)
end
if union_readonly == nil then
union_readonly = m_readonly
elseif union_readonly ~= m_readonly then
errors[#errors + 1] = { kind = "reguse_mixed_const" }
return nil, errors
end
local names, new_inner = parse_reg_names(inner, m_after)
if not names or #names == 0 then
errors[#errors + 1] = { kind = "reguse_malformed" }
return nil, errors
end
for _, n in ipairs(names) do members[#members + 1] = n end
inner_pos = new_inner
::continue_inner::
end
if #members == 0 then
errors[#errors + 1] = { kind = "reguse_malformed" }
return nil, errors
end
local after_close = duffle.skip_ws_and_cmt(body, after_braces)
local inst_name, inst_end = duffle.read_ident(body, after_close)
local aliases = {}
local slot_name
if inst_name then
slot_name = inst_name
for _, m in ipairs(members) do
local path = inst_name .. "." .. m
if not add_alias(path, slot_name) then return nil, errors end
aliases[#aliases + 1] = path
end
after_close = inst_end
else
slot_name = members[1]
for _, m in ipairs(members) do
if not add_alias(m, slot_name) then return nil, errors end
aliases[#aliases + 1] = m
end
end
if not add_slot(slot_name, aliases, union_readonly) then return nil, errors end
after_close = duffle.skip_ws_and_cmt(body, after_close)
if body:sub(after_close, after_close) == ";" then after_close = after_close + 1 end
pos = after_close
elseif first == "Reg" or first == "Reg_" then
local typed_fields = nil
if first == "Reg_" then
if body:sub(after, after) ~= "(" then
errors[#errors + 1] = { kind = "reguse_malformed" }
return nil, errors
end
local type_inner, after_paren = duffle.read_parens(body, after)
if not type_inner then
errors[#errors + 1] = { kind = "reguse_malformed" }
return nil, errors
end
local type_ident = duffle.trim(type_inner)
typed_fields = fields_for_reg_type(type_ident, type_registry)
if not typed_fields then
if require_types then
errors[#errors + 1] = { kind = "reguse_unknown_reg_type", type_name = type_ident }
else
pending = true
end
end
after = duffle.skip_ws_and_cmt(body, after_paren)
end
local readonly = false
local maybe_const, maybe_end = duffle.read_ident(body, after)
if maybe_const == "const" then
readonly = true
after = duffle.skip_ws_and_cmt(body, maybe_end)
end
local names, new_pos = parse_reg_names(body, after)
if not names or #names == 0 then
errors[#errors + 1] = { kind = "reguse_malformed" }
return nil, errors
end
for _, n in ipairs(names) do
if first == "Reg_" then
if typed_fields then
for _, field in ipairs(typed_fields) do
local path = n .. "." .. field
if not add_alias(path, path) then return nil, errors end
if not add_slot(path, { path }, readonly) then return nil, errors end
end
end
else
if not add_alias(n, n) then return nil, errors end
if not add_slot(n, { n }, readonly) then return nil, errors end
end
end
pos = new_pos
else
errors[#errors + 1] = { kind = "reguse_malformed" }
return nil, errors
end
::continue::
end
if #slots == 0 then
if pending and not require_types then
return { slots = slots, alias_to_slot = alias_to_slot, pending = true }, errors
end
if #errors == 0 then
errors[#errors + 1] = { kind = "reguse_malformed" }
end
return nil, errors
end
return { slots = slots, alias_to_slot = alias_to_slot, pending = pending }, errors
end
--- Parse: `typedef` declarations. --- Parse: `typedef` declarations.
--- ---
--- Recognizes four shapes: --- Recognizes four shapes:
@@ -1546,6 +1842,21 @@ local function parse_typedef_binds(source, pos, ident_end, line_of, out)
local body, after_brace = find_body_braces(source, after_paren, open_paren + 1) local body, after_brace = find_body_braces(source, after_paren, open_paren + 1)
if not body then return after_brace end if not body then return after_brace end
register_struct_type(body, name, pos, line_of, out) register_struct_type(body, name, pos, line_of, out)
if name:sub(1, 7) == "RegUse_" then
local schema, schema_errors = parse_reg_use_schema_body(body, out.type_name_registry)
if schema then
schema.name = name
schema.source_file = out._source_file
schema.source_line = line_of(pos)
out.reg_use_schemas[name] = schema
end
for _, err in ipairs(schema_errors or {}) do
err.schema_name = name
err.source_file = out._source_file
err.source_line = line_of(pos)
out.reg_use_errors[#out.reg_use_errors + 1] = err
end
end
attach_debug_skip_marker(out, "unrelated") attach_debug_skip_marker(out, "unrelated")
return after_brace return after_brace
@@ -1872,10 +2183,11 @@ end
-- Adding a new construct = 1 row here + 1 parser function above. -- Adding a new construct = 1 row here + 1 parser function above.
local DECL_PARSERS = { local DECL_PARSERS = {
MipsAtom_ = parse_mips_atom, MipsAtom_ = parse_decl_form,
MipsAtom_Proc_ = parse_mips_atom_proc, MipsAtom_Proc_ = parse_decl_form,
MipsAtomComp_ = parse_mips_atom_comp, MipsAtomComp_ = parse_decl_form,
MipsAtomComp_Proc_ = parse_mips_atom_comp_proc, MipsAtomComp_Proc_ = parse_decl_form,
MipsAtomComp_ProcMap_ = parse_decl_form,
-- `atom_dbg_skip` is the only debug-skip parser entry. Every other -- `atom_dbg_skip` is the only debug-skip parser entry. Every other
-- identifier follows the ordinary unrelated-token path; there is no alias. -- identifier follows the ordinary unrelated-token path; there is no alias.
atom_dbg_skip = parse_dbg_skip_marker, atom_dbg_skip = parse_dbg_skip_marker,
@@ -1895,6 +2207,131 @@ local DECL_PARSERS = {
-- Only the bare `atom_dbg_skip` marker reaches `parse_dbg_skip_marker`. -- Only the bare `atom_dbg_skip` marker reaches `parse_dbg_skip_marker`.
-- Unknown identifiers follow the same unrelated-token path as every other unsupported source token. -- Unknown identifiers follow the same unrelated-token path as every other unsupported source token.
local function tape_skip_ident(ident)
return ident and (DECL_FORMS[ident] or ident == "Struct_" or ident == "Enum_")
end
local function collect_addrs_assigns_REMOVED(text)
local addrs = {}
local pos = 1
local n = #text
while pos <= n do
pos = duffle.skip_ws_and_cmt(text, pos)
if pos > n then break end
local ident, ident_end = duffle.read_ident(text, pos)
if ident == "addrs" then
local after = duffle.skip_ws_and_cmt(text, ident_end)
if text:sub(after, after) == "[" then
local inner, after_br = duffle.read_brackets(text, after)
local idx = inner and tonumber(duffle.trim(inner))
after_br = duffle.skip_ws_and_cmt(text, after_br or after)
if idx and text:sub(after_br, after_br) == "=" then
local rhs = duffle.skip_ws_and_cmt(text, after_br + 1)
local rhs_ident = duffle.read_ident(text, rhs)
if rhs_ident then addrs[idx] = rhs_ident end
pos = rhs
else
pos = after_br or (after + 1)
end
else
pos = ident_end
end
elseif ident then
pos = ident_end
else
pos = pos + 1
end
end
return addrs
end
local function collect_tb_emits(body, addrs)
local names = {}
local pos = 1
local n = #body
while pos <= n do
pos = duffle.skip_ws_and_cmt(body, pos)
if pos > n then break end
local ident, ident_end = duffle.read_ident(body, pos)
if ident == "tb_emit_" or ident == "tb_emit" then
local after = duffle.skip_ws_and_cmt(body, ident_end)
if body:sub(after, after) == "(" then
local inner, after_p = duffle.read_parens(body, after)
local name
if ident == "tb_emit_" then
name = duffle.trim(inner or ""):match("^([%w_]+)")
else
local args = duffle.split_top_level_commas(inner or "")
local last = duffle.trim(args[#args] or "")
local idx = last:match("^addrs%s*%[%s*(%d+)%s*%]$")
if idx then
name = addrs[tonumber(idx)]
else
name = last:match("([%w_]+)$")
end
end
if name then names[#names + 1] = name end
pos = after_p or (after + 1)
else
pos = ident_end
end
elseif ident then
pos = ident_end
else
pos = pos + 1
end
end
return names
end
-- Linear appearance order of tb_emit / tb_emit_ in each C function body.
-- Commented-out emits are skipped by skip_ws_and_cmt. No C if/loop CFG.
local function scan_tape_chains(source)
local addrs = collect_addrs_assigns(source)
local chains = {}
local pos = 1
local n = #source
while pos <= n do
pos = duffle.skip_ws_and_cmt(source, pos)
if pos > n then break end
local ident, ident_end = duffle.read_ident(source, pos)
if tape_skip_ident(ident) then
local after = duffle.skip_ws_and_cmt(source, ident_end)
if source:sub(after, after) == "(" then
local _, after_p = duffle.read_parens(source, after)
after = duffle.skip_ws_and_cmt(source, after_p or after)
end
if source:sub(after, after) == "{" then
local _, after_b = duffle.read_braces(source, after)
pos = after_b or (after + 1)
else
pos = after
end
elseif ident then
local after = duffle.skip_ws_and_cmt(source, ident_end)
if source:sub(after, after) == "(" then
local _, after_p = duffle.read_parens(source, after)
after = duffle.skip_ws_and_cmt(source, after_p or after)
if source:sub(after, after) == "{" then
local body, after_b = duffle.read_braces(source, after)
local names = collect_tb_emits(body or "", addrs)
if #names > 0 then
chains[#chains + 1] = names
end
pos = after_b or (after + 1)
else
pos = after
end
else
pos = ident_end
end
else
pos = pos + 1
end
end
return chains
end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- The single source walker -- The single source walker
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -1913,6 +2350,7 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
raw_atoms = {}, raw_atoms = {},
binds = {}, binds = {},
atom_infos = {}, atom_infos = {},
component_atom_infos = {},
macros = {}, macros = {},
-- Raw marker evidence for annotation validation. The `debug_skip` boolean -- Raw marker evidence for annotation validation. The `debug_skip` boolean
-- is stamped on the declaration record itself; the projection lives on AtomEntry.debug_skip. -- is stamped on the declaration record itself; the projection lives on AtomEntry.debug_skip.
@@ -1937,6 +2375,12 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
-- typedef chain walking (cycle-guarded, depth <= 8), and struct field sums. -- typedef chain walking (cycle-guarded, depth <= 8), and struct field sums.
-- See `propagate_type_sizes()` below. -- See `propagate_type_sizes()` below.
type_name_registry = {}, type_name_registry = {},
reg_use_schemas = {},
tape_chains = {},
_addrs = {},
_chain = nil,
_brace_depth = 0,
reg_use_errors = {},
-- Shared `R_*_Code -> integer code` registry -- Shared `R_*_Code -> integer code` registry
-- (passed in from M.run pass 1; same reference so preprocessor intercept writes are visible to the enum-value resolver). -- (passed in from M.run pass 1; same reference so preprocessor intercept writes are visible to the enum-value resolver).
-- Stripped from `src.scan` before return. -- Stripped from `src.scan` before return.
@@ -1970,6 +2414,48 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
local parser = DECL_PARSERS[ident] local parser = DECL_PARSERS[ident]
if parser then if parser then
pos = parser(source, pos, ident_end, line_of, out) pos = parser(source, pos, ident_end, line_of, out)
elseif ident == "addrs" then
local after = duffle.skip_ws_and_cmt(source, ident_end)
if source:sub(after, after) == "[" then
local inner, after_br = duffle.read_brackets(source, after)
local idx = inner and tonumber(duffle.trim(inner))
after_br = duffle.skip_ws_and_cmt(source, after_br or after)
if idx and source:sub(after_br, after_br) == "=" then
local rhs = duffle.skip_ws_and_cmt(source, after_br + 1)
local rhs_ident = duffle.read_ident(source, rhs)
if rhs_ident then out._addrs[idx] = rhs_ident end
pos = rhs
else
pos = after_br or (after + 1)
end
else
pos = ident_end
end
elseif ident == "tb_emit_" or ident == "tb_emit" then
local after = duffle.skip_ws_and_cmt(source, ident_end)
if source:sub(after, after) == "(" then
local inner, after_p = duffle.read_parens(source, after)
local name
if ident == "tb_emit_" then
name = duffle.trim(inner or ""):match("^([%w_]+)")
else
local args = duffle.split_top_level_commas(inner or "")
local last = duffle.trim(args[#args] or "")
local idx = last:match("^addrs%s*%[%s*(%d+)%s*%]$")
if idx then
name = out._addrs[tonumber(idx)]
else
name = last:match("([%w_]+)$")
end
end
if name then
out._chain = out._chain or {}
out._chain[#out._chain + 1] = name
end
pos = after_p or (after + 1)
else
pos = ident_end
end
else else
-- Unsupported identifiers follow the unrelated-token path. If a -- Unsupported identifiers follow the unrelated-token path. If a
-- pending marker is still open, consume it so it cannot drift to a -- pending marker is still open, consume it so it cannot drift to a
@@ -1988,12 +2474,22 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
else else
local markers = out.debug_skip_markers local markers = out.debug_skip_markers
local marker = markers[#markers] local marker = markers[#markers]
if marker and marker.pending and marker.proc_prelude then
local c = source:sub(pos, pos) local c = source:sub(pos, pos)
if marker and marker.pending and marker.proc_prelude then
if c == "{" or c == ";" then if c == "{" or c == ";" then
attach_debug_skip_marker(out, "unrelated") attach_debug_skip_marker(out, "unrelated")
end end
end end
if c == "{" then
out._brace_depth = out._brace_depth + 1
elseif c == "}" then
if out._brace_depth == 1 and out._chain and #out._chain > 0 then
out.tape_chains[#out.tape_chains + 1] = out._chain
end
out._chain = nil
out._brace_depth = out._brace_depth - 1
if out._brace_depth < 0 then out._brace_depth = 0 end
end
pos = pos + 1 pos = pos + 1
end end
end end
@@ -2004,6 +2500,12 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
-- Runs AFTER the source walk so all typedef / Struct_ / Enum_ declarations have been parsed into `out.type_name_registry`. -- Runs AFTER the source walk so all typedef / Struct_ / Enum_ declarations have been parsed into `out.type_name_registry`.
-- Mutates each entry's `byte_size` field in place; fields with pointer_depth > 0 already carry byte_size = 4 from parse time and are unaffected. -- Mutates each entry's `byte_size` field in place; fields with pointer_depth > 0 already carry byte_size = 4 from parse time and are unaffected.
propagate_type_sizes(out) propagate_type_sizes(out)
if out._chain and #out._chain > 0 then
out.tape_chains[#out.tape_chains + 1] = out._chain
end
out._addrs = nil
out._chain = nil
out._brace_depth = nil
return out return out
end end
@@ -2166,16 +2668,21 @@ local function merge_corpus_registries(corpus)
corpus.atom_ctxs = corpus.atom_ctxs or {} corpus.atom_ctxs = corpus.atom_ctxs or {}
corpus.atom_phases = corpus.atom_phases or {} corpus.atom_phases = corpus.atom_phases or {}
corpus.atom_infos = corpus.atom_infos or {} corpus.atom_infos = corpus.atom_infos or {}
corpus.component_atom_infos = corpus.component_atom_infos or {}
corpus.atom_auto_regs = corpus.atom_auto_regs or {} corpus.atom_auto_regs = corpus.atom_auto_regs or {}
corpus.phase_auto_regs = corpus.phase_auto_regs or {} corpus.phase_auto_regs = corpus.phase_auto_regs or {}
corpus.collisions = corpus.collisions or {} corpus.collisions = corpus.collisions or {}
corpus.reg_use_schemas = corpus.reg_use_schemas or {}
corpus.reg_use_errors = corpus.reg_use_errors or {}
corpus.tape_chains = corpus.tape_chains or {}
-- Replace the existing corpus collections with empty tables so a re-run on the same corpus produces identical state (deterministic merge). -- Replace the existing corpus collections with empty tables so a re-run on the same corpus produces identical state (deterministic merge).
-- This is safe because M.run is the only writer to these tables within a single orchestrator invocation. -- This is safe because M.run is the only writer to these tables within a single orchestrator invocation.
for _, key in ipairs({ for _, key in ipairs({
"register_alias_registry", "type_name_registry", "binds_by_name", "register_alias_registry", "type_name_registry", "binds_by_name",
"atoms_by_name", "atom_views", "atom_ctxs", "atom_phases", "atoms_by_name", "atom_views", "atom_ctxs", "atom_phases",
"atom_infos", "collisions", "atom_infos", "component_atom_infos", "collisions", "reg_use_schemas", "reg_use_errors",
"tape_chains",
}) do }) do
corpus[key] = {} corpus[key] = {}
end end
@@ -2268,6 +2775,65 @@ local function merge_corpus_registries(corpus)
for _, info in ipairs(scan.atom_infos or {}) do for _, info in ipairs(scan.atom_infos or {}) do
corpus.atom_infos[#corpus.atom_infos + 1] = info corpus.atom_infos[#corpus.atom_infos + 1] = info
end end
for _, info in ipairs(scan.component_atom_infos or {}) do
corpus.component_atom_infos[#corpus.component_atom_infos + 1] = info
end
for name, schema in pairs(scan.reg_use_schemas or {}) do
if corpus.reg_use_schemas[name] == nil then
corpus.reg_use_schemas[name] = schema
end
end
for _, err in ipairs(scan.reg_use_errors or {}) do
corpus.reg_use_errors[#corpus.reg_use_errors + 1] = err
end
for _, chain in ipairs(scan.tape_chains or {}) do
corpus.tape_chains[#corpus.tape_chains + 1] = chain
end
end
end
end
local SCHEMA_BODY_ERROR = {
reguse_malformed = true,
reguse_unknown_reg_type = true,
reguse_duplicate_alias = true,
reguse_duplicate_slot = true,
reguse_const_reg_spelling = true,
reguse_mixed_const = true,
}
-- Re-parse every RegUse_* body against the merged type_name_registry.
-- Scan-time expansion still runs when Reg_T is in the same source.
-- Missing Reg_T after merge is reguse_unknown_reg_type, not a fallback table.
local function resolve_reg_use_schemas(corpus)
local kept = {}
for _, err in ipairs(corpus.reg_use_errors or {}) do
if not SCHEMA_BODY_ERROR[err.kind] then
kept[#kept + 1] = err
end
end
corpus.reg_use_errors = kept
for name, type_entry in pairs(corpus.type_name_registry or {}) do
if name:sub(1, 7) == "RegUse_" and type_entry.body then
local fresh, errs = parse_reg_use_schema_body(
type_entry.body, corpus.type_name_registry, { require_types = true })
if fresh then
fresh.name = name
local old = corpus.reg_use_schemas[name]
fresh.source_file = (old and old.source_file) or type_entry.source_file
fresh.source_line = (old and old.source_line) or type_entry.source_line
corpus.reg_use_schemas[name] = fresh
else
corpus.reg_use_schemas[name] = nil
end
for _, err in ipairs(errs or {}) do
err.schema_name = name
err.source_file = type_entry.source_file
err.source_line = type_entry.source_line
corpus.reg_use_errors[#corpus.reg_use_errors + 1] = err
end
end end
end end
end end
@@ -2356,6 +2922,7 @@ function M.run(ctx)
-- Merge per-source scans into the corpus registries (see merge_corpus_registries for first-wins + collision discipline). -- Merge per-source scans into the corpus registries (see merge_corpus_registries for first-wins + collision discipline).
merge_corpus_registries(corpus) merge_corpus_registries(corpus)
resolve_reg_use_schemas(corpus)
-- code_macros and code_macro_bodies are function-local; the GC reclaims them on M.run return. -- code_macros and code_macro_bodies are function-local; the GC reclaims them on M.run return.
return { outputs = {}, errors = {}, warnings = {} } return { outputs = {}, errors = {}, warnings = {} }
File diff suppressed because it is too large Load Diff
+14 -25
View File
@@ -91,11 +91,9 @@ if (-not (Test-Path -LiteralPath $path_pcsx_packages)) {
New-Item -ItemType Directory -Path $path_pcsx_packages -Force | Out-Null New-Item -ItemType Directory -Path $path_pcsx_packages -Force | Out-Null
} }
# Download anything missing. Skip the package entirely if its dir already has # Download anything missing.
# any contents (the legacy packages.config style means the targets file # Skip the package entirely if its dir already has any contents (the legacy packages.config style means the targets file location varies per package
# location varies per package — `luajit.native` puts it at build/native/, # — `luajit.native` puts it at build/native/, `glfw` puts it elsewhere — so we can't probe a specific path; just check whether the dir is non-empty).
# `glfw` puts it elsewhere — so we can't probe a specific path; just check
# whether the dir is non-empty).
Add-Type -AssemblyName System.IO.Compression.FileSystem Add-Type -AssemblyName System.IO.Compression.FileSystem
foreach ($pkg in $required_packages.Values) { foreach ($pkg in $required_packages.Values) {
$pkgDir = Join-Path $path_pcsx_packages ('{0}.{1}' -f $pkg.id, $pkg.version) $pkgDir = Join-Path $path_pcsx_packages ('{0}.{1}' -f $pkg.id, $pkg.version)
@@ -122,24 +120,18 @@ foreach ($pkg in $required_packages.Values) {
} }
} }
# ════════════════════════════════════════════════════════════════════════════ # ════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════
# isoffi.lua size guard — `core.vcxproj` #includes src/core/isoffi.lua into # isoffi.lua size guard — `core.vcxproj` #includes src/core/isoffi.lua into luaiso.cc via the `-- lualoader, R"EOF(...)EOF"` trick.
# luaiso.cc via the `-- lualoader, R"EOF(...)EOF"` trick. The raw string # The raw string literal between R"EOF(-- and -- )EOF" must stay under ~16,379 bytes or MSVC (19.44) fails with C2026 (its actual raw-string limit is 16,384, minus 5 bytes for the `-- lualoader, ` prefix).
# literal between R"EOF(-- and -- )EOF" must stay under ~16,379 bytes or # If the upstream file grows past that, trim it: remove license header, trailing whitespace, blank separators, inline comments, and shrink 4-space indent to 2-space.
# MSVC (19.44) fails with C2026 (its actual raw-string limit is 16,384,
# minus 5 bytes for the `-- lualoader, ` prefix). If the upstream file
# grows past that, trim it: remove license header, trailing whitespace,
# blank separators, inline comments, and shrink 4-space indent to 2-space.
# Idempotent — only writes when the raw string exceeds the limit. # Idempotent — only writes when the raw string exceeds the limit.
# ════════════════════════════════════════════════════════════════════════════ # ════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════
$path_isoffi = join-path $path_pcsx_redux 'src\core\isoffi.lua' $path_isoffi = join-path $path_pcsx_redux 'src\core\isoffi.lua'
if (Test-Path -LiteralPath $path_isoffi) { if (Test-Path -LiteralPath $path_isoffi) {
$content = Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8 $content = Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8
$startMarker = $content.IndexOf('R"EOF(--') $startMarker = $content.IndexOf('R"EOF(--')
$endMarker = $content.IndexOf('-- )EOF"') $endMarker = $content.IndexOf('-- )EOF"')
$literalLen = if ($startMarker -ge 0 -and $endMarker -gt $startMarker) { $literalLen = if ($startMarker -ge 0 -and $endMarker -gt $startMarker) { $endMarker - ($startMarker + 8) } else { -1 }
$endMarker - ($startMarker + 8)
} else { -1 }
# Effective MSVC raw-string limit for the lualoader prefix is 16379 bytes. # Effective MSVC raw-string limit for the lualoader prefix is 16379 bytes.
if ($literalLen -gt 16379) { if ($literalLen -gt 16379) {
Write-Host "isoffi.lua raw string is $literalLen bytes (>16379); trimming for MSVC C2026 limit." Write-Host "isoffi.lua raw string is $literalLen bytes (>16379); trimming for MSVC C2026 limit."
@@ -168,8 +160,7 @@ if (Test-Path -LiteralPath $path_isoffi) {
$newLines += $line $newLines += $line
} }
($newLines -join "`n") | Out-File -LiteralPath $path_isoffi -Encoding utf8 -NoNewline ($newLines -join "`n") | Out-File -LiteralPath $path_isoffi -Encoding utf8 -NoNewline
$newLen = ((Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8) ` $newLen = ((Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8) -replace '.*R"EOF\(--', '' -replace '-- \)EOF".*', '').Length
-replace '.*R"EOF\(--', '' -replace '-- \)EOF".*', '').Length
Write-Host "isoffi.lua trimmed: $literalLen -> $newLen bytes of raw string content." Write-Host "isoffi.lua trimmed: $literalLen -> $newLen bytes of raw string content."
} }
} }
@@ -231,13 +222,11 @@ $lfs_dll_import = join-path $luajit_lib_dir 'libluajit-5.1.dll.a'
$path_openbios = join-path $path_pcsx_redux 'src\mips\openbios' $path_openbios = join-path $path_pcsx_redux 'src\mips\openbios'
# Wipe stale *.dep files across src\mips. These cache absolute paths to the # Wipe stale *.dep files across src\mips.
# GCC headers directory; if the toolchain was upgraded (e.g. v14.2.0 → v16.1.0) # These cache absolute paths to the GCC headers directory; if the toolchain was upgraded (e.g. v14.2.0 → v16.1.0)
# Make reads the stale paths and aborts with "no rule to make target .../stddef.h". # Make reads the stale paths and aborts with "no rule to make target .../stddef.h".
# `make clean` in openbios only clears its own dir — subdirs like # `make clean` in openbios only clears its own dir — subdirs like common/crt0/, modplayer/, and shell/ keep their stale .dep files.
# common/crt0/, modplayer/, and shell/ keep their stale .dep files. Easier to # Easier to just delete the lot before each build than to teach every Makefile about deepclean recursion.
# just delete the lot before each build than to teach every Makefile about
# deepclean recursion.
Get-ChildItem -Path (join-path $path_pcsx_redux 'src\mips') -Recurse -Filter '*.dep' -ErrorAction SilentlyContinue | Get-ChildItem -Path (join-path $path_pcsx_redux 'src\mips') -Recurse -Filter '*.dep' -ErrorAction SilentlyContinue |
ForEach-Object { Remove-Item -LiteralPath $_.FullName -Force } ForEach-Object { Remove-Item -LiteralPath $_.FullName -Force }