mirror of
https://github.com/Ed94/pikuma_ps1.git
synced 2026-09-08 17:29:05 +00:00
Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4fbf550d3c |
Vendored
-26
@@ -1,26 +0,0 @@
|
|||||||
# Cozy and Windy
|
|
||||||
|
|
||||||
Editor theme ported from the Rider scheme of the same name.
|
|
||||||
|
|
||||||
It colors the editor surface, C/C++ syntax, and tape-atom DSL keywords
|
|
||||||
emitted by `local.tape-atom-syntax`. It does not change workbench chrome.
|
|
||||||
|
|
||||||
## Install
|
|
||||||
|
|
||||||
```powershell
|
|
||||||
cd C:\projects\Pikuma\ps1\.vscode\cozy-and-windy
|
|
||||||
npm run package
|
|
||||||
code --install-extension .\cozy-and-windy-0.1.0.vsix --force
|
|
||||||
```
|
|
||||||
|
|
||||||
Reload the window. Select **Cozy and Windy** as the color theme, or set
|
|
||||||
`workbench.colorTheme` to `Cozy and Windy` in the PS1 workspace settings.
|
|
||||||
|
|
||||||
Keep `local.tape-atom-syntax` installed. This theme colors those token
|
|
||||||
types; it does not classify them.
|
|
||||||
|
|
||||||
## Inspect
|
|
||||||
|
|
||||||
Open `hello_camera.atom.c` and run **Developer: Inspect Editor Tokens and Scopes**
|
|
||||||
on `MipsAtom_`, an atom name, `atom_info`, `R_PrimCursor`, a `gte_*` call,
|
|
||||||
and a `mac_*` call.
|
|
||||||
Binary file not shown.
Vendored
-25
@@ -1,25 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "cozy-and-windy",
|
|
||||||
"displayName": "Cozy and Windy",
|
|
||||||
"description": "Editor theme ported from the Rider Cozy and Windy scheme. Colors C/C++ and tape-atom DSL keywords.",
|
|
||||||
"publisher": "local",
|
|
||||||
"version": "0.1.0",
|
|
||||||
"engines": {
|
|
||||||
"vscode": "^1.80.0"
|
|
||||||
},
|
|
||||||
"categories": [
|
|
||||||
"Themes"
|
|
||||||
],
|
|
||||||
"scripts": {
|
|
||||||
"package": "npx --yes @vscode/vsce@3.6.1 package --allow-missing-repository --skip-license --out cozy-and-windy-0.1.0.vsix"
|
|
||||||
},
|
|
||||||
"contributes": {
|
|
||||||
"themes": [
|
|
||||||
{
|
|
||||||
"label": "Cozy and Windy",
|
|
||||||
"uiTheme": "vs-dark",
|
|
||||||
"path": "./themes/cozy-and-windy-color-theme.json"
|
|
||||||
}
|
|
||||||
]
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,132 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "Cozy and Windy",
|
|
||||||
"type": "dark",
|
|
||||||
"semanticHighlighting": true,
|
|
||||||
"colors": {
|
|
||||||
// 121212
|
|
||||||
// 111212
|
|
||||||
// 211f1e
|
|
||||||
// 191817
|
|
||||||
"editor.background": "#191817",
|
|
||||||
"editor.foreground": "#dfc6ba",
|
|
||||||
"editor.lineHighlightBackground": "#1c1c1c",
|
|
||||||
"editor.selectionBackground": "#164371",
|
|
||||||
"editor.selectionForeground": "#c8c8c8",
|
|
||||||
"editorLineNumber.foreground": "#43c3c3",
|
|
||||||
"editorLineNumber.activeForeground": "#00fff4",
|
|
||||||
"editorIndentGuide.background1": "#181818",
|
|
||||||
"editorIndentGuide.activeBackground1": "#202020",
|
|
||||||
"editorRuler.foreground": "#505050",
|
|
||||||
"editorGutter.background": "#211f1e",
|
|
||||||
"editorBracketMatch.background": "#3b514d",
|
|
||||||
"editor.foldBackground": "#0c0c0c6a",
|
|
||||||
"editor.wordHighlightBackground": "#211f1e4d",
|
|
||||||
"editor.wordHighlightStrongBackground": "#303030",
|
|
||||||
"editorCursor.foreground": "#00fff4",
|
|
||||||
"editorWhitespace.foreground": "#181818",
|
|
||||||
// "editorLineHighlightBorder": "#1c1c1c",
|
|
||||||
"editorWidget.background": "#211f1e",
|
|
||||||
"editorSuggestWidget.background": "#2c334b",
|
|
||||||
"editorHoverWidget.background": "#2c334b"
|
|
||||||
},
|
|
||||||
"semanticTokenColors": {
|
|
||||||
"comment": { "foreground": "#868686", "fontStyle": "italic" },
|
|
||||||
"keyword": { "foreground": "#d8bd5b" },
|
|
||||||
"string": { "foreground": "#d46a54" },
|
|
||||||
"number": { "foreground": "#b5cea8" },
|
|
||||||
"operator": { "foreground": "#be8e78" },
|
|
||||||
"class": { "foreground": "#54a4d6" },
|
|
||||||
"struct": { "foreground": "#54a4d6" },
|
|
||||||
"enum": { "foreground": "#54a4d6" },
|
|
||||||
"type": { "foreground": "#54a4d6" },
|
|
||||||
"interface": { "foreground": "#7984ab" },
|
|
||||||
"function": { "foreground": "#cccab5" },
|
|
||||||
// "function": { "foreground": "#6090a9" },
|
|
||||||
"method": { "foreground": "#6090a9" },
|
|
||||||
"variable": { "foreground": "#bc966c" },
|
|
||||||
"parameter": { "foreground": "#ce8365" },
|
|
||||||
"property": { "foreground": "#acb8c8" },
|
|
||||||
"*.static": { "foreground": "#9e95c6" },
|
|
||||||
"macro": { "foreground": "#5ea852" },
|
|
||||||
"namespace": { "foreground": "#8e8e8e" },
|
|
||||||
"typeParameter": { "foreground": "#b8d7a3" },
|
|
||||||
"enumMember": { "foreground": "#a373b0" },
|
|
||||||
"label": { "foreground": "#c8c8c8", "fontStyle": "bold" },
|
|
||||||
"tapeAtomKeyword": { "foreground": "#d8bd5b", "fontStyle": "bold" },
|
|
||||||
"tapeAtomName": { "foreground": "#b1b7d6", "fontStyle": "bold" },
|
|
||||||
"tapeComponentKeyword": { "foreground": "#d68a36", "fontStyle": "bold" },
|
|
||||||
"tapeComponentName": { "foreground": "#b1b7d6", "fontStyle": "bold" },
|
|
||||||
"tapeAnnotation": { "foreground": "#d8bd5b" },
|
|
||||||
"tapeBindType": { "foreground": "#54a4d6" },
|
|
||||||
"tapePhase": { "foreground": "#b8d7a3", "fontStyle": "italic" },
|
|
||||||
"tapeLabel": { "foreground": "#959595", "fontStyle": "bold" },
|
|
||||||
// "tapeCpuInstruction": { "foreground": "#6d9aa0" },
|
|
||||||
// "tapeCpuInstruction": { "foreground": "#cf7539" },
|
|
||||||
// "tapeCpuInstruction": { "foreground": "#d16b3a" },
|
|
||||||
"tapeCpuInstruction": { "foreground": "#d5895a" },
|
|
||||||
"tapeGteInstruction": { "foreground": "#988bcb" },
|
|
||||||
"tapeGpuInstruction": { "foreground": "#bf7dac" },
|
|
||||||
"tapeComponentInstruction": { "foreground": "#8baa5d" },
|
|
||||||
// "tapeGprRegister": { "foreground": "#92d4d9" },
|
|
||||||
"tapeGprRegister": { "foreground": "#a2bfa8" },
|
|
||||||
"tapeCop2Register": { "foreground": "#945cd9" },
|
|
||||||
"tapeDuffleType": { "foreground": "#54a4d6" },
|
|
||||||
"tapeAttribute": { "foreground": "#73a07c" },
|
|
||||||
"tapeGprRegister.tapeRead": { "foreground": "#5bb8b0", "fontStyle": "italic" },
|
|
||||||
"tapeGprRegister.tapeWrite": { "foreground": "#2d8f8c", "fontStyle": "bold" },
|
|
||||||
"tapeCop2Register.tapeRead": { "foreground": "#b08ae0", "fontStyle": "italic" },
|
|
||||||
"tapeCop2Register.tapeWrite": { "foreground": "#7b3ec4", "fontStyle": "bold" },
|
|
||||||
// "*.tapeAuto": { },
|
|
||||||
"tapeControlFlow": { "foreground": "#63d169", "fontStyle": "bold" },
|
|
||||||
"tapeDelaySlot": { "foreground": "#ff5647" }
|
|
||||||
},
|
|
||||||
"tokenColors": [
|
|
||||||
{ "scope": ["comment", "comment.block", "comment.line", "comment.block.documentation"], "settings": { "foreground": "#868686", "fontStyle": "italic" } },
|
|
||||||
{ "scope": ["keyword", "keyword.control", "keyword.other"], "settings": { "foreground": "#d8bd5b" } },
|
|
||||||
{ "scope": ["string", "string.quoted"], "settings": { "foreground": "#d46a54" } },
|
|
||||||
{ "scope": ["string.quoted.other"], "settings": { "foreground": "#d69d85" } },
|
|
||||||
{ "scope": ["constant.numeric"], "settings": { "foreground": "#b5cea8" } },
|
|
||||||
{ "scope": ["punctuation", "keyword.operator"], "settings": { "foreground": "#be8e78" } },
|
|
||||||
{ "scope": ["keyword.operator.overload"], "settings": { "foreground": "#b87e76" } },
|
|
||||||
{ "scope": ["entity.name.type", "entity.name.type.class", "entity.name.type.struct", "entity.name.type.enum"], "settings": { "foreground": "#54a4d6" } },
|
|
||||||
{ "scope": ["entity.name.type.interface"], "settings": { "foreground": "#7984ab" } },
|
|
||||||
{ "scope": ["entity.name.function"], "settings": { "foreground": "#cccab5" } },
|
|
||||||
{ "scope": ["entity.name.function.member"], "settings": { "foreground": "#6090a9" } },
|
|
||||||
{ "scope": ["variable.other.local"], "settings": { "foreground": "#bc966c" } },
|
|
||||||
{ "scope": ["variable.parameter"], "settings": { "foreground": "#ce8365" } },
|
|
||||||
{ "scope": ["variable.other.property"], "settings": { "foreground": "#acb8c8" } },
|
|
||||||
{ "scope": ["variable.other.constant"], "settings": { "foreground": "#9e95c6" } },
|
|
||||||
{ "scope": ["variable.other.global", "variable.other.defaultLibrary"], "settings": { "foreground": "#bf7dac" } },
|
|
||||||
{ "scope": ["support.type", "support.function"], "settings": { "foreground": "#8baa5d" } },
|
|
||||||
{ "scope": ["meta.preprocessor"], "settings": { "foreground": "#5ea852" } },
|
|
||||||
// { "scope": ["variable.parameter.preprocessor"], "settings": { "foreground": "#636363" } },
|
|
||||||
{ "scope": ["variable.parameter.preprocessor"], "settings": { "foreground": "#bc966c" } },
|
|
||||||
{ "scope": ["keyword.control.directive"], "settings": { "foreground": "#d68a36" } },
|
|
||||||
{ "scope": ["entity.name.namespace"], "settings": { "foreground": "#8e8e8e" } },
|
|
||||||
{ "scope": ["entity.name.type.parameter"], "settings": { "foreground": "#b8d7a3" } },
|
|
||||||
{ "scope": ["variable.other.enummember"], "settings": { "foreground": "#a373b0" } },
|
|
||||||
{ "scope": ["entity.name.type.concept"], "settings": { "foreground": "#76ff7d" } },
|
|
||||||
{ "scope": ["entity.name.type.dependent"], "settings": { "foreground": "#448b5a", "fontStyle": "bold" } },
|
|
||||||
{ "scope": ["entity.name.label"], "settings": { "foreground": "#c8c8c8", "fontStyle": "bold" } },
|
|
||||||
{ "scope": ["invalid"], "settings": { "foreground": "#ff5647" } },
|
|
||||||
{ "scope": ["keyword.codetag.todo"], "settings": { "foreground": "#c10000", "fontStyle": "bold italic" } },
|
|
||||||
{ "scope": ["keyword.control.duffle.atom"], "settings": { "foreground": "#d8bd5b", "fontStyle": "bold" } },
|
|
||||||
{ "scope": ["entity.name.function.duffle.atom"], "settings": { "foreground": "#cccab5", "fontStyle": "bold" } },
|
|
||||||
{ "scope": ["keyword.control.duffle.component"], "settings": { "foreground": "#d68a36", "fontStyle": "bold" } },
|
|
||||||
{ "scope": ["entity.name.function.duffle.component"], "settings": { "foreground": "#6090a9", "fontStyle": "bold" } },
|
|
||||||
{ "scope": ["support.function.duffle.annotation"], "settings": { "foreground": "#d8bd5b" } },
|
|
||||||
{ "scope": ["entity.name.type.duffle.bind"], "settings": { "foreground": "#54a4d6" } },
|
|
||||||
{ "scope": ["entity.name.tag.duffle.phase"], "settings": { "foreground": "#b8d7a3", "fontStyle": "italic" } },
|
|
||||||
{ "scope": ["entity.name.label.duffle.atom"], "settings": { "foreground": "#c8c8c8", "fontStyle": "bold" } },
|
|
||||||
{ "scope": ["support.function.duffle.cpu"], "settings": { "foreground": "#6d9aa0" } },
|
|
||||||
{ "scope": ["support.function.duffle.gte"], "settings": { "foreground": "#988bcb" } },
|
|
||||||
{ "scope": ["support.function.duffle.gpu"], "settings": { "foreground": "#bf7dac" } },
|
|
||||||
{ "scope": ["support.function.duffle.component"], "settings": { "foreground": "#8baa5d" } },
|
|
||||||
{ "scope": ["keyword.control.duffle.branch"], "settings": { "foreground": "#76ff7d", "fontStyle": "bold" } },
|
|
||||||
{ "scope": ["keyword.operator.duffle.delayslot"], "settings": { "foreground": "#ff5647" } },
|
|
||||||
{ "scope": ["variable.other.constant.duffle.gpr"], "settings": { "foreground": "#3fa8a6" } },
|
|
||||||
{ "scope": ["variable.other.constant.duffle.cop2"], "settings": { "foreground": "#945cd9" } },
|
|
||||||
{ "scope": ["storage.type.duffle.type"], "settings": { "foreground": "#54a4d6" } },
|
|
||||||
{ "scope": ["storage.modifier.duffle.attr"], "settings": { "foreground": "#73a07c" } }
|
|
||||||
]
|
|
||||||
}
|
|
||||||
Vendored
-43
@@ -1,43 +0,0 @@
|
|||||||
# Package and install the local VS Code Insiders extensions under .vscode/.
|
|
||||||
# Usage:
|
|
||||||
# .\install_extensions.ps1
|
|
||||||
# .\install_extensions.ps1 -SkipPackage
|
|
||||||
|
|
||||||
param([switch] $SkipPackage)
|
|
||||||
|
|
||||||
$path_vscode = $PSScriptRoot
|
|
||||||
$code_insiders = "C:\apps\Microsoft VS Code Insiders\bin\code-insiders.cmd"
|
|
||||||
if (-not (test-path -literalpath $code_insiders)) {
|
|
||||||
$found = get-command code-insiders -erroraction silentlycontinue
|
|
||||||
if ($found) { $code_insiders = $found.source }
|
|
||||||
}
|
|
||||||
|
|
||||||
if (-not (test-path -literalpath $code_insiders)) { throw "code-insiders not found. Install VS Code Insiders or add it to PATH." }
|
|
||||||
|
|
||||||
$extensions = @(
|
|
||||||
(join-path $path_vscode "tape-atom-syntax"),
|
|
||||||
(join-path $path_vscode "cozy-and-windy")
|
|
||||||
)
|
|
||||||
|
|
||||||
foreach ($extension in $extensions) {
|
|
||||||
$package_json = join-path $extension "package.json"
|
|
||||||
if (-not (test-path -literalpath $package_json)) { throw "missing $package_json" }
|
|
||||||
|
|
||||||
$manifest = get-content -literalpath $package_json -raw | convertfrom-json
|
|
||||||
$vsix = join-path $extension ("{0}-{1}.vsix" -f $manifest.name, $manifest.version)
|
|
||||||
|
|
||||||
if (-not $SkipPackage) {
|
|
||||||
if (-not $manifest.scripts.package) { throw "$package_json has no scripts.package" }
|
|
||||||
write-host "packaging $($manifest.displayName) ($($manifest.name)@$($manifest.version))"
|
|
||||||
& npm --prefix $extension run package
|
|
||||||
if ($LASTEXITCODE -ne 0) { throw "npm run package failed for $extension" }
|
|
||||||
}
|
|
||||||
|
|
||||||
if (-not (test-path -literalpath $vsix)) { throw "missing $vsix" }
|
|
||||||
|
|
||||||
write-host "installing $vsix"
|
|
||||||
& $code_insiders --install-extension $vsix --force
|
|
||||||
if ($LASTEXITCODE -ne 0) { throw "install failed for $vsix" }
|
|
||||||
}
|
|
||||||
|
|
||||||
write-host "done. reload the Insiders window (Developer: Reload Window)."
|
|
||||||
Vendored
+33
@@ -177,6 +177,39 @@
|
|||||||
"tbreak main",
|
"tbreak main",
|
||||||
"continue"
|
"continue"
|
||||||
]
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"name": "Debug: Hello Camera! (attach only)",
|
||||||
|
"type": "gdb",
|
||||||
|
"request": "attach",
|
||||||
|
"target": "localhost:3333",
|
||||||
|
"remote": true,
|
||||||
|
"cwd": "${workspaceRoot}",
|
||||||
|
"valuesFormatting": "parseText",
|
||||||
|
"registerLimit": "1-32",
|
||||||
|
"frameFilters": false,
|
||||||
|
"showDevDebugOutput": false,
|
||||||
|
"printCalls": false,
|
||||||
|
"stopAtConnect": true,
|
||||||
|
"gdbpath": "gdb-multiarch",
|
||||||
|
"windows": {
|
||||||
|
"gdbpath": "gdb-multiarch.exe"
|
||||||
|
},
|
||||||
|
"osx": {
|
||||||
|
"gdbpath": "gdb"
|
||||||
|
},
|
||||||
|
"executable": "${workspaceRoot}/build/hello_camera.dwarf-injected.elf",
|
||||||
|
"setupCommands": [
|
||||||
|
{ "text": "set mi-async off" },
|
||||||
|
{ "text": "set remotetimeout 0" },
|
||||||
|
{ "text": "set logging file build/gen/hello_camera.gdb.log" },
|
||||||
|
{ "text": "set logging redirect on" }
|
||||||
|
],
|
||||||
|
"autorun": [
|
||||||
|
"source scripts/gdb/gdb_tape_atoms.gdb",
|
||||||
|
"tbreak hot_reload_entry",
|
||||||
|
"continue"
|
||||||
|
]
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
|
|||||||
BIN
Binary file not shown.
-222
@@ -1,222 +0,0 @@
|
|||||||
"use strict";
|
|
||||||
|
|
||||||
const { nearestCall } = require("./lexer");
|
|
||||||
const { mergeIndexes, scanSource } = require("./source-index");
|
|
||||||
|
|
||||||
const TOKEN_TYPES = [
|
|
||||||
"tapeAtomKeyword",
|
|
||||||
"tapeAtomName",
|
|
||||||
"tapeComponentKeyword",
|
|
||||||
"tapeComponentName",
|
|
||||||
"tapeAnnotation",
|
|
||||||
"tapeBindType",
|
|
||||||
"tapePhase",
|
|
||||||
"tapeLabel",
|
|
||||||
"tapeCpuInstruction",
|
|
||||||
"tapeControlFlow",
|
|
||||||
"tapeGteInstruction",
|
|
||||||
"tapeGpuInstruction",
|
|
||||||
"tapeComponentInstruction",
|
|
||||||
"tapeDelaySlot",
|
|
||||||
"tapeGprRegister",
|
|
||||||
"tapeCop2Register",
|
|
||||||
"tapeDuffleType",
|
|
||||||
"tapeAttribute",
|
|
||||||
"keyword",
|
|
||||||
"macro",
|
|
||||||
];
|
|
||||||
|
|
||||||
const TOKEN_MODIFIERS = ["declaration", "tapeRead", "tapeWrite", "tapeAuto"];
|
|
||||||
const TOKEN_TYPE_INDEX = new Map(TOKEN_TYPES.map((name, index) => [name, index]));
|
|
||||||
const TOKEN_MODIFIER_INDEX = new Map(TOKEN_MODIFIERS.map((name, index) => [name, index]));
|
|
||||||
|
|
||||||
const ATOM_KEYWORDS = new Set(["MipsAtom_", "MipsAtom_Proc_"]);
|
|
||||||
const COMPONENT_KEYWORDS = new Set(["MipsAtomComp_", "MipsAtomComp_Proc_"]);
|
|
||||||
const ANNOTATIONS = new Set([
|
|
||||||
"atom_info", "atom_bind", "atom_reads", "atom_writes", "atom_label",
|
|
||||||
"atom_offset", "atom_reg", "atom_type", "atom_ctx", "atom_phase",
|
|
||||||
"atom_auto_reg", "phase_auto_reg", "atom_dbg_skip",
|
|
||||||
]);
|
|
||||||
|
|
||||||
const DSL_KEYWORDS = new Set([
|
|
||||||
"FI_", "I_", "NI_", "Relative_", "Struct_", "Enum_", "Union_", "Array_",
|
|
||||||
"Slice_", "TypeR_", "TypeV_", "align_", "internal", "local_persist", "global",
|
|
||||||
"RO_", "LP_", "gknown", "expect_", "cexpr_",
|
|
||||||
"asm", "asm_words", "asm_rpins", "asm_clobber",
|
|
||||||
"O_", "S_", "C_", "T_", "tmpl", "glue", "r_", "v_", "tr_", "tv_",
|
|
||||||
"rgcc", "r_use", "r_set", "r_mod", "r_imm", "r_mem",
|
|
||||||
"u1_", "u2_", "u4_", "u8_", "s1_", "s2_", "s4_", "s8_",
|
|
||||||
"u1_r", "u2_r", "u4_r", "u8_r", "u1_v", "u2_v", "u4_v", "u8_v",
|
|
||||||
]);
|
|
||||||
|
|
||||||
const DELAY_SLOT_KEYWORDS = new Set(["LdSlot_", "BdSlot_", "DmaSlot_", "GteDelay_"]);
|
|
||||||
|
|
||||||
const CONTROL_FLOW_PREFIXES = /^(?:branch_|jump_|call_)/;
|
|
||||||
|
|
||||||
const ROLE_TO_TYPE = {
|
|
||||||
atomName: "tapeAtomName",
|
|
||||||
componentName: "tapeComponentName",
|
|
||||||
bindType: "tapeBindType",
|
|
||||||
duffleType: "tapeDuffleType",
|
|
||||||
gprRegister: "tapeGprRegister",
|
|
||||||
cop2Register: "tapeCop2Register",
|
|
||||||
};
|
|
||||||
|
|
||||||
function registerType(name, index) {
|
|
||||||
const kind = index.registers.get(name);
|
|
||||||
if (kind === "gpr" || /^R_[A-Za-z0-9_]+$/.test(name)) return "tapeGprRegister";
|
|
||||||
if (kind === "cop2" || /^(?:C2_|gte_cr_)[A-Za-z0-9_]+$/.test(name)) return "tapeCop2Register";
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
function instructionType(name, index) {
|
|
||||||
const domain = index.macros.get(name);
|
|
||||||
if (domain === "control") return "tapeControlFlow";
|
|
||||||
if (domain === "cpu") return "tapeCpuInstruction";
|
|
||||||
if (domain === "gte") return "tapeGteInstruction";
|
|
||||||
if (domain === "gpu") return "tapeGpuInstruction";
|
|
||||||
if (domain === "component") {
|
|
||||||
if (/^mac_gte_/.test(name)) return "tapeGteInstruction";
|
|
||||||
if (/^mac_gp/.test(name)) return "tapeGpuInstruction";
|
|
||||||
if (/^mac_/.test(name)) return "tapeComponentInstruction";
|
|
||||||
return "macro";
|
|
||||||
}
|
|
||||||
if (domain === "utility") return "macro";
|
|
||||||
if (/^gte_(?!cr_)/.test(name)) return "tapeGteInstruction";
|
|
||||||
if (/^gp[01]_/.test(name)) return "tapeGpuInstruction";
|
|
||||||
if (/^mac_gte_/.test(name)) return "tapeGteInstruction";
|
|
||||||
if (/^mac_gp/.test(name)) return "tapeGpuInstruction";
|
|
||||||
if (/^mac_/.test(name)) return "tapeComponentInstruction";
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
function modifierMask(modifiers) {
|
|
||||||
let mask = 0;
|
|
||||||
for (const modifier of modifiers) {
|
|
||||||
const index = TOKEN_MODIFIER_INDEX.get(modifier);
|
|
||||||
if (index !== undefined) mask |= (1 << index);
|
|
||||||
}
|
|
||||||
return mask;
|
|
||||||
}
|
|
||||||
|
|
||||||
function isRegUseAccess(tokens, tokenIndex) {
|
|
||||||
const prev = tokens[tokenIndex - 1];
|
|
||||||
if (!prev || prev.text !== ".") return false;
|
|
||||||
const prevPrev = tokens[tokenIndex - 2];
|
|
||||||
if (!prevPrev || prevPrev.kind !== "identifier") return false;
|
|
||||||
const next = tokens[tokenIndex + 1];
|
|
||||||
if (next && next.text === ".") return false;
|
|
||||||
if (prevPrev.text === "r") return true;
|
|
||||||
const prev3 = tokens[tokenIndex - 3];
|
|
||||||
const prev4 = tokens[tokenIndex - 4];
|
|
||||||
if (prev3 && prev3.text === "." && prev4 && prev4.kind === "identifier" && prev4.text === "r") return true;
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
function classifyDocument(source, filePath, workspaceIndex, shouldCancel = () => false) {
|
|
||||||
const scanned = scanSource(source, filePath);
|
|
||||||
const index = mergeIndexes(workspaceIndex, scanned.index);
|
|
||||||
const spans = [];
|
|
||||||
|
|
||||||
for (let tokenIndex = 0; tokenIndex < scanned.tokens.length; tokenIndex += 1) {
|
|
||||||
if (shouldCancel()) break;
|
|
||||||
const token = scanned.tokens[tokenIndex];
|
|
||||||
if (token.kind !== "identifier") continue;
|
|
||||||
|
|
||||||
let type = null;
|
|
||||||
let modifiers = [];
|
|
||||||
const declaration = scanned.declarations.get(token.start);
|
|
||||||
const context = nearestCall(scanned.contexts, tokenIndex);
|
|
||||||
|
|
||||||
if (declaration) {
|
|
||||||
type = ROLE_TO_TYPE[declaration.role] || null;
|
|
||||||
modifiers = declaration.modifiers.slice();
|
|
||||||
} else if (ATOM_KEYWORDS.has(token.text)) {
|
|
||||||
type = "tapeAtomKeyword";
|
|
||||||
} else if (COMPONENT_KEYWORDS.has(token.text)) {
|
|
||||||
type = "keyword";
|
|
||||||
} else if (ANNOTATIONS.has(token.text)) {
|
|
||||||
type = "tapeAnnotation";
|
|
||||||
} else if (context && context.callee === "atom_bind" && context.argIndex === 0) {
|
|
||||||
type = "tapeBindType";
|
|
||||||
} else if (context && context.callee === "atom_phase" && context.argIndex === 0) {
|
|
||||||
type = "tapePhase";
|
|
||||||
modifiers = ["declaration"];
|
|
||||||
} else if (context && context.callee === "atom_ctx" && context.argIndex === 0) {
|
|
||||||
type = "tapeAtomName";
|
|
||||||
} else if (context && context.callee === "atom_label" && context.argIndex === 0) {
|
|
||||||
type = "tapeLabel";
|
|
||||||
modifiers = ["declaration"];
|
|
||||||
} else if (context && context.callee === "atom_offset" && context.argIndex <= 1) {
|
|
||||||
type = "tapeLabel";
|
|
||||||
} else if (context && context.callee === "atom_reads") {
|
|
||||||
type = registerType(token.text, index);
|
|
||||||
if (type) modifiers = ["tapeRead"];
|
|
||||||
} else if (context && context.callee === "atom_writes") {
|
|
||||||
type = registerType(token.text, index);
|
|
||||||
if (type) modifiers = ["tapeWrite"];
|
|
||||||
} else if (context && context.callee === "atom_auto_reg") {
|
|
||||||
if (context.argIndex === 0) type = "tapeAtomName";
|
|
||||||
if (context.argIndex === 1) {
|
|
||||||
type = "tapeGprRegister";
|
|
||||||
modifiers = ["declaration", "tapeAuto"];
|
|
||||||
}
|
|
||||||
} else if (context && context.callee === "phase_auto_reg") {
|
|
||||||
if (context.argIndex === 0) type = "tapePhase";
|
|
||||||
if (context.argIndex === 1) {
|
|
||||||
type = "tapeGprRegister";
|
|
||||||
modifiers = ["declaration", "tapeAuto"];
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if (!type && index.bindTypes.has(token.text)) type = "tapeBindType";
|
|
||||||
if (!type && DSL_KEYWORDS.has(token.text)) type = "keyword";
|
|
||||||
if (!type && index.types.has(token.text)) type = "tapeDuffleType";
|
|
||||||
if (!type && index.attributes.has(token.text)) type = "tapeAttribute";
|
|
||||||
if (!type) type = registerType(token.text, index);
|
|
||||||
if (!type && DELAY_SLOT_KEYWORDS.has(token.text)) type = "tapeDelaySlot";
|
|
||||||
if (!type) {
|
|
||||||
const domain = index.macros.get(token.text);
|
|
||||||
if (domain === "control" || (domain && CONTROL_FLOW_PREFIXES.test(token.text))) {
|
|
||||||
type = "tapeControlFlow";
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (!type && isRegUseAccess(scanned.tokens, tokenIndex)) type = "tapeGprRegister";
|
|
||||||
if (!type) type = instructionType(token.text, index);
|
|
||||||
if (!type && /^(?:Slice_|A[0-9]+_)/.test(token.text)) type = "tapeDuffleType";
|
|
||||||
if (!type && /_[RV]$/.test(token.text)) type = "tapeDuffleType";
|
|
||||||
if (!type && index.atoms.has(token.text)) type = "tapeAtomName";
|
|
||||||
if (!type && index.components.has(token.text)) type = "tapeComponentName";
|
|
||||||
if (!type && index.phases.has(token.text)) type = "tapePhase";
|
|
||||||
if (!type && index.labels.has(token.text)) type = "tapeLabel";
|
|
||||||
if (!type) continue;
|
|
||||||
|
|
||||||
spans.push({
|
|
||||||
text: token.text,
|
|
||||||
type,
|
|
||||||
typeIndex: TOKEN_TYPE_INDEX.get(type),
|
|
||||||
modifiers,
|
|
||||||
modifierMask: modifierMask(modifiers),
|
|
||||||
start: token.start,
|
|
||||||
length: token.end - token.start,
|
|
||||||
line: token.line,
|
|
||||||
character: token.character,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
spans.sort((left, right) => left.start - right.start || left.length - right.length);
|
|
||||||
const nonOverlapping = [];
|
|
||||||
for (const span of spans) {
|
|
||||||
const previous = nonOverlapping[nonOverlapping.length - 1];
|
|
||||||
if (!previous || previous.start + previous.length <= span.start) nonOverlapping.push(span);
|
|
||||||
}
|
|
||||||
|
|
||||||
return { spans: nonOverlapping, errors: scanned.errors };
|
|
||||||
}
|
|
||||||
|
|
||||||
module.exports = {
|
|
||||||
TOKEN_MODIFIERS,
|
|
||||||
TOKEN_TYPES,
|
|
||||||
classifyDocument,
|
|
||||||
modifierMask,
|
|
||||||
};
|
|
||||||
-111
@@ -1,111 +0,0 @@
|
|||||||
"use strict";
|
|
||||||
|
|
||||||
const vscode = require("vscode");
|
|
||||||
const { TOKEN_MODIFIERS, TOKEN_TYPES, classifyDocument } = require("./classifier");
|
|
||||||
const { createIndex, mergeIndexes, scanSource } = require("./source-index");
|
|
||||||
|
|
||||||
const SOURCE_GLOB = "**/*.{c,h,cc,cpp,cxx,hh,hpp,hxx}";
|
|
||||||
const EXCLUDE_GLOB = "**/{gen,build,.slop_cache,toolchain,node_modules}/**";
|
|
||||||
const EXCLUDED_SEGMENTS = new Set(["gen", "build", ".slop_cache", "toolchain", "node_modules"]);
|
|
||||||
|
|
||||||
function isExcluded(uri) {
|
|
||||||
const segments = uri.fsPath.replaceAll("\\", "/").split("/");
|
|
||||||
return segments.some((segment) => EXCLUDED_SEGMENTS.has(segment));
|
|
||||||
}
|
|
||||||
|
|
||||||
function formatError(filePath, error) {
|
|
||||||
return `${filePath}:${error.offset}: ${error.kind}`;
|
|
||||||
}
|
|
||||||
|
|
||||||
async function activate(context) {
|
|
||||||
const output = vscode.window.createOutputChannel("Tape Atom DSL");
|
|
||||||
const emitter = new vscode.EventEmitter();
|
|
||||||
const legend = new vscode.SemanticTokensLegend(TOKEN_TYPES, TOKEN_MODIFIERS);
|
|
||||||
let workspaceIndex = createIndex();
|
|
||||||
let rebuildGeneration = 0;
|
|
||||||
let debounceHandle = null;
|
|
||||||
|
|
||||||
async function rebuildIndex() {
|
|
||||||
const generation = ++rebuildGeneration;
|
|
||||||
const files = await vscode.workspace.findFiles(SOURCE_GLOB, EXCLUDE_GLOB);
|
|
||||||
let nextIndex = createIndex();
|
|
||||||
|
|
||||||
for (const uri of files) {
|
|
||||||
if (generation !== rebuildGeneration) return;
|
|
||||||
if (isExcluded(uri)) continue;
|
|
||||||
try {
|
|
||||||
const bytes = await vscode.workspace.fs.readFile(uri);
|
|
||||||
const source = Buffer.from(bytes).toString("utf8");
|
|
||||||
const result = scanSource(source, uri.fsPath);
|
|
||||||
nextIndex = mergeIndexes(nextIndex, result.index);
|
|
||||||
for (const error of result.errors) output.appendLine(formatError(uri.fsPath, error));
|
|
||||||
} catch (error) {
|
|
||||||
output.appendLine(`${uri.fsPath}: ${error.stack || error.message || error}`);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if (generation !== rebuildGeneration) return;
|
|
||||||
workspaceIndex = nextIndex;
|
|
||||||
emitter.fire();
|
|
||||||
}
|
|
||||||
|
|
||||||
function scheduleRebuild(uri) {
|
|
||||||
if (uri && isExcluded(uri)) return;
|
|
||||||
if (debounceHandle !== null) clearTimeout(debounceHandle);
|
|
||||||
debounceHandle = setTimeout(() => {
|
|
||||||
debounceHandle = null;
|
|
||||||
rebuildIndex().catch((error) => output.appendLine(error.stack || String(error)));
|
|
||||||
}, 100);
|
|
||||||
}
|
|
||||||
|
|
||||||
const provider = {
|
|
||||||
onDidChangeSemanticTokens: emitter.event,
|
|
||||||
provideDocumentSemanticTokens(document, cancellationToken) {
|
|
||||||
try {
|
|
||||||
const result = classifyDocument(
|
|
||||||
document.getText(),
|
|
||||||
document.uri.fsPath,
|
|
||||||
workspaceIndex,
|
|
||||||
() => cancellationToken.isCancellationRequested
|
|
||||||
);
|
|
||||||
const builder = new vscode.SemanticTokensBuilder(legend);
|
|
||||||
for (const span of result.spans) {
|
|
||||||
if (cancellationToken.isCancellationRequested) break;
|
|
||||||
builder.push(span.line, span.character, span.length, span.typeIndex, span.modifierMask);
|
|
||||||
}
|
|
||||||
for (const error of result.errors) {
|
|
||||||
output.appendLine(formatError(document.uri.fsPath || document.uri.toString(), error));
|
|
||||||
}
|
|
||||||
return builder.build();
|
|
||||||
} catch (error) {
|
|
||||||
output.appendLine(`${document.uri}: ${error.stack || error.message || error}`);
|
|
||||||
return new vscode.SemanticTokensBuilder(legend).build();
|
|
||||||
}
|
|
||||||
},
|
|
||||||
};
|
|
||||||
|
|
||||||
const selector = [
|
|
||||||
{ language: "c", scheme: "file" },
|
|
||||||
{ language: "c", scheme: "untitled" },
|
|
||||||
{ language: "cpp", scheme: "file" },
|
|
||||||
{ language: "cpp", scheme: "untitled" },
|
|
||||||
];
|
|
||||||
const watcher = vscode.workspace.createFileSystemWatcher(SOURCE_GLOB);
|
|
||||||
|
|
||||||
context.subscriptions.push(
|
|
||||||
output,
|
|
||||||
emitter,
|
|
||||||
watcher,
|
|
||||||
watcher.onDidCreate(scheduleRebuild),
|
|
||||||
watcher.onDidChange(scheduleRebuild),
|
|
||||||
watcher.onDidDelete(scheduleRebuild),
|
|
||||||
vscode.languages.registerDocumentSemanticTokensProvider(selector, provider, legend),
|
|
||||||
{ dispose() { if (debounceHandle !== null) clearTimeout(debounceHandle); } }
|
|
||||||
);
|
|
||||||
|
|
||||||
await rebuildIndex();
|
|
||||||
}
|
|
||||||
|
|
||||||
function deactivate() {}
|
|
||||||
|
|
||||||
module.exports = { activate, deactivate };
|
|
||||||
Vendored
-186
@@ -1,186 +0,0 @@
|
|||||||
"use strict";
|
|
||||||
|
|
||||||
function isIdentifierStart(code) {
|
|
||||||
return code === 95 ||
|
|
||||||
(code >= 65 && code <= 90) ||
|
|
||||||
(code >= 97 && code <= 122);
|
|
||||||
}
|
|
||||||
|
|
||||||
function isIdentifierContinue(code) {
|
|
||||||
return isIdentifierStart(code) || (code >= 48 && code <= 57);
|
|
||||||
}
|
|
||||||
|
|
||||||
function lex(source) {
|
|
||||||
if (typeof source !== "string") throw new TypeError("source must be a string");
|
|
||||||
|
|
||||||
const tokens = [];
|
|
||||||
const errors = [];
|
|
||||||
let offset = 0;
|
|
||||||
let line = 0;
|
|
||||||
let character = 0;
|
|
||||||
|
|
||||||
function advance() {
|
|
||||||
if (source[offset] === "\r" && source[offset + 1] === "\n") {
|
|
||||||
offset += 2;
|
|
||||||
line += 1;
|
|
||||||
character = 0;
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (source[offset] === "\n") {
|
|
||||||
offset += 1;
|
|
||||||
line += 1;
|
|
||||||
character = 0;
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
offset += 1;
|
|
||||||
character += 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
function pushToken(kind, start, startLine, startCharacter) {
|
|
||||||
tokens.push({
|
|
||||||
kind,
|
|
||||||
text: source.slice(start, offset),
|
|
||||||
start,
|
|
||||||
end: offset,
|
|
||||||
line: startLine,
|
|
||||||
character: startCharacter,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
while (offset < source.length) {
|
|
||||||
const ch = source[offset];
|
|
||||||
|
|
||||||
if (/\s/.test(ch)) {
|
|
||||||
advance();
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (ch === "/" && source[offset + 1] === "/") {
|
|
||||||
while (offset < source.length && source[offset] !== "\r" && source[offset] !== "\n") advance();
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (ch === "/" && source[offset + 1] === "*") {
|
|
||||||
const start = offset;
|
|
||||||
advance();
|
|
||||||
advance();
|
|
||||||
let closed = false;
|
|
||||||
while (offset < source.length) {
|
|
||||||
if (source[offset] === "*" && source[offset + 1] === "/") {
|
|
||||||
advance();
|
|
||||||
advance();
|
|
||||||
closed = true;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
advance();
|
|
||||||
}
|
|
||||||
if (!closed) errors.push({ kind: "unterminated-block-comment", offset: start });
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (ch === "\"" || ch === "'") {
|
|
||||||
const quote = ch;
|
|
||||||
const start = offset;
|
|
||||||
advance();
|
|
||||||
let closed = false;
|
|
||||||
while (offset < source.length) {
|
|
||||||
if (source[offset] === "\\") {
|
|
||||||
advance();
|
|
||||||
if (offset < source.length) advance();
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
if (source[offset] === quote) {
|
|
||||||
advance();
|
|
||||||
closed = true;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
if (source[offset] === "\n" || source[offset] === "\r") break;
|
|
||||||
advance();
|
|
||||||
}
|
|
||||||
if (!closed) errors.push({ kind: "unterminated-literal", offset: start });
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
const code = source.charCodeAt(offset);
|
|
||||||
if (isIdentifierStart(code)) {
|
|
||||||
const start = offset;
|
|
||||||
const startLine = line;
|
|
||||||
const startCharacter = character;
|
|
||||||
advance();
|
|
||||||
while (offset < source.length && isIdentifierContinue(source.charCodeAt(offset))) advance();
|
|
||||||
pushToken("identifier", start, startLine, startCharacter);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
const start = offset;
|
|
||||||
const startLine = line;
|
|
||||||
const startCharacter = character;
|
|
||||||
advance();
|
|
||||||
pushToken("punctuation", start, startLine, startCharacter);
|
|
||||||
}
|
|
||||||
|
|
||||||
return { tokens, errors };
|
|
||||||
}
|
|
||||||
|
|
||||||
function buildCallContexts(tokens) {
|
|
||||||
const contexts = Array.from({ length: tokens.length }, () => []);
|
|
||||||
const calls = [];
|
|
||||||
const errors = [];
|
|
||||||
const stack = [];
|
|
||||||
|
|
||||||
for (let tokenIndex = 0; tokenIndex < tokens.length; tokenIndex += 1) {
|
|
||||||
const token = tokens[tokenIndex];
|
|
||||||
|
|
||||||
if (token.text === ")") {
|
|
||||||
if (stack.length === 0) {
|
|
||||||
errors.push({ kind: "unmatched-close-paren", offset: token.start });
|
|
||||||
} else {
|
|
||||||
const frame = stack.pop();
|
|
||||||
if (frame.callee !== null) calls.push({ ...frame, closeTokenIndex: tokenIndex });
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
contexts[tokenIndex] = stack
|
|
||||||
.filter((frame) => frame.callee !== null)
|
|
||||||
.map((frame) => ({
|
|
||||||
callee: frame.callee,
|
|
||||||
calleeTokenIndex: frame.calleeTokenIndex,
|
|
||||||
openTokenIndex: frame.openTokenIndex,
|
|
||||||
argIndex: frame.argIndex,
|
|
||||||
}));
|
|
||||||
|
|
||||||
if (token.text === "(") {
|
|
||||||
const previous = tokens[tokenIndex - 1];
|
|
||||||
const hasCallee = previous && previous.kind === "identifier";
|
|
||||||
stack.push({
|
|
||||||
callee: hasCallee ? previous.text : null,
|
|
||||||
calleeTokenIndex: hasCallee ? tokenIndex - 1 : -1,
|
|
||||||
openTokenIndex: tokenIndex,
|
|
||||||
argIndex: 0,
|
|
||||||
});
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (token.text === "," && stack.length > 0) {
|
|
||||||
const frame = stack[stack.length - 1];
|
|
||||||
if (frame.callee !== null) frame.argIndex += 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
for (const frame of stack) {
|
|
||||||
errors.push({ kind: "unmatched-open-paren", offset: tokens[frame.openTokenIndex].start });
|
|
||||||
}
|
|
||||||
|
|
||||||
return { contexts, calls, errors };
|
|
||||||
}
|
|
||||||
|
|
||||||
function nearestCall(contexts, tokenIndex, callee) {
|
|
||||||
const entries = contexts[tokenIndex] || [];
|
|
||||||
for (let contextIndex = entries.length - 1; contextIndex >= 0; contextIndex -= 1) {
|
|
||||||
const entry = entries[contextIndex];
|
|
||||||
if (callee === undefined || entry.callee === callee) return entry;
|
|
||||||
}
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
module.exports = { buildCallContexts, lex, nearestCall };
|
|
||||||
-85
@@ -1,85 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "atomasm-psx",
|
|
||||||
"displayName": "AtomAsm-PSX",
|
|
||||||
"description": "Semantic highlighting for the PS1 Tape/Atom MIPS macro DSL",
|
|
||||||
"publisher": "local",
|
|
||||||
"version": "0.3.0",
|
|
||||||
"engines": { "vscode": "^1.80.0" },
|
|
||||||
"categories": ["Programming Languages"],
|
|
||||||
"activationEvents": ["onLanguage:c", "onLanguage:cpp"],
|
|
||||||
"main": "./extension.js",
|
|
||||||
"files": [
|
|
||||||
"classifier.js",
|
|
||||||
"extension.js",
|
|
||||||
"lexer.js",
|
|
||||||
"source-index.js",
|
|
||||||
"syntaxes/tape_atom.tmLanguage.json"
|
|
||||||
],
|
|
||||||
"scripts": {
|
|
||||||
"test": "node --test test/*.test.js",
|
|
||||||
"package": "npx --yes @vscode/vsce@3.6.1 package --allow-missing-repository --skip-license --out atomasm-psx-0.3.0.vsix"
|
|
||||||
},
|
|
||||||
"contributes": {
|
|
||||||
"semanticTokenTypes": [
|
|
||||||
{ "id": "tapeAtomKeyword", "superType": "keyword", "description": "Tape atom declaration keyword" },
|
|
||||||
{ "id": "tapeAtomName", "superType": "function", "description": "Tape atom name" },
|
|
||||||
{ "id": "tapeComponentKeyword", "superType": "keyword", "description": "Tape atom component declaration keyword" },
|
|
||||||
{ "id": "tapeComponentName", "superType": "function", "description": "Tape atom component name" },
|
|
||||||
{ "id": "tapeAnnotation", "superType": "macro", "description": "Tape atom annotation" },
|
|
||||||
{ "id": "tapeBindType", "superType": "type", "description": "Tape bind structure type" },
|
|
||||||
{ "id": "tapePhase", "superType": "label", "description": "Tape atom phase" },
|
|
||||||
{ "id": "tapeLabel", "superType": "label", "description": "Tape atom branch label" },
|
|
||||||
{ "id": "tapeCpuInstruction", "superType": "macro", "description": "MIPS CPU instruction emitter" },
|
|
||||||
{ "id": "tapeControlFlow", "superType": "keyword", "description": "MIPS branch or jump instruction" },
|
|
||||||
{ "id": "tapeGteInstruction", "superType": "macro", "description": "GTE instruction emitter" },
|
|
||||||
{ "id": "tapeGpuInstruction", "superType": "macro", "description": "GPU command emitter" },
|
|
||||||
{ "id": "tapeComponentInstruction", "superType": "macro", "description": "Tape atom component invocation" },
|
|
||||||
{ "id": "tapeDelaySlot", "superType": "keyword", "description": "Load or branch delay slot annotation" },
|
|
||||||
{ "id": "tapeGprRegister", "superType": "variable", "description": "MIPS GPR alias" },
|
|
||||||
{ "id": "tapeCop2Register", "superType": "variable", "description": "COP2 data or control register alias" },
|
|
||||||
{ "id": "tapeDuffleType", "superType": "type", "description": "Duffle type or type constructor" },
|
|
||||||
{ "id": "tapeAttribute", "superType": "keyword", "description": "Duffle linkage or storage attribute" },
|
|
||||||
{ "id": "keyword", "description": "Standard keyword (DSL built-in macros)" },
|
|
||||||
{ "id": "macro", "description": "Standard macro (utility #define with no instruction domain)" }
|
|
||||||
],
|
|
||||||
"semanticTokenModifiers": [
|
|
||||||
{ "id": "tapeRead", "description": "Register declared in atom_reads" },
|
|
||||||
{ "id": "tapeWrite", "description": "Register declared in atom_writes" },
|
|
||||||
{ "id": "tapeAuto", "description": "Auto-allocated register" }
|
|
||||||
],
|
|
||||||
"semanticTokenScopes": [
|
|
||||||
{
|
|
||||||
"language": "c",
|
|
||||||
"scopes": {
|
|
||||||
"tapeAtomKeyword": ["keyword.control.duffle.atom"],
|
|
||||||
"tapeAtomName": ["entity.name.function.duffle.atom"],
|
|
||||||
"tapeComponentKeyword": ["keyword.control.duffle.component"],
|
|
||||||
"tapeComponentName": ["entity.name.function.duffle.component"],
|
|
||||||
"tapeAnnotation": ["support.function.duffle.annotation"],
|
|
||||||
"tapeBindType": ["entity.name.type.duffle.bind"],
|
|
||||||
"tapePhase": ["entity.name.tag.duffle.phase"],
|
|
||||||
"tapeLabel": ["entity.name.label.duffle.atom"],
|
|
||||||
"tapeCpuInstruction": ["support.function.duffle.cpu"],
|
|
||||||
"tapeControlFlow": ["keyword.control.duffle.branch"],
|
|
||||||
"tapeGteInstruction": ["support.function.duffle.gte"],
|
|
||||||
"tapeGpuInstruction": ["support.function.duffle.gpu"],
|
|
||||||
"tapeComponentInstruction": ["support.function.duffle.component"],
|
|
||||||
"tapeDelaySlot": ["keyword.operator.duffle.delayslot"],
|
|
||||||
"tapeGprRegister": ["variable.other.constant.duffle.gpr"],
|
|
||||||
"tapeCop2Register": ["variable.other.constant.duffle.cop2"],
|
|
||||||
"tapeDuffleType": ["storage.type.duffle.type"],
|
|
||||||
"tapeAttribute": ["storage.modifier.duffle.attr"],
|
|
||||||
"keyword": ["keyword"],
|
|
||||||
"macro": ["entity.name.function.preprocessor"]
|
|
||||||
}
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"grammars": [
|
|
||||||
{
|
|
||||||
"scopeName": "tape_atom.injection",
|
|
||||||
"path": "./syntaxes/tape_atom.tmLanguage.json",
|
|
||||||
"injectTo": ["source.c", "source.cpp"]
|
|
||||||
}
|
|
||||||
]
|
|
||||||
}
|
|
||||||
}
|
|
||||||
-341
@@ -1,341 +0,0 @@
|
|||||||
"use strict";
|
|
||||||
|
|
||||||
const path = require("node:path");
|
|
||||||
const { buildCallContexts, lex, nearestCall } = require("./lexer");
|
|
||||||
|
|
||||||
const BASE_TYPES = [
|
|
||||||
"B1", "B2", "B4", "B8", "F4", "F8", "S1", "S2", "S4", "S8",
|
|
||||||
"U1", "U2", "U4", "U8", "MipsAtom", "MipsCode", "Reg",
|
|
||||||
];
|
|
||||||
|
|
||||||
const C_BUILTINS = new Set([
|
|
||||||
"void", "type", "char", "short", "int", "long", "float", "double",
|
|
||||||
"unsigned", "signed", "bool", "size_t", "uint8_t", "uint16_t", "uint32_t",
|
|
||||||
"int8_t", "int16_t", "int32_t",
|
|
||||||
]);
|
|
||||||
|
|
||||||
const BASE_ATTRIBUTES = [
|
|
||||||
"FI_", "I_", "NI_", "Relative_", "Struct_", "Enum_", "Union_", "Array_",
|
|
||||||
"Slice_", "TypeR_", "TypeV_", "align_", "internal", "local_persist", "global",
|
|
||||||
"RO_", "LP_", "gknown", "expect_", "cexpr_",
|
|
||||||
"asm", "asm_words", "asm_rpins", "asm_clobber",
|
|
||||||
"O_", "S_", "C_", "T_", "tmpl", "glue", "r_", "v_", "tr_", "tv_",
|
|
||||||
"rgcc", "r_use", "r_set", "r_mod", "r_imm", "r_mem",
|
|
||||||
"u1_", "u2_", "u4_", "u8_", "s1_", "s2_", "s4_", "s8_",
|
|
||||||
"u1_r", "u2_r", "u4_r", "u8_r", "u1_v", "u2_v", "u4_v", "u8_v",
|
|
||||||
];
|
|
||||||
|
|
||||||
function createIndex() {
|
|
||||||
return {
|
|
||||||
atoms: new Set(),
|
|
||||||
components: new Set(),
|
|
||||||
componentAliases: new Set(),
|
|
||||||
macros: new Map(),
|
|
||||||
registers: new Map(),
|
|
||||||
bindTypes: new Set(),
|
|
||||||
types: new Set(BASE_TYPES),
|
|
||||||
phases: new Set(),
|
|
||||||
labels: new Set(),
|
|
||||||
attributes: new Set(BASE_ATTRIBUTES),
|
|
||||||
componentCallees: new Map(),
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
function cloneIndex(source) {
|
|
||||||
const result = createIndex();
|
|
||||||
for (const key of ["atoms", "components", "componentAliases", "bindTypes", "types", "phases", "labels", "attributes"]) {
|
|
||||||
for (const value of source[key]) result[key].add(value);
|
|
||||||
}
|
|
||||||
for (const [name, domain] of source.macros) result.macros.set(name, domain);
|
|
||||||
for (const [name, domain] of source.registers) result.registers.set(name, domain);
|
|
||||||
for (const [name, callees] of source.componentCallees) result.componentCallees.set(name, callees.slice());
|
|
||||||
return result;
|
|
||||||
}
|
|
||||||
|
|
||||||
function mergeIndexes(...sources) {
|
|
||||||
const result = createIndex();
|
|
||||||
for (const source of sources) {
|
|
||||||
if (!source) continue;
|
|
||||||
for (const key of ["atoms", "components", "componentAliases", "bindTypes", "types", "phases", "labels", "attributes"]) {
|
|
||||||
for (const value of source[key]) result[key].add(value);
|
|
||||||
}
|
|
||||||
for (const [name, domain] of source.macros) {
|
|
||||||
const existing = result.macros.get(name);
|
|
||||||
if (!existing || domainRank(domain) >= domainRank(existing)) result.macros.set(name, domain);
|
|
||||||
}
|
|
||||||
for (const [name, domain] of source.registers) result.registers.set(name, domain);
|
|
||||||
for (const [name, callees] of source.componentCallees) {
|
|
||||||
const existing = result.componentCallees.get(name) || [];
|
|
||||||
result.componentCallees.set(name, existing.concat(callees));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return resolveComponentDomains(result);
|
|
||||||
}
|
|
||||||
|
|
||||||
function domainFromPath(filePath) {
|
|
||||||
const base = path.basename(filePath.replaceAll("\\", "/")).toLowerCase();
|
|
||||||
if (base === "mips.h") return "cpu";
|
|
||||||
if (base === "gte.h") return "gte";
|
|
||||||
if (base === "gp.h") return "gpu";
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
function prefixDomain(name) {
|
|
||||||
if (/^(?:branch_|jump_|call_)/.test(name)) return "control";
|
|
||||||
if (/^gte_(?!cr_)/.test(name) || name.startsWith("mac_gte_") || name.startsWith("ac_gte_")) return "gte";
|
|
||||||
if (/^gp[01]_/.test(name) || name.startsWith("mac_gp_") || name.startsWith("ac_gp_")) return "gpu";
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
function collectBraceIdentifiers(tokens, openBraceIndex) {
|
|
||||||
const names = [];
|
|
||||||
let depth = 0;
|
|
||||||
for (let tokenIndex = openBraceIndex; tokenIndex < tokens.length; tokenIndex += 1) {
|
|
||||||
if (tokens[tokenIndex].text === "{") depth += 1;
|
|
||||||
if (tokens[tokenIndex].text === "}") {
|
|
||||||
depth -= 1;
|
|
||||||
if (depth === 0) break;
|
|
||||||
}
|
|
||||||
if (tokens[tokenIndex].kind === "identifier") names.push(tokens[tokenIndex].text);
|
|
||||||
}
|
|
||||||
return names;
|
|
||||||
}
|
|
||||||
|
|
||||||
function resolveComponentDomains(index) {
|
|
||||||
const hardwareRank = { cpu: 1, gpu: 2, gte: 3, control: 4 };
|
|
||||||
let changed = true;
|
|
||||||
while (changed) {
|
|
||||||
changed = false;
|
|
||||||
for (const [alias, callees] of index.componentCallees) {
|
|
||||||
let best = index.macros.get(alias) || "component";
|
|
||||||
let bestRank = hardwareRank[best] || 0;
|
|
||||||
for (const callee of callees) {
|
|
||||||
const domain = prefixDomain(callee) || index.macros.get(callee);
|
|
||||||
const rank = hardwareRank[domain] || 0;
|
|
||||||
if (rank > bestRank) {
|
|
||||||
best = domain;
|
|
||||||
bestRank = rank;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (bestRank > 0 && index.macros.get(alias) !== best) {
|
|
||||||
index.macros.set(alias, best);
|
|
||||||
changed = true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return index;
|
|
||||||
}
|
|
||||||
|
|
||||||
function domainRank(domain) {
|
|
||||||
if (domain === "control") return 4;
|
|
||||||
if (domain === "cpu" || domain === "gte" || domain === "gpu") return 3;
|
|
||||||
if (domain === "component") return 2;
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
function registerKind(name) {
|
|
||||||
if (/^R_[A-Za-z0-9_]+$/.test(name)) return "gpr";
|
|
||||||
if (/^(?:C2_|gte_cr_)[A-Za-z0-9_]+$/.test(name)) return "cop2";
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
function componentAlias(name) {
|
|
||||||
return name.startsWith("ac_") ? `mac_${name.slice(3)}` : null;
|
|
||||||
}
|
|
||||||
|
|
||||||
function findFunctionNameBefore(tokens, calleeTokenIndex) {
|
|
||||||
let closeIndex = calleeTokenIndex - 1;
|
|
||||||
while (closeIndex >= 0 && tokens[closeIndex].kind === "identifier" && tokens[closeIndex].text === "atom_dbg_skip") {
|
|
||||||
closeIndex -= 1;
|
|
||||||
}
|
|
||||||
if (!tokens[closeIndex] || tokens[closeIndex].text !== ")") return null;
|
|
||||||
|
|
||||||
let depth = 1;
|
|
||||||
for (let tokenIndex = closeIndex - 1; tokenIndex >= 0; tokenIndex -= 1) {
|
|
||||||
if (tokens[tokenIndex].text === ")") depth += 1;
|
|
||||||
if (tokens[tokenIndex].text === "(") depth -= 1;
|
|
||||||
if (depth !== 0) continue;
|
|
||||||
const name = tokens[tokenIndex - 1];
|
|
||||||
return name && name.kind === "identifier" ? name : null;
|
|
||||||
}
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
function scanSource(source, filePath) {
|
|
||||||
const lexical = lex(source);
|
|
||||||
const balanced = buildCallContexts(lexical.tokens);
|
|
||||||
const tokens = lexical.tokens;
|
|
||||||
const contexts = balanced.contexts;
|
|
||||||
const index = createIndex();
|
|
||||||
const declarations = new Map();
|
|
||||||
const domain = domainFromPath(filePath);
|
|
||||||
|
|
||||||
function mark(token, role, modifiers = ["declaration"]) {
|
|
||||||
declarations.set(token.start, { role, modifiers });
|
|
||||||
}
|
|
||||||
|
|
||||||
function addComponent(token) {
|
|
||||||
index.components.add(token.text);
|
|
||||||
mark(token, "componentName");
|
|
||||||
const alias = componentAlias(token.text);
|
|
||||||
if (alias) {
|
|
||||||
index.componentAliases.add(alias);
|
|
||||||
index.macros.set(alias, prefixDomain(alias) || prefixDomain(token.text) || "component");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
function bindComponentCallees(alias, callees) {
|
|
||||||
if (!alias) return;
|
|
||||||
index.componentAliases.add(alias);
|
|
||||||
index.componentCallees.set(alias, callees);
|
|
||||||
if (!index.macros.has(alias)) index.macros.set(alias, "component");
|
|
||||||
}
|
|
||||||
|
|
||||||
for (let tokenIndex = 0; tokenIndex < tokens.length; tokenIndex += 1) {
|
|
||||||
const token = tokens[tokenIndex];
|
|
||||||
if (token.kind !== "identifier") continue;
|
|
||||||
|
|
||||||
const kind = registerKind(token.text);
|
|
||||||
if (kind) {
|
|
||||||
index.registers.set(token.text, kind);
|
|
||||||
if (tokens[tokenIndex + 1] && tokens[tokenIndex + 1].text === "=") {
|
|
||||||
mark(token, kind === "gpr" ? "gprRegister" : "cop2Register");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
const context = nearestCall(contexts, tokenIndex);
|
|
||||||
if (context && context.argIndex === 0) {
|
|
||||||
if (context.callee === "MipsAtom_") {
|
|
||||||
index.atoms.add(token.text);
|
|
||||||
mark(token, "atomName");
|
|
||||||
}
|
|
||||||
if (context.callee === "MipsAtomComp_") addComponent(token);
|
|
||||||
if (context.callee === "atom_bind") index.bindTypes.add(token.text);
|
|
||||||
if (context.callee === "atom_phase" || context.callee === "phase_auto_reg") index.phases.add(token.text);
|
|
||||||
if (context.callee === "atom_label" || context.callee === "atom_offset") index.labels.add(token.text);
|
|
||||||
}
|
|
||||||
|
|
||||||
const isWrappedType = context && (
|
|
||||||
((context.callee === "Struct_" || context.callee === "Union_") && context.argIndex === 0) ||
|
|
||||||
(context.callee === "Enum_" && context.argIndex === 1)
|
|
||||||
);
|
|
||||||
if (isWrappedType) {
|
|
||||||
index.types.add(token.text);
|
|
||||||
mark(token, token.text.startsWith("Binds_") ? "bindType" : "duffleType");
|
|
||||||
if (token.text.startsWith("Binds_")) index.bindTypes.add(token.text);
|
|
||||||
}
|
|
||||||
|
|
||||||
if (context && context.callee === "atom_offset" && context.argIndex === 1) index.labels.add(token.text);
|
|
||||||
|
|
||||||
if (context && context.callee === "atom_auto_reg") {
|
|
||||||
if (context.argIndex === 0) index.atoms.add(token.text);
|
|
||||||
if (context.argIndex === 1) {
|
|
||||||
index.registers.set(token.text, "gpr");
|
|
||||||
mark(token, "gprRegister", ["declaration", "tapeAuto"]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if (context && context.callee === "phase_auto_reg" && context.argIndex === 1) {
|
|
||||||
index.registers.set(token.text, "gpr");
|
|
||||||
mark(token, "gprRegister", ["declaration", "tapeAuto"]);
|
|
||||||
}
|
|
||||||
|
|
||||||
if (token.text === "define" && tokens[tokenIndex - 1] && tokens[tokenIndex - 1].text === "#") {
|
|
||||||
const name = tokens[tokenIndex + 1];
|
|
||||||
if (name && name.kind === "identifier" && name.line === token.line) {
|
|
||||||
if (/^(?:RegUse_|Struct_|Enum_|Union_|TypeR_|TypeV_|Relative_|Binds_)/.test(name.text)) {
|
|
||||||
index.types.add(name.text);
|
|
||||||
} else if (/^(?:ac_|mac_)/.test(name.text)) {
|
|
||||||
const alias = name.text.startsWith("ac_") ? componentAlias(name.text) : name.text;
|
|
||||||
const rest = [];
|
|
||||||
for (let restIndex = tokenIndex + 2; restIndex < tokens.length && tokens[restIndex].line === name.line; restIndex += 1) {
|
|
||||||
if (tokens[restIndex].kind === "identifier") rest.push(tokens[restIndex].text);
|
|
||||||
}
|
|
||||||
if (alias) {
|
|
||||||
index.componentAliases.add(alias);
|
|
||||||
index.macros.set(alias, prefixDomain(alias) || "component");
|
|
||||||
if (rest.length) index.componentCallees.set(alias, rest);
|
|
||||||
}
|
|
||||||
} else {
|
|
||||||
index.macros.set(name.text, domain || "utility");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if (token.text === "typedef") {
|
|
||||||
let endIndex = tokenIndex + 1;
|
|
||||||
let hasBrace = false;
|
|
||||||
let lastIdentifier = null;
|
|
||||||
while (endIndex < tokens.length && tokens[endIndex].text !== ";") {
|
|
||||||
if (tokens[endIndex].text === "{") hasBrace = true;
|
|
||||||
if (tokens[endIndex].kind === "identifier" && !C_BUILTINS.has(tokens[endIndex].text)) lastIdentifier = tokens[endIndex];
|
|
||||||
endIndex += 1;
|
|
||||||
}
|
|
||||||
if (!hasBrace && lastIdentifier && !C_BUILTINS.has(lastIdentifier.text)) {
|
|
||||||
index.types.add(lastIdentifier.text);
|
|
||||||
mark(lastIdentifier, "duffleType");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if (token.text === "MipsAtom_Proc_") {
|
|
||||||
const functionName = findFunctionNameBefore(tokens, tokenIndex);
|
|
||||||
if (functionName) {
|
|
||||||
const atomName = functionName.text.endsWith("_proc")
|
|
||||||
? functionName.text.slice(0, -5)
|
|
||||||
: functionName.text;
|
|
||||||
index.atoms.add(atomName);
|
|
||||||
index.atoms.add(functionName.text);
|
|
||||||
mark(functionName, "atomName");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if (token.text === "MipsAtomComp_Proc_") {
|
|
||||||
const functionName = findFunctionNameBefore(tokens, tokenIndex);
|
|
||||||
if (functionName) addComponent(functionName);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
for (const call of balanced.calls) {
|
|
||||||
if (call.callee === "MipsAtomComp_") {
|
|
||||||
const name = tokens[call.openTokenIndex + 1];
|
|
||||||
const brace = tokens[call.closeTokenIndex + 1];
|
|
||||||
if (name && name.kind === "identifier" && brace && brace.text === "{") {
|
|
||||||
bindComponentCallees(componentAlias(name.text), collectBraceIdentifiers(tokens, call.closeTokenIndex + 1));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (call.callee === "MipsAtomComp_Proc_") {
|
|
||||||
const functionName = findFunctionNameBefore(tokens, call.calleeTokenIndex);
|
|
||||||
let braceIndex = -1;
|
|
||||||
for (let tokenIndex = call.openTokenIndex + 1; tokenIndex < call.closeTokenIndex; tokenIndex += 1) {
|
|
||||||
if (tokens[tokenIndex].text === "{") {
|
|
||||||
braceIndex = tokenIndex;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (functionName && braceIndex >= 0) {
|
|
||||||
bindComponentCallees(componentAlias(functionName.text), collectBraceIdentifiers(tokens, braceIndex));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (!domain) continue;
|
|
||||||
const name = tokens[call.calleeTokenIndex];
|
|
||||||
const after = tokens[call.closeTokenIndex + 1];
|
|
||||||
if (!name || !after || after.text !== "{") continue;
|
|
||||||
if (/^(?:gp0_|gp1_|gte_|mac_)/.test(name.text)) index.macros.set(name.text, domain);
|
|
||||||
}
|
|
||||||
|
|
||||||
return {
|
|
||||||
index: resolveComponentDomains(cloneIndex(index)),
|
|
||||||
declarations,
|
|
||||||
tokens,
|
|
||||||
contexts,
|
|
||||||
errors: [...lexical.errors, ...balanced.errors],
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
module.exports = {
|
|
||||||
createIndex,
|
|
||||||
domainFromPath,
|
|
||||||
mergeIndexes,
|
|
||||||
resolveComponentDomains,
|
|
||||||
scanSource,
|
|
||||||
};
|
|
||||||
@@ -1,71 +0,0 @@
|
|||||||
{
|
|
||||||
"scopeName": "tape_atom.injection",
|
|
||||||
"injectionSelector": "L:source.c -comment -string, L:source.cpp -comment -string",
|
|
||||||
"patterns": [
|
|
||||||
{ "include": "#atom-declarations" },
|
|
||||||
{ "include": "#component-declarations" },
|
|
||||||
{ "include": "#annotation-arguments" },
|
|
||||||
{ "include": "#annotations" },
|
|
||||||
{ "include": "#delay-slots" },
|
|
||||||
{ "include": "#types" },
|
|
||||||
{ "include": "#attributes" }
|
|
||||||
],
|
|
||||||
"repository": {
|
|
||||||
"atom-declarations": {
|
|
||||||
"patterns": [
|
|
||||||
{
|
|
||||||
"match": "\\b(MipsAtom_)\\s*\\(\\s*([A-Za-z_][A-Za-z0-9_]*)",
|
|
||||||
"captures": {
|
|
||||||
"1": { "name": "keyword.control.duffle.atom" },
|
|
||||||
"2": { "name": "entity.name.function.duffle.atom" }
|
|
||||||
}
|
|
||||||
},
|
|
||||||
{ "match": "\\bMipsAtom_Proc_\\b", "name": "keyword.control.duffle.atom" },
|
|
||||||
{ "match": "\\b[A-Za-z_][A-Za-z0-9_]*_proc\\b", "name": "entity.name.function.duffle.atom" }
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"component-declarations": {
|
|
||||||
"patterns": [
|
|
||||||
{
|
|
||||||
"match": "\\b(MipsAtomComp_)\\s*\\(\\s*(ac_[A-Za-z0-9_]*)",
|
|
||||||
"captures": {
|
|
||||||
"1": { "name": "keyword" },
|
|
||||||
"2": { "name": "entity.name.function.duffle.component" }
|
|
||||||
}
|
|
||||||
},
|
|
||||||
{ "match": "\\bMipsAtomComp_Proc_\\b", "name": "keyword" }
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"annotation-arguments": {
|
|
||||||
"patterns": [
|
|
||||||
{
|
|
||||||
"match": "\\b(atom_offset)\\s*\\(\\s*([A-Za-z_][A-Za-z0-9_]*)\\s*,\\s*([A-Za-z_][A-Za-z0-9_]*)",
|
|
||||||
"captures": {
|
|
||||||
"1": { "name": "support.function.duffle.annotation" },
|
|
||||||
"2": { "name": "entity.name.label.duffle.atom" },
|
|
||||||
"3": { "name": "entity.name.label.duffle.atom" }
|
|
||||||
}
|
|
||||||
},
|
|
||||||
{ "match": "(?<=\\batom_bind\\()\\s*Binds_[A-Za-z0-9_]+", "name": "entity.name.type.duffle.bind" },
|
|
||||||
{ "match": "(?<=\\batom_phase\\()\\s*[A-Za-z_][A-Za-z0-9_]*", "name": "entity.name.tag.duffle.phase" },
|
|
||||||
{ "match": "(?<=\\batom_label\\()\\s*[A-Za-z_][A-Za-z0-9_]*", "name": "entity.name.label.duffle.atom" }
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"annotations": {
|
|
||||||
"match": "\\b(atom_info|atom_bind|atom_reads|atom_writes|atom_label|atom_offset|atom_reg|atom_type|atom_ctx|atom_phase|atom_auto_reg|phase_auto_reg|atom_dbg_skip)\\b",
|
|
||||||
"name": "support.function.duffle.annotation"
|
|
||||||
},
|
|
||||||
"delay-slots": {
|
|
||||||
"match": "\\b(LdSlot_|BdSlot_|DmaSlot_|GteDelay_)\\b",
|
|
||||||
"name": "keyword.operator.duffle.delayslot"
|
|
||||||
},
|
|
||||||
"types": {
|
|
||||||
"match": "\\b(?:Binds_[A-Za-z0-9_]+|RegUse_[A-Za-z0-9_]+)\\b",
|
|
||||||
"name": "storage.type.duffle.type"
|
|
||||||
},
|
|
||||||
"attributes": {
|
|
||||||
"match": "\\b(?:FI_|I_|NI_|Relative_|Struct_|Enum_|Union_|Array_|Slice_|TypeR_|TypeV_|align_|internal|local_persist|global|RO_|LP_|gknown|expect_|cexpr_|asm|asm_words|asm_rpins|asm_clobber|O_|S_|C_|T_|tmpl|glue|r_|v_|tr_|tv_|rgcc|r_use|r_set|r_mod|r_imm|r_mem|u[1248]_|u[1248]_r|u[1248]_v|s[1248]_)\\b",
|
|
||||||
"name": "keyword"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
Binary file not shown.
-128
@@ -1,128 +0,0 @@
|
|||||||
"use strict";
|
|
||||||
|
|
||||||
const assert = require("node:assert/strict");
|
|
||||||
const test = require("node:test");
|
|
||||||
|
|
||||||
const { classifyDocument } = require("../classifier");
|
|
||||||
const { createIndex } = require("../source-index");
|
|
||||||
|
|
||||||
function byText(result, text) {
|
|
||||||
return result.spans.filter((span) => span.text === text);
|
|
||||||
}
|
|
||||||
|
|
||||||
test("classifyDocument distinguishes declaration, annotation, phase, bind, and label roles", () => {
|
|
||||||
const source = [
|
|
||||||
"typedef Struct_(Binds_CubeTri) { U4 PrimCursor; };",
|
|
||||||
"MipsAtom_(cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4),",
|
|
||||||
"\tatom_reads(R_PrimCursor), atom_writes(R_FaceCursor)) {",
|
|
||||||
"\tbranch_le_zero(R_T0, atom_offset(cull, exit)),",
|
|
||||||
"\tatom_label(exit)",
|
|
||||||
"};",
|
|
||||||
].join("\n");
|
|
||||||
|
|
||||||
const result = classifyDocument(source, "C:/x/code/hello_camera/hello_camera.atom.c", createIndex());
|
|
||||||
|
|
||||||
assert.equal(byText(result, "MipsAtom_")[0].type, "tapeAtomKeyword");
|
|
||||||
assert.deepEqual(byText(result, "cube_g4_face")[0].modifiers, ["declaration"]);
|
|
||||||
assert.equal(byText(result, "atom_bind")[0].type, "tapeAnnotation");
|
|
||||||
assert.equal(byText(result, "Binds_CubeTri").at(-1).type, "tapeBindType");
|
|
||||||
assert.equal(byText(result, "cube_g4")[0].type, "tapePhase");
|
|
||||||
assert.deepEqual(byText(result, "cube_g4")[0].modifiers, ["declaration"]);
|
|
||||||
assert.equal(byText(result, "cull")[0].type, "tapeLabel");
|
|
||||||
assert.equal(byText(result, "exit").every((span) => span.type === "tapeLabel"), true);
|
|
||||||
});
|
|
||||||
|
|
||||||
test("classifyDocument applies read and write modifiers to GPRs", () => {
|
|
||||||
const source = "atom_info(atom_reads(R_PrimCursor), atom_writes(R_FaceCursor))";
|
|
||||||
const result = classifyDocument(source, "C:/x/code/test.atom.c", createIndex());
|
|
||||||
|
|
||||||
assert.deepEqual(byText(result, "R_PrimCursor")[0].modifiers, ["tapeRead"]);
|
|
||||||
assert.deepEqual(byText(result, "R_FaceCursor")[0].modifiers, ["tapeWrite"]);
|
|
||||||
});
|
|
||||||
|
|
||||||
test("classifyDocument separates CPU, GTE, GPU, and component domains", () => {
|
|
||||||
const workspace = createIndex();
|
|
||||||
workspace.macros.set("load_word", "cpu");
|
|
||||||
workspace.macros.set("gte_cmdw_rtpt", "gte");
|
|
||||||
workspace.macros.set("gp1_word_DisplayOn", "gpu");
|
|
||||||
workspace.macros.set("mac_yield", "control");
|
|
||||||
workspace.componentAliases.add("mac_yield");
|
|
||||||
|
|
||||||
const source = "load_word(R_T0, R_T1, 0), gte_cmdw_rtpt, gp1_word_DisplayOn(), mac_yield(), C2_MAC0, gte_cr_OFX_Code";
|
|
||||||
const result = classifyDocument(source, "C:/x/code/test.c", workspace);
|
|
||||||
|
|
||||||
assert.equal(byText(result, "load_word")[0].type, "tapeCpuInstruction");
|
|
||||||
assert.equal(byText(result, "gte_cmdw_rtpt")[0].type, "tapeGteInstruction");
|
|
||||||
assert.equal(byText(result, "gp1_word_DisplayOn")[0].type, "tapeGpuInstruction");
|
|
||||||
assert.equal(byText(result, "mac_yield")[0].type, "tapeControlFlow");
|
|
||||||
assert.equal(byText(result, "C2_MAC0")[0].type, "tapeCop2Register");
|
|
||||||
assert.equal(byText(result, "gte_cr_OFX_Code")[0].type, "tapeCop2Register");
|
|
||||||
});
|
|
||||||
|
|
||||||
test("component invocations keep the domain resolved from their emitted instructions", () => {
|
|
||||||
const workspace = createIndex();
|
|
||||||
workspace.macros.set("mac_load_word_imm", "cpu");
|
|
||||||
workspace.macros.set("mac_gcmd_push", "gpu");
|
|
||||||
workspace.macros.set("mac_gte_store_f3", "gte");
|
|
||||||
workspace.macros.set("mac_load_v3s4", "cpu");
|
|
||||||
|
|
||||||
const source = "mac_load_word_imm(dst, imm), mac_gcmd_push(cmd), mac_gte_store_f3(cursor), mac_load_v3s4()";
|
|
||||||
const result = classifyDocument(source, "C:/x/code/hello_camera/hello_camera.atom.c", workspace);
|
|
||||||
|
|
||||||
assert.equal(byText(result, "mac_load_word_imm")[0].type, "tapeCpuInstruction");
|
|
||||||
assert.equal(byText(result, "mac_gcmd_push")[0].type, "tapeGpuInstruction");
|
|
||||||
assert.equal(byText(result, "mac_gte_store_f3")[0].type, "tapeGteInstruction");
|
|
||||||
assert.equal(byText(result, "mac_load_v3s4")[0].type, "tapeCpuInstruction");
|
|
||||||
});
|
|
||||||
|
|
||||||
test("utility macros without a hardware domain use the standard macro token", () => {
|
|
||||||
const workspace = createIndex();
|
|
||||||
workspace.macros.set("load_word", "cpu");
|
|
||||||
workspace.macros.set("assert", "utility");
|
|
||||||
workspace.macros.set("stringify", "utility");
|
|
||||||
workspace.macros.set("u4_hi", "utility");
|
|
||||||
|
|
||||||
const source = "load_word(R_T0, R_T1, 0), assert(ok), stringify(name), u4_hi(imm)";
|
|
||||||
const result = classifyDocument(source, "C:/x/code/hello_camera/hello_camera.c", workspace);
|
|
||||||
|
|
||||||
assert.equal(byText(result, "load_word")[0].type, "tapeCpuInstruction");
|
|
||||||
assert.equal(byText(result, "assert")[0].type, "macro");
|
|
||||||
assert.equal(byText(result, "stringify")[0].type, "macro");
|
|
||||||
assert.equal(byText(result, "u4_hi")[0].type, "macro");
|
|
||||||
});
|
|
||||||
|
|
||||||
test("document-local declarations override an empty workspace index", () => {
|
|
||||||
const source = [
|
|
||||||
"MipsAtomComp_(ac_new_component) { nop };",
|
|
||||||
"MipsAtomComp_Proc_(ab, { nop })",
|
|
||||||
"mac_new_component(),",
|
|
||||||
].join("\n");
|
|
||||||
const result = classifyDocument(source, "C:/x/code/duffle/math.atom.c", createIndex());
|
|
||||||
|
|
||||||
assert.equal(byText(result, "MipsAtomComp_")[0].type, "keyword");
|
|
||||||
assert.equal(byText(result, "MipsAtomComp_Proc_")[0].type, "keyword");
|
|
||||||
assert.equal(byText(result, "ac_new_component")[0].type, "tapeComponentName");
|
|
||||||
assert.equal(byText(result, "mac_new_component")[0].type, "tapeComponentInstruction");
|
|
||||||
});
|
|
||||||
|
|
||||||
test("delay slot markers share the tapeDelaySlot token", () => {
|
|
||||||
const source = "LdSlot_ nop, BdSlot_ nop, DmaSlot_ nop2, GteDelay_ nop";
|
|
||||||
const result = classifyDocument(source, "C:/x/code/duffle/gte.atom.c", createIndex());
|
|
||||||
|
|
||||||
assert.equal(byText(result, "LdSlot_")[0].type, "tapeDelaySlot");
|
|
||||||
assert.equal(byText(result, "BdSlot_")[0].type, "tapeDelaySlot");
|
|
||||||
assert.equal(byText(result, "DmaSlot_")[0].type, "tapeDelaySlot");
|
|
||||||
assert.equal(byText(result, "GteDelay_")[0].type, "tapeDelaySlot");
|
|
||||||
});
|
|
||||||
|
|
||||||
test("classifier returns ordered non-overlapping spans and partial malformed output", () => {
|
|
||||||
const source = "atom_reads(R_A /* broken";
|
|
||||||
const result = classifyDocument(source, "C:/x/code/test.atom.c", createIndex());
|
|
||||||
|
|
||||||
assert.equal(result.errors.some((error) => error.kind === "unterminated-block-comment"), true);
|
|
||||||
for (let spanIndex = 1; spanIndex < result.spans.length; spanIndex += 1) {
|
|
||||||
const previous = result.spans[spanIndex - 1];
|
|
||||||
const current = result.spans[spanIndex];
|
|
||||||
assert.equal(previous.start + previous.length <= current.start, true);
|
|
||||||
}
|
|
||||||
});
|
|
||||||
-88
@@ -1,88 +0,0 @@
|
|||||||
"use strict";
|
|
||||||
|
|
||||||
const assert = require("node:assert/strict");
|
|
||||||
const fs = require("node:fs");
|
|
||||||
const path = require("node:path");
|
|
||||||
const test = require("node:test");
|
|
||||||
|
|
||||||
const { TOKEN_MODIFIERS, TOKEN_TYPES } = require("../classifier");
|
|
||||||
|
|
||||||
const ROOT = path.resolve(__dirname, "..");
|
|
||||||
|
|
||||||
function readJson(filePath) {
|
|
||||||
const raw = fs.readFileSync(filePath, "utf8");
|
|
||||||
const stripped = raw.replace(/\/\/.*$/gm, "").replace(/,\s*([}\]])/g, "$1");
|
|
||||||
return JSON.parse(stripped);
|
|
||||||
}
|
|
||||||
|
|
||||||
function collectScopeNames(value, output = new Set()) {
|
|
||||||
if (Array.isArray(value)) {
|
|
||||||
for (const entry of value) collectScopeNames(entry, output);
|
|
||||||
return output;
|
|
||||||
}
|
|
||||||
if (!value || typeof value !== "object") return output;
|
|
||||||
if (typeof value.name === "string") output.add(value.name);
|
|
||||||
for (const child of Object.values(value)) collectScopeNames(child, output);
|
|
||||||
return output;
|
|
||||||
}
|
|
||||||
|
|
||||||
test("package semantic legend matches classifier exports", () => {
|
|
||||||
const packageJson = readJson(path.join(ROOT, "package.json"));
|
|
||||||
const contributedTypes = packageJson.contributes.semanticTokenTypes.map((entry) => entry.id);
|
|
||||||
const contributedModifiers = packageJson.contributes.semanticTokenModifiers.map((entry) => entry.id);
|
|
||||||
|
|
||||||
assert.equal(packageJson.version, "0.3.0");
|
|
||||||
assert.deepEqual(contributedTypes, TOKEN_TYPES);
|
|
||||||
assert.deepEqual(contributedModifiers, TOKEN_MODIFIERS.filter((name) => name !== "declaration"));
|
|
||||||
});
|
|
||||||
|
|
||||||
test("package includes runtime files only and acknowledges local-only metadata", () => {
|
|
||||||
const packageJson = readJson(path.join(ROOT, "package.json"));
|
|
||||||
|
|
||||||
assert.deepEqual(packageJson.files, [
|
|
||||||
"classifier.js",
|
|
||||||
"extension.js",
|
|
||||||
"lexer.js",
|
|
||||||
"source-index.js",
|
|
||||||
"syntaxes/tape_atom.tmLanguage.json",
|
|
||||||
]);
|
|
||||||
assert.equal(packageJson.scripts.package.includes("--allow-missing-repository"), true);
|
|
||||||
assert.equal(packageJson.scripts.package.includes("--skip-license"), true);
|
|
||||||
});
|
|
||||||
|
|
||||||
test("every semantic token has a scope mapping; DSL-specific tokens also have grammar scopes", () => {
|
|
||||||
const packageJson = readJson(path.join(ROOT, "package.json"));
|
|
||||||
const grammar = readJson(path.join(ROOT, "syntaxes", "tape_atom.tmLanguage.json"));
|
|
||||||
const mappings = packageJson.contributes.semanticTokenScopes[0].scopes;
|
|
||||||
const grammarScopes = collectScopeNames(grammar);
|
|
||||||
|
|
||||||
const grammarRequired = new Set([
|
|
||||||
"tapeAtomKeyword", "tapeAtomName", "tapeComponentName",
|
|
||||||
"tapeAnnotation", "tapeBindType", "tapePhase", "tapeLabel",
|
|
||||||
"tapeDelaySlot", "tapeDuffleType", "keyword",
|
|
||||||
]);
|
|
||||||
|
|
||||||
for (const tokenType of TOKEN_TYPES) {
|
|
||||||
assert.equal(Array.isArray(mappings[tokenType]), true, `missing scope mapping: ${tokenType}`);
|
|
||||||
if (grammarRequired.has(tokenType)) {
|
|
||||||
assert.equal(mappings[tokenType].some((scope) => grammarScopes.has(scope)), true, `grammar does not emit: ${tokenType}`);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
test("TextMate offset labels stay scoped to atom_offset calls", () => {
|
|
||||||
const grammar = readJson(path.join(ROOT, "syntaxes", "tape_atom.tmLanguage.json"));
|
|
||||||
const serialized = JSON.stringify(grammar);
|
|
||||||
const offsetRule = grammar.repository["annotation-arguments"].patterns
|
|
||||||
.find((rule) => rule.match.includes("atom_offset"));
|
|
||||||
|
|
||||||
assert.equal(serialized.includes("(?<=,)"), false);
|
|
||||||
assert.equal(offsetRule.captures[1].name, "support.function.duffle.annotation");
|
|
||||||
assert.equal(offsetRule.captures[2].name, "entity.name.label.duffle.atom");
|
|
||||||
assert.equal(offsetRule.captures[3].name, "entity.name.label.duffle.atom");
|
|
||||||
});
|
|
||||||
|
|
||||||
test("workspace enables semantic highlighting", () => {
|
|
||||||
const settings = readJson(path.resolve(ROOT, "..", "settings.json"));
|
|
||||||
assert.equal(settings["editor.semanticHighlighting.enabled"], true);
|
|
||||||
});
|
|
||||||
-77
@@ -1,77 +0,0 @@
|
|||||||
"use strict";
|
|
||||||
|
|
||||||
const assert = require("node:assert/strict");
|
|
||||||
const test = require("node:test");
|
|
||||||
|
|
||||||
const { buildCallContexts, lex, nearestCall } = require("../lexer");
|
|
||||||
|
|
||||||
test("lex skips comments, strings, and character literals", () => {
|
|
||||||
const source = [
|
|
||||||
"MipsAtom_(visible)",
|
|
||||||
"// MipsAtom_(line_comment)",
|
|
||||||
"const char *s = \"atom_reads(R_Hidden)\";",
|
|
||||||
"char c = '\\''; /* gte_cmdw_hidden */",
|
|
||||||
"atom_reads(R_Visible)",
|
|
||||||
].join("\n");
|
|
||||||
|
|
||||||
const result = lex(source);
|
|
||||||
const identifiers = result.tokens
|
|
||||||
.filter((token) => token.kind === "identifier")
|
|
||||||
.map((token) => token.text);
|
|
||||||
|
|
||||||
assert.deepEqual(result.errors, []);
|
|
||||||
assert.equal(identifiers.includes("visible"), true);
|
|
||||||
assert.equal(identifiers.includes("R_Visible"), true);
|
|
||||||
assert.equal(identifiers.includes("line_comment"), false);
|
|
||||||
assert.equal(identifiers.includes("R_Hidden"), false);
|
|
||||||
assert.equal(identifiers.includes("gte_cmdw_hidden"), false);
|
|
||||||
});
|
|
||||||
|
|
||||||
test("lex reports unterminated block comments without returning comment tokens", () => {
|
|
||||||
const result = lex("R_Visible /* atom_reads(R_Hidden)");
|
|
||||||
|
|
||||||
assert.equal(result.tokens.some((token) => token.text === "R_Visible"), true);
|
|
||||||
assert.equal(result.tokens.some((token) => token.text === "R_Hidden"), false);
|
|
||||||
assert.deepEqual(result.errors.map((error) => error.kind), ["unterminated-block-comment"]);
|
|
||||||
});
|
|
||||||
|
|
||||||
test("line comments stop at CRLF boundaries", () => {
|
|
||||||
const result = lex("// atom_reads(R_Hidden)\r\natom_reads(R_Visible)\r\n");
|
|
||||||
const identifiers = result.tokens
|
|
||||||
.filter((token) => token.kind === "identifier")
|
|
||||||
.map((token) => token.text);
|
|
||||||
|
|
||||||
assert.equal(identifiers.includes("R_Hidden"), false);
|
|
||||||
assert.equal(identifiers.includes("R_Visible"), true);
|
|
||||||
});
|
|
||||||
|
|
||||||
test("balanced contexts retain multiline nesting and argument indexes", () => {
|
|
||||||
const source = [
|
|
||||||
"atom_info(",
|
|
||||||
"\tatom_phase(cube_g4),",
|
|
||||||
"\tatom_reads(R_A, nested(R_B, R_C)),",
|
|
||||||
"\tatom_writes(R_D)",
|
|
||||||
")",
|
|
||||||
].join("\n");
|
|
||||||
const lexical = lex(source);
|
|
||||||
const balanced = buildCallContexts(lexical.tokens);
|
|
||||||
|
|
||||||
const byText = new Map();
|
|
||||||
lexical.tokens.forEach((token, index) => {
|
|
||||||
if (token.kind === "identifier") byText.set(token.text, index);
|
|
||||||
});
|
|
||||||
|
|
||||||
assert.equal(nearestCall(balanced.contexts, byText.get("cube_g4")).callee, "atom_phase");
|
|
||||||
assert.equal(nearestCall(balanced.contexts, byText.get("R_A")).callee, "atom_reads");
|
|
||||||
assert.equal(nearestCall(balanced.contexts, byText.get("R_A")).argIndex, 0);
|
|
||||||
assert.equal(nearestCall(balanced.contexts, byText.get("R_C")).callee, "nested");
|
|
||||||
assert.equal(nearestCall(balanced.contexts, byText.get("R_D")).callee, "atom_writes");
|
|
||||||
assert.deepEqual(balanced.errors, []);
|
|
||||||
});
|
|
||||||
|
|
||||||
test("balanced contexts report unmatched parentheses", () => {
|
|
||||||
const lexical = lex("atom_reads(R_A");
|
|
||||||
const balanced = buildCallContexts(lexical.tokens);
|
|
||||||
|
|
||||||
assert.deepEqual(balanced.errors.map((error) => error.kind), ["unmatched-open-paren"]);
|
|
||||||
});
|
|
||||||
-134
@@ -1,134 +0,0 @@
|
|||||||
"use strict";
|
|
||||||
|
|
||||||
const assert = require("node:assert/strict");
|
|
||||||
const test = require("node:test");
|
|
||||||
|
|
||||||
const {
|
|
||||||
createIndex,
|
|
||||||
domainFromPath,
|
|
||||||
mergeIndexes,
|
|
||||||
scanSource,
|
|
||||||
} = require("../source-index");
|
|
||||||
|
|
||||||
test("scanSource discovers current atom and component forms", () => {
|
|
||||||
const source = [
|
|
||||||
"MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4), atom_reads(R_PrimCursor)) { mac_yield() };",
|
|
||||||
"MipsAtomComp_(ac_load_pair) { load_word(R_T0, R_T1, 0) };",
|
|
||||||
"internal MipsAtom* normalize_proc(AtomArena_R aa) MipsAtom_Proc_(aa, { mac_yield() })",
|
|
||||||
"FI_ void ac_store_pair(MipsAtomBuilder_R ab) atom_dbg_skip MipsAtomComp_Proc_(ab, { store_word(R_T0, R_T1, 0) })",
|
|
||||||
].join("\n");
|
|
||||||
|
|
||||||
const result = scanSource(source, "C:/projects/Pikuma/ps1/code/duffle/mips.atom.c");
|
|
||||||
|
|
||||||
assert.equal(result.index.atoms.has("cube_g4_face"), true);
|
|
||||||
assert.equal(result.index.atoms.has("normalize"), true);
|
|
||||||
assert.equal(result.index.components.has("ac_load_pair"), true);
|
|
||||||
assert.equal(result.index.components.has("ac_store_pair"), true);
|
|
||||||
assert.equal(result.index.componentAliases.has("mac_load_pair"), true);
|
|
||||||
assert.equal(result.index.componentAliases.has("mac_store_pair"), true);
|
|
||||||
assert.equal(result.index.macros.get("mac_store_pair"), "component");
|
|
||||||
assert.equal(result.index.componentCallees.get("mac_store_pair").includes("store_word"), true);
|
|
||||||
assert.equal(result.index.phases.has("cube_g4"), true);
|
|
||||||
assert.equal(result.index.registers.get("R_PrimCursor"), "gpr");
|
|
||||||
assert.deepEqual(result.errors, []);
|
|
||||||
});
|
|
||||||
|
|
||||||
test("scanSource discovers binds, labels, registers, typedefs, and macro domains", () => {
|
|
||||||
const source = [
|
|
||||||
"typedef Struct_(Binds_CubeTri) { U4 PrimCursor; };",
|
|
||||||
"typedef Enum_(U4, PadStatus) { PadStatus_Ok };",
|
|
||||||
"typedef U4 const MipsCode;",
|
|
||||||
"enum { R_PrimCursor = R_T7 atom_reg, C2_Custom = 12, gte_cr_Custom = 13 };",
|
|
||||||
"#define load_word(rt, base, off) enc_i(rt, base, off)",
|
|
||||||
"atom_bind(Binds_CubeTri)",
|
|
||||||
"atom_label(exit)",
|
|
||||||
"atom_offset(entry, exit)",
|
|
||||||
].join("\n");
|
|
||||||
|
|
||||||
const result = scanSource(source, "C:/projects/Pikuma/ps1/code/duffle/mips.h");
|
|
||||||
|
|
||||||
assert.equal(result.index.bindTypes.has("Binds_CubeTri"), true);
|
|
||||||
assert.equal(result.index.types.has("PadStatus"), true);
|
|
||||||
assert.equal(result.index.types.has("MipsCode"), true);
|
|
||||||
assert.equal(result.index.registers.get("R_PrimCursor"), "gpr");
|
|
||||||
assert.equal(result.index.registers.get("C2_Custom"), "cop2");
|
|
||||||
assert.equal(result.index.registers.get("gte_cr_Custom"), "cop2");
|
|
||||||
assert.equal(result.index.macros.get("load_word"), "cpu");
|
|
||||||
assert.equal(result.index.labels.has("entry"), true);
|
|
||||||
assert.equal(result.index.labels.has("exit"), true);
|
|
||||||
});
|
|
||||||
|
|
||||||
test("domainFromPath uses the declaration file rather than parent directory names", () => {
|
|
||||||
assert.equal(domainFromPath("C:/x/code/hello_gte/hello_gte.atom.c"), null);
|
|
||||||
assert.equal(domainFromPath("C:/x/code/duffle/mips.h"), "cpu");
|
|
||||||
assert.equal(domainFromPath("C:/x/code/duffle/gte.h"), "gte");
|
|
||||||
assert.equal(domainFromPath("C:/x/code/duffle/gp.h"), "gpu");
|
|
||||||
});
|
|
||||||
|
|
||||||
test("component aliases inherit the domain of the instructions they emit", () => {
|
|
||||||
const headers = mergeIndexes(
|
|
||||||
scanSource("#define load_word(a,b,c) 1\n#define store_word(a,b,c) 1\n#define shift_aright_var(a,b,c) 1\n#define jump_reg(rd) 1\n", "C:/x/code/duffle/mips.h").index,
|
|
||||||
scanSource("#define gte_sw(rt, base, off) 1\n", "C:/x/code/duffle/gte.h").index
|
|
||||||
);
|
|
||||||
const math = scanSource(
|
|
||||||
[
|
|
||||||
"MipsAtomComp_(ac_load_v3s4) { load_word(R_T0, R_T1, 0) };",
|
|
||||||
"#define mac_load_p3s4 mac_load_v3s4",
|
|
||||||
].join("\n"),
|
|
||||||
"C:/x/code/duffle/math.atom.c"
|
|
||||||
);
|
|
||||||
const shift = scanSource(
|
|
||||||
"MipsAtomComp_(ac_shift_aright_var_v3_self) { shift_aright_var(R_T0, R_T0, R_T1) };",
|
|
||||||
"C:/x/code/duffle/gte.atom.c"
|
|
||||||
);
|
|
||||||
const gte = scanSource(
|
|
||||||
"MipsAtomComp_(ac_gte_store_f3) { gte_sw(C2_SXY0, R_T0, 0) };",
|
|
||||||
"C:/x/code/duffle/gte.atom.c"
|
|
||||||
);
|
|
||||||
const yieldAtom = scanSource(
|
|
||||||
"MipsAtomComp_(ac_yield) { load_word(R_AtomJmp, R_TapePtr, 0), jump_reg(R_AtomJmp), nop };",
|
|
||||||
"C:/x/code/duffle/lottes_tape.h"
|
|
||||||
);
|
|
||||||
|
|
||||||
const merged = mergeIndexes(headers, math.index, shift.index, gte.index, yieldAtom.index);
|
|
||||||
assert.equal(merged.macros.get("mac_load_v3s4"), "cpu");
|
|
||||||
assert.equal(merged.macros.get("mac_load_p3s4"), "cpu");
|
|
||||||
assert.equal(merged.macros.get("mac_shift_aright_var_v3_self"), "cpu");
|
|
||||||
assert.equal(merged.macros.get("mac_gte_store_f3"), "gte");
|
|
||||||
assert.equal(merged.macros.get("mac_yield"), "control");
|
|
||||||
});
|
|
||||||
|
|
||||||
test("scanSource tags utility header defines as utility, not a hardware domain", () => {
|
|
||||||
const source = [
|
|
||||||
"#define assert(cond) ((void)(cond))",
|
|
||||||
"#define stringify(name) #name",
|
|
||||||
"#define u4_hi(imm) ((imm) >> 16)",
|
|
||||||
].join("\n");
|
|
||||||
const result = scanSource(source, "C:/projects/Pikuma/ps1/code/duffle/dsl.h");
|
|
||||||
|
|
||||||
assert.equal(result.index.macros.get("assert"), "utility");
|
|
||||||
assert.equal(result.index.macros.get("stringify"), "utility");
|
|
||||||
assert.equal(result.index.macros.get("u4_hi"), "utility");
|
|
||||||
});
|
|
||||||
|
|
||||||
test("mergeIndexes prefers a hardware domain over a later utility define", () => {
|
|
||||||
const left = createIndex();
|
|
||||||
left.macros.set("sub_s", "utility");
|
|
||||||
const right = createIndex();
|
|
||||||
right.macros.set("sub_s", "cpu");
|
|
||||||
|
|
||||||
assert.equal(mergeIndexes(left, right).macros.get("sub_s"), "cpu");
|
|
||||||
assert.equal(mergeIndexes(right, left).macros.get("sub_s"), "cpu");
|
|
||||||
});
|
|
||||||
|
|
||||||
test("mergeIndexes preserves domain-specific aliases", () => {
|
|
||||||
const left = createIndex();
|
|
||||||
left.macros.set("load_word", "cpu");
|
|
||||||
const right = createIndex();
|
|
||||||
right.componentAliases.add("mac_gte_store");
|
|
||||||
right.macros.set("mac_gte_store", "gte");
|
|
||||||
|
|
||||||
const merged = mergeIndexes(left, right);
|
|
||||||
assert.equal(merged.macros.get("load_word"), "cpu");
|
|
||||||
assert.equal(merged.macros.get("mac_gte_store"), "gte");
|
|
||||||
});
|
|
||||||
@@ -1,17 +1,24 @@
|
|||||||
Copyright (C) 2026 Edward R. Gonzalez
|
This is free and unencumbered software released into the public domain.
|
||||||
|
|
||||||
This software is provided 'as-is', without any express or implied
|
Anyone is free to copy, modify, publish, use, compile, sell, or
|
||||||
warranty. In no event will the authors be held liable for any damages
|
distribute this software, either in source code form or as a compiled
|
||||||
arising from the use of this software.
|
binary, for any purpose, commercial or non-commercial, and by any
|
||||||
|
means.
|
||||||
|
|
||||||
Permission is granted to anyone to use this software for any purpose,
|
In jurisdictions that recognize copyright laws, the author or authors
|
||||||
including commercial applications, and to alter it and redistribute it
|
of this software dedicate any and all copyright interest in the
|
||||||
freely, subject to the following restrictions:
|
software to the public domain. We make this dedication for the benefit
|
||||||
|
of the public at large and to the detriment of our heirs and
|
||||||
|
successors. We intend this dedication to be an overt act of
|
||||||
|
relinquishment in perpetuity of all present and future rights to this
|
||||||
|
software under copyright law.
|
||||||
|
|
||||||
1. The origin of this software must not be misrepresented; you must not
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
claim that you wrote the original software. If you use this software
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
in a product, an acknowledgment in the product documentation would be
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||||
appreciated but is not required.
|
IN NO EVENT SHALL THE AUTHORS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||||
2. Altered source versions must be plainly marked as such, and must not be
|
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||||
misrepresented as being the original software.
|
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||||
3. This notice may not be removed or altered from any source distribution.
|
OTHER DEALINGS IN THE SOFTWARE.
|
||||||
|
|
||||||
|
For more information, please refer to <https://unlicense.org>
|
||||||
|
|||||||
@@ -1,15 +0,0 @@
|
|||||||
#ifdef INTELLISENSE_DIRECTIVES
|
|
||||||
# pragma once
|
|
||||||
#endif
|
|
||||||
|
|
||||||
enum {
|
|
||||||
bios_init_pad_2 = 0x12,
|
|
||||||
bios_start_pad_2 = 0x13,
|
|
||||||
bios_flushcache = 0x44,
|
|
||||||
bios_table_addr = 0xA0,
|
|
||||||
bios_btable_addr = 0xB0,
|
|
||||||
};
|
|
||||||
|
|
||||||
enum {
|
|
||||||
bios_pad_buffer_size = 0x22,
|
|
||||||
};
|
|
||||||
+2
-22
@@ -70,31 +70,11 @@
|
|||||||
/* ----------------------------------------------------------------------------
|
/* ----------------------------------------------------------------------------
|
||||||
* atom_reg (per-enum opt-in marker for the DWARF register-alias registry)
|
* atom_reg (per-enum opt-in marker for the DWARF register-alias registry)
|
||||||
*
|
*
|
||||||
* Bare `atom_reg` token adjacent to an enum entry that alias as debug-visible for scan_source's register_alias_registry.
|
* The bare `atom_reg` token adjacent to an enum entry in mips.h / lottes_tape.h flags that alias as debug-visible for scan_source's register_alias_registry.
|
||||||
* Lua scanner reads the bare token.
|
* The C preprocessor strips it to a comment so no runtime symbol is created; the Lua scanner reads the bare token.
|
||||||
* ----------------------------------------------------------------------------*/
|
* ----------------------------------------------------------------------------*/
|
||||||
#define atom_reg /* atom_reg: opt the preceding enum entry into the DWARF registry */
|
#define atom_reg /* atom_reg: opt the preceding enum entry into the DWARF registry */
|
||||||
|
|
||||||
// ----------------------------------------------------------------------------
|
|
||||||
// atom_auto_reg(atom, sym) — per-atom auto-allocated GPR binding.
|
|
||||||
// enum {
|
|
||||||
// atom_auto_reg(cube_g4_face, R_Fwdx), // expands to: R_Fwdx = R_Fwdx_Code /* atom_auto_reg: cube_g4_face */,
|
|
||||||
// atom_auto_reg(cube_g4_face, R_Eye_z) atom_type(S4), // atom_type chains after
|
|
||||||
// };
|
|
||||||
// (The macro IS the entire enum entry — no separate LHS=RHS. The `atom` scope is
|
|
||||||
// preserved in a trailing C-comment on the RHS so the Lua scanner can recover
|
|
||||||
// it after preprocessing strips the macro form. R_<Sym>_Code is resolved from gen/auto_reg.h which the .c file #include's before the enum declaration.)
|
|
||||||
#define atom_auto_reg(atom, sym) sym = sym ## _Code /* atom_auto_reg: atom */
|
|
||||||
|
|
||||||
// ----------------------------------------------------------------------------
|
|
||||||
// phase_auto_reg(phase, sym) — per-phase auto-allocated GPR binding.
|
|
||||||
// enum {
|
|
||||||
// phase_auto_reg(cube_g4, R_Temp0), // expands to: R_Temp0 = R_Temp0_Code /* phase_auto_reg: cube_g4 */,
|
|
||||||
// phase_auto_reg(cube_g4, R_Temp1),
|
|
||||||
// };
|
|
||||||
// (Same macro-as-enum-entry form as atom_auto_reg above; the `phase` scope is preserved in a trailing C-comment on the RHS for the Lua scanner to recover.)
|
|
||||||
#define phase_auto_reg(phase, sym) sym = sym ## _Code /* phase_auto_reg: phase */
|
|
||||||
|
|
||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
* atom_info :
|
* atom_info :
|
||||||
* MipsAtom_(cube_tri) atom_info(
|
* MipsAtom_(cube_tri) atom_info(
|
||||||
|
|||||||
+22
-29
@@ -3,7 +3,7 @@
|
|||||||
# include "assert.h"
|
# include "assert.h"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#define offset_of(type, member) cast(U8,__builtin_offsetof(type,member)) // Compiler builtin version of O_
|
#define offset_of(type, member) cast(U8,__builtin_offsetof(type,member))
|
||||||
#define static_assert _Static_assert
|
#define static_assert _Static_assert
|
||||||
#define typeof __typeof__
|
#define typeof __typeof__
|
||||||
#define typeof_ptr(ptr) typeof((ptr)[0])
|
#define typeof_ptr(ptr) typeof((ptr)[0])
|
||||||
@@ -28,9 +28,8 @@
|
|||||||
#define internal static // internal
|
#define internal static // internal
|
||||||
|
|
||||||
#define asm __asm__
|
#define asm __asm__
|
||||||
|
|
||||||
#define A_(data) (& data)
|
|
||||||
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
||||||
|
|
||||||
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
||||||
#define C_(type,data) ((type)(data)) // for enforced precedence
|
#define C_(type,data) ((type)(data)) // for enforced precedence
|
||||||
#define expect_(x, y) __builtin_expect(x, y) // so compiler knows the common path
|
#define expect_(x, y) __builtin_expect(x, y) // so compiler knows the common path
|
||||||
@@ -91,13 +90,12 @@
|
|||||||
#define PtrSet_(type) TypeR_(type); typedef TypeV_(type)
|
#define PtrSet_(type) TypeR_(type); typedef TypeV_(type)
|
||||||
#define TSet_(type) type; typedef PtrSet_(type)
|
#define TSet_(type) type; typedef PtrSet_(type)
|
||||||
|
|
||||||
#define Array_len(a) (U4)(sizeof(a) / sizeof(typeof((a)[0])))
|
#define array_len(a) (U4)(sizeof(a) / sizeof(typeof((a)[0])))
|
||||||
#define Array_decl(type, ...) (type[]){__VA_ARGS__}
|
#define array_decl(type, ...) (type[]){__VA_ARGS__}
|
||||||
#define Array_sym(type,len) A ## len ## _ ## type
|
#define Array_sym(type,len) A ## len ## _ ## type
|
||||||
#define Array_expand(type,len) type Array_sym(type, len)[len]; typedef PtrSet_(Array_sym(type, len))
|
#define Array_expand(type,len) type Array_sym(type, len)[len]; typedef PtrSet_(Array_sym(type, len))
|
||||||
#define Array_(type,len) Array_expand(type,len)
|
#define Array_(type,len) Array_expand(type,len)
|
||||||
#define Bit_(id,b) id = (1 << b), tmpl(id,pos) = b
|
#define Bit_(id,b) id = (1 << b), tmpl(id,pos) = b
|
||||||
#define Bitmask_(b) (1u << b)
|
|
||||||
#define Enum_(underlying_type, symbol) underlying_type TSet_(symbol); enum symbol
|
#define Enum_(underlying_type, symbol) underlying_type TSet_(symbol); enum symbol
|
||||||
#define Proc_(symbol) symbol
|
#define Proc_(symbol) symbol
|
||||||
#define Relative_(symbol) // Does nothing but annotate that a symbol is associated with another.
|
#define Relative_(symbol) // Does nothing but annotate that a symbol is associated with another.
|
||||||
@@ -135,22 +133,22 @@ typedef __UINT32_TYPE__ TSet_(B4);
|
|||||||
#define u4_v(value) C_(U4 V_*, value)
|
#define u4_v(value) C_(U4 V_*, value)
|
||||||
enum { false = 0, true = 1, true_overflow, };
|
enum { false = 0, true = 1, true_overflow, };
|
||||||
|
|
||||||
#define u4_lo(value) (u4_(value) & 0xFFFFU)
|
#define u4_lo(value) ((value) & 0xFFFFU)
|
||||||
#define u4_hi(value) (u4_(value) >> (S_(U2) * 8))
|
#define u4_hi(value) ((value) >> 12)
|
||||||
|
|
||||||
typedef void Proc_(VoidFn) (void);
|
typedef void Proc_(VoidFn) (void);
|
||||||
|
|
||||||
#define Kilo_(n) (C_(U4, n) << 10)
|
#define kilo(n) (C_(U4, n) << 10)
|
||||||
#define Mega_(n) (C_(U4, n) << 20)
|
#define mega(n) (C_(U4, n) << 20)
|
||||||
#define Giga_(n) (C_(U4, n) << 30)
|
#define giga(n) (C_(U4, n) << 30)
|
||||||
#define Tera_(n) (C_(U4, n) << 40)
|
#define tera(n) (C_(U4, n) << 40)
|
||||||
|
|
||||||
#define null C_(U4, 0)
|
#define null C_(U4, 0)
|
||||||
#define nullptr C_(void*, 0)
|
#define nullptr C_(void*, 0)
|
||||||
#define O_(type, field) C_(U4, & C_(type*,0)->field)
|
#define O_(type, field) C_(U4, & C_(type*,0)->field)
|
||||||
#define OA_(type, aexpr) C_(U4, & C_(type*,0) aexpr)
|
#define OA_(type, member, idx) C_(U4, & C_(type*,0)->member[idx])
|
||||||
#define OT_(field) O_(typeof_ptr(& field), field))
|
#define OT_(field) O_(typeof_ptr(& field), filed))
|
||||||
#define S_(data) C_(U4, sizeof(data))
|
#define S_(data) C_(U4, sizeof(data))
|
||||||
|
|
||||||
#define sop_1(op,a,b) C_(U1, s1_(a) op s1_(b))
|
#define sop_1(op,a,b) C_(U1, s1_(a) op s1_(b))
|
||||||
#define sop_2(op,a,b) C_(U2, s2_(a) op s2_(b))
|
#define sop_2(op,a,b) C_(U2, s2_(a) op s2_(b))
|
||||||
@@ -170,8 +168,6 @@ def_signed_ops(le, <=)
|
|||||||
#undef def_signed_ops
|
#undef def_signed_ops
|
||||||
#undef def_signed_op
|
#undef def_signed_op
|
||||||
|
|
||||||
// Unused, we arent' doing any C-like asm since we have the asm dsl. We'll keep the non-generics if we somehow do.
|
|
||||||
#if 0
|
|
||||||
#define def_generic_sop(op, a, ...) _Generic((a), U1: op ## _s1, U2: op ## _s2, U4: op ## _s4) (a, __VA_ARGS__)
|
#define def_generic_sop(op, a, ...) _Generic((a), U1: op ## _s1, U2: op ## _s2, U4: op ## _s4) (a, __VA_ARGS__)
|
||||||
#define add_s(a,b) def_generic_sop(add,a,b)
|
#define add_s(a,b) def_generic_sop(add,a,b)
|
||||||
#define sub_s(a,b) def_generic_sop(sub,a,b)
|
#define sub_s(a,b) def_generic_sop(sub,a,b)
|
||||||
@@ -181,12 +177,11 @@ def_signed_ops(le, <=)
|
|||||||
#define ge_s(a,b) def_generic_sop(ge, a,b)
|
#define ge_s(a,b) def_generic_sop(ge, a,b)
|
||||||
#define le_s(a,b) def_generic_sop(le, a,b)
|
#define le_s(a,b) def_generic_sop(le, a,b)
|
||||||
#undef def_generic_sop
|
#undef def_generic_sop
|
||||||
#endif
|
|
||||||
|
|
||||||
#define alignas _Alignas
|
#define alignas _Alignas
|
||||||
#define alignof _Alignof
|
#define alignof _Alignof
|
||||||
#define byte_pad(amount, ...) B1 glue(_PAD_, __VA_ARGS__) [amount]
|
#define byte_pad(amount, ...) B1 glue(_PAD_, __VA_ARGS__) [amount]
|
||||||
#define C_ptr(type, data) (C_(type*, & (data)) [0])
|
#define pcast(type, data) (C_(type*, & (data)) [0])
|
||||||
|
|
||||||
#define dbg_args(...) __VA_ARGS__
|
#define dbg_args(...) __VA_ARGS__
|
||||||
|
|
||||||
@@ -201,8 +196,6 @@ def_signed_ops(le, <=)
|
|||||||
#define defer_info(type,expr, ...) for(type info= {__VA_ARGS__}; info.once!=1;++info.once,(expr)) // Defer with tracked state
|
#define defer_info(type,expr, ...) for(type info= {__VA_ARGS__}; info.once!=1;++info.once,(expr)) // Defer with tracked state
|
||||||
|
|
||||||
#define do_while(cond) for (U8 once=0; once!=1 || (cond); ++once)
|
#define do_while(cond) for (U8 once=0; once!=1 || (cond); ++once)
|
||||||
|
|
||||||
#define Jmp_nZero_(cond,label) if (cond) goto label;
|
|
||||||
#pragma endregion Control Flow & Iteration
|
#pragma endregion Control Flow & Iteration
|
||||||
|
|
||||||
#define span_iter(type, iter, m_begin, op, m_end) ( \
|
#define span_iter(type, iter, m_begin, op, m_end) ( \
|
||||||
@@ -219,16 +212,16 @@ def_signed_ops(le, <=)
|
|||||||
typedef Span_(S4);
|
typedef Span_(S4);
|
||||||
typedef Span_(U4);
|
typedef Span_(U4);
|
||||||
|
|
||||||
|
#if 0
|
||||||
#pragma region Debug
|
#pragma region Debug
|
||||||
#define debug_trap() __builtin_trap()
|
#define debug_trap() __builtin_debugtrap()
|
||||||
#if BUILD_DEBUG
|
#if BUILD_DEBUG
|
||||||
#define assert(cond) if(cond == false){debug_trap();}
|
IA_ void assert(U8 cond) { if(cond){return;} else{debug_trap(); ms_exit_process(1);} }
|
||||||
#else
|
#else
|
||||||
# ifndef assert
|
#define assert(cond)
|
||||||
# include <assert.h>
|
|
||||||
# endif
|
|
||||||
#endif
|
#endif
|
||||||
#pragma endregion Debug
|
#pragma endregion Debug
|
||||||
|
#endif
|
||||||
|
|
||||||
#define GCC_OPTIMIZATION_DISABLE _Pragma("GCC push_options") _Pragma("GCC optimize(\"O0\")")
|
#define GCC_OPTIMIZATION_DISABLE _Pragma("GCC push_options") _Pragma("GCC optimize(\"O0\")")
|
||||||
#define GCC_OPTIMIZATION_ENABLE _Pragma("GCC pop_options")
|
#define GCC_OPTIMIZATION_ENABLE _Pragma("GCC pop_options")
|
||||||
|
|||||||
+31
-254
@@ -14,10 +14,8 @@
|
|||||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h
|
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
|
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
|
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
|
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
|
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.h
|
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
|
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
|
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
|
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
|
||||||
@@ -35,11 +33,15 @@
|
|||||||
* These do NOT yield. They are expanded inline inside Tape Atoms.
|
* These do NOT yield. They are expanded inline inside Tape Atoms.
|
||||||
* ---------------------------------------------------------------------------*/
|
* ---------------------------------------------------------------------------*/
|
||||||
// The 'Yield' sequence for Tape Atoms (mac_yield).
|
// The 'Yield' sequence for Tape Atoms (mac_yield).
|
||||||
|
// - mac_yield() is the safe default for atom-endings: 4 words, BD-slot of jr is mandatory nop.
|
||||||
|
// - mac_yield_load() + mac_yield_tail():
|
||||||
|
// - unconditional branch: mac_yield_load fills the branch's BD-slot (replaces a nop);
|
||||||
|
// - mac_yield_tail runs at the branch target (does NOT re-load R_AtomJmp).
|
||||||
#define mac_yield(...) \
|
#define mac_yield(...) \
|
||||||
load_word(R_AtomJmp, R_TapePtr, 0) \
|
load_word(R_AtomJmp, R_TapePtr, 0) \
|
||||||
, add_ui_self( R_TapePtr, S_(MipsCode)) \
|
, add_ui_self( R_TapePtr, S_(MipsCode)) \
|
||||||
, jump_reg( R_AtomJmp) \
|
, jump_reg( R_AtomJmp) \
|
||||||
, BdSlot_ nop
|
, nop
|
||||||
WORD_COUNT(mac_yield, 4)
|
WORD_COUNT(mac_yield, 4)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
@@ -50,25 +52,14 @@ WORD_COUNT(mac_yield_load, 1)
|
|||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
#define mac_yield_tail(...) \
|
#define mac_yield_tail(...) \
|
||||||
add_ui_self(R_TapePtr, S_(MipsCode)) \
|
add_ui_self(R_TapePtr, S_(MipsCode)) \
|
||||||
, jump_reg( R_AtomJmp) \
|
, jump_reg( R_AtomJmp) \
|
||||||
, BdSlot_ nop
|
, nop
|
||||||
WORD_COUNT(mac_yield_tail, 3)
|
WORD_COUNT(mac_yield_tail, 3)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_load_half_v3(tx, ty, tz, base, offset) \
|
|
||||||
load_half(tx, base, offset + OA_(U2,[0])) \
|
|
||||||
, load_half(ty, base, offset + OA_(U2,[1])) \
|
|
||||||
, load_half(tz, base, offset + OA_(U2,[2]))
|
|
||||||
WORD_COUNT(mac_load_half_v3, 3)
|
|
||||||
|
|
||||||
#define mac_load_v3s2(transfer, base, offset) \
|
|
||||||
mac_load_half_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
|
||||||
WORD_COUNT(mac_load_v3s2, 3)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
#define mac_load_v2s2(rs_x, rs_y, r_base, offset) \
|
#define mac_load_v2s2(rs_x, rs_y, r_base, offset) \
|
||||||
load_half(rs_x, r_base, offset + O_(V3_S2,x)) \
|
load_half( rs_x, r_base, O_(V3_S2,x)) \
|
||||||
, load_half(rs_y, r_base, offset + O_(V3_S2,y))
|
, load_half( rs_y, r_base, O_(V3_S2,y))
|
||||||
WORD_COUNT(mac_load_v2s2, 2)
|
WORD_COUNT(mac_load_v2s2, 2)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
@@ -77,76 +68,6 @@ WORD_COUNT(mac_load_v2s2, 2)
|
|||||||
, store_half(rt_y, base, offset + O_(V2_S2,y))
|
, store_half(rt_y, base, offset + O_(V2_S2,y))
|
||||||
WORD_COUNT(mac_store_v2s2, 2)
|
WORD_COUNT(mac_store_v2s2, 2)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_load_word_v3(tx, ty, tz, base, offset) \
|
|
||||||
load_word(tx, base, offset + OA_(U4,[0])) \
|
|
||||||
, load_word(ty, base, offset + OA_(U4,[1])) \
|
|
||||||
, load_word(tz, base, offset + OA_(U4,[2]))
|
|
||||||
WORD_COUNT(mac_load_word_v3, 3)
|
|
||||||
|
|
||||||
#define mac_load_v3s4(transfer, base, offset) \
|
|
||||||
mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
|
||||||
WORD_COUNT(mac_load_v3s4, 3)
|
|
||||||
|
|
||||||
#define mac_load_p3s4(transfer, base, offset) \
|
|
||||||
mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
|
||||||
WORD_COUNT(mac_load_p3s4, 3)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_store_half_v3(tx, ty, tz, base, offset) \
|
|
||||||
store_half(tx, base, offset + OA_(U2,[0])) \
|
|
||||||
, store_half(ty, base, offset + OA_(U2,[1])) \
|
|
||||||
, store_half(tz, base, offset + OA_(U2,[2]))
|
|
||||||
WORD_COUNT(mac_store_half_v3, 3)
|
|
||||||
|
|
||||||
#define mac_store_v3s2(transfer, base, offset) \
|
|
||||||
mac_store_half_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
|
||||||
WORD_COUNT(mac_store_v3s2, 3)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_store_word_v3(tx, ty, tz, base, offset) \
|
|
||||||
store_word(tx, base, offset + OA_(U4,[0])) \
|
|
||||||
, store_word(ty, base, offset + OA_(U4,[1])) \
|
|
||||||
, store_word(tz, base, offset + OA_(U4,[2]))
|
|
||||||
WORD_COUNT(mac_store_word_v3, 3)
|
|
||||||
|
|
||||||
#define mac_store_v3s4(transfer, base, offset) \
|
|
||||||
mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
|
||||||
WORD_COUNT(mac_store_v3s4, 3)
|
|
||||||
|
|
||||||
#define mac_store_p3s4(transfer, base, offset) \
|
|
||||||
mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
|
||||||
WORD_COUNT(mac_store_p3s4, 3)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_add_si_v3s4(rt_x, rt_y, rt_z, base, offset) \
|
|
||||||
add_si(rt_x, base, O_(V3_S4,x)) \
|
|
||||||
, add_si(rt_y, base, O_(V3_S4,y)) \
|
|
||||||
, add_si(rt_z, base, O_(V3_S4,z))
|
|
||||||
WORD_COUNT(mac_add_si_v3s4, 3)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_sub_s_v3(dx, dy, dz, sx, sy, sz, tx, ty, tz) \
|
|
||||||
sub_s(dx, sx, tx) \
|
|
||||||
, sub_s(dy, sy, ty) \
|
|
||||||
, sub_s(dz, sz, tz)
|
|
||||||
WORD_COUNT(mac_sub_s_v3, 3)
|
|
||||||
|
|
||||||
#define mac_sub_v3s4(d, s, t) \
|
|
||||||
mac_sub_s_v3(d.x, d.y, d.z, s.x, s.y, s.z, t.x, t.y, t.z)
|
|
||||||
WORD_COUNT(mac_sub_v3s4, 3)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_sub_s_v3_self(ds_x, ds_y, ds_z, tx, ty, tz) \
|
|
||||||
sub_s(ds_x, ds_x, tx) \
|
|
||||||
, sub_s(ds_y, ds_y, ty) \
|
|
||||||
, sub_s(ds_z, ds_z, tz)
|
|
||||||
WORD_COUNT(mac_sub_s_v3_self, 3)
|
|
||||||
|
|
||||||
#define mac_sub_v3s4_self(ds, t) \
|
|
||||||
mac_sub_s_v3_self(ds.x, ds.y, ds.z, t.x, t.y, t.z)
|
|
||||||
WORD_COUNT(mac_sub_v3s4_self, 3)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
#define mac_store_rects2(rt_x, rt_y, rt_width, rt_height, base, offset) \
|
#define mac_store_rects2(rt_x, rt_y, rt_width, rt_height, base, offset) \
|
||||||
store_half(rt_x, base, offset + O_(Rect_S2,x)) \
|
store_half(rt_x, base, offset + O_(Rect_S2,x)) \
|
||||||
@@ -155,39 +76,6 @@ WORD_COUNT(mac_sub_v3s4_self, 3)
|
|||||||
, store_half(rt_height, base, offset + O_(Rect_S2,height))
|
, store_half(rt_height, base, offset + O_(Rect_S2,height))
|
||||||
WORD_COUNT(mac_store_rects2, 4)
|
WORD_COUNT(mac_store_rects2, 4)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_load_word_imm(dst, imm) \
|
|
||||||
load_upper_i(dst, u4_hi(imm)) \
|
|
||||||
, or_i_self( dst, u4_lo(imm))
|
|
||||||
WORD_COUNT(mac_load_word_imm, 2)
|
|
||||||
|
|
||||||
#define mac_shift_aright_v3_self(dt_x, dt_y, dt_z, shift_amount) \
|
|
||||||
shift_aright(dt_x, dt_x, shift_amount) \
|
|
||||||
, shift_aright(dt_y, dt_y, shift_amount) \
|
|
||||||
, shift_aright(dt_z, dt_z, shift_amount)
|
|
||||||
WORD_COUNT(mac_shift_aright_v3_self, 3)
|
|
||||||
|
|
||||||
#define mac_shift_aright_v3s4_self(dt, shift) \
|
|
||||||
mac_shift_aright_v3_self(dt.x, dt.y, dt.z, shift)
|
|
||||||
WORD_COUNT(mac_shift_aright_v3s4_self, 3)
|
|
||||||
|
|
||||||
#define mac_shift_aright_var_v3(rd_v0, rd_v1, rd_v2, rs_v0, rs_v1, rs_v2, r_shift) \
|
|
||||||
shift_aright_var(rd_v0, rs_v0, r_shift) \
|
|
||||||
, shift_aright_var(rd_v1, rs_v1, r_shift) \
|
|
||||||
, shift_aright_var(rd_v2, rs_v2, r_shift)
|
|
||||||
WORD_COUNT(mac_shift_aright_var_v3, 3)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_shift_aright_var_v3_self(rds_v0, rds_v1, rds_v2, r_shift) \
|
|
||||||
shift_aright_var(rds_v0, rds_v0, r_shift) \
|
|
||||||
, shift_aright_var(rds_v1, rds_v1, r_shift) \
|
|
||||||
, shift_aright_var(rds_v2, rds_v2, r_shift)
|
|
||||||
WORD_COUNT(mac_shift_aright_var_v3_self, 3)
|
|
||||||
|
|
||||||
#define mac_shift_aright_var_v3s4_self(ds, shift) \
|
|
||||||
mac_shift_aright_var_v3_self(ds.x, ds.y, ds.z, shift)
|
|
||||||
WORD_COUNT(mac_shift_aright_var_v3s4_self, 3)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
#define mac_load_tri_indices(r_face_cusor, r_i0, r_i1, r_i2) \
|
#define mac_load_tri_indices(r_face_cusor, r_i0, r_i1, r_i2) \
|
||||||
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)) \
|
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)) \
|
||||||
@@ -195,30 +83,6 @@ WORD_COUNT(mac_shift_aright_var_v3s4_self, 3)
|
|||||||
, load_half_u(r_i2, r_face_cusor, 2 * S_(S2))
|
, load_half_u(r_i2, r_face_cusor, 2 * S_(S2))
|
||||||
WORD_COUNT(mac_load_tri_indices, 3)
|
WORD_COUNT(mac_load_tri_indices, 3)
|
||||||
|
|
||||||
#define mac_gte_mv_to_cr_diag_v3s4(v) \
|
|
||||||
gte_mv_to_ctrl_r(v.y, gte_cr_RT13) \
|
|
||||||
, gte_mv_to_ctrl_r(v.z, gte_cr_RT22) \
|
|
||||||
, gte_mv_to_ctrl_r(v.x, gte_cr_RT11)
|
|
||||||
WORD_COUNT(mac_gte_mv_to_cr_diag_v3s4, 3)
|
|
||||||
|
|
||||||
#define mac_gte_ld_ir123_v3s4(v) \
|
|
||||||
gte_mv_to_data_r(v.x, C2_IR1) \
|
|
||||||
, gte_mv_to_data_r(v.y, C2_IR2) \
|
|
||||||
, gte_mv_to_data_r(v.z, C2_IR3)
|
|
||||||
WORD_COUNT(mac_gte_ld_ir123_v3s4, 3)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_gte_op_cross_v3s4(a, b) \
|
|
||||||
mac_gte_mv_to_cr_diag_v3s4(a) \
|
|
||||||
GteDelay_ /* RT diagonal: D1 = a.x, D2 = a.y, D3 = a.z */ \
|
|
||||||
, mac_gte_ld_ir123_v3s4(b) \
|
|
||||||
GteDelay_ /* IR: second operand (b.xyz) */ \
|
|
||||||
, gte_cmdw_cross /* OP: MAC1/2/3 = a × b (S12.20) */ \
|
|
||||||
, mac_gte_mv_from_mac123_v3s4(a) \
|
|
||||||
GteDelay_ /* Read MAC1/2/3 → a.xyz (overwrites source-A's load targets) */ \
|
|
||||||
, mac_shift_aright_v3s4_self(a, 12) /* Right-shift MAC by 12 (S12.20 → S12.0 OuterProduct12) */
|
|
||||||
WORD_COUNT(mac_gte_op_cross_v3s4, 16)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
#define mac_gte_store_f3(r_primitive_cursor) \
|
#define mac_gte_store_f3(r_primitive_cursor) \
|
||||||
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)) \
|
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)) \
|
||||||
@@ -232,19 +96,19 @@ WORD_COUNT(mac_gte_store_f3, 3)
|
|||||||
, add_u_self(R_AT, r_vert_base) \
|
, add_u_self(R_AT, r_vert_base) \
|
||||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||||
, LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY0) \
|
, gte_mv_to_data_r(R_V0, C2_VXY0) \
|
||||||
, gte_mv_to_data_r(R_V1, C2_VZ0) \
|
, gte_mv_to_data_r(R_V1, C2_VZ0) \
|
||||||
, shift_lleft(R_AT, r_v1, v3s2_byteoff) \
|
, shift_lleft(R_AT, r_v1, v3s2_byteoff) \
|
||||||
, add_u_self(R_AT, r_vert_base) \
|
, add_u_self(R_AT, r_vert_base) \
|
||||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||||
, LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY1) \
|
, gte_mv_to_data_r(R_V0, C2_VXY1) \
|
||||||
, gte_mv_to_data_r(R_V1, C2_VZ1) \
|
, gte_mv_to_data_r(R_V1, C2_VZ1) \
|
||||||
, shift_lleft(R_AT, r_v2, v3s2_byteoff) \
|
, shift_lleft(R_AT, r_v2, v3s2_byteoff) \
|
||||||
, add_u_self(R_AT, r_vert_base) \
|
, add_u_self(R_AT, r_vert_base) \
|
||||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||||
, LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY2) \
|
, gte_mv_to_data_r(R_V0, C2_VXY2) \
|
||||||
, gte_mv_to_data_r(R_V1, C2_VZ2)
|
, gte_mv_to_data_r(R_V1, C2_VZ2)
|
||||||
WORD_COUNT(mac_gte_load_tri_verts, 18)
|
WORD_COUNT(mac_gte_load_tri_verts, 18)
|
||||||
|
|
||||||
@@ -260,84 +124,10 @@ WORD_COUNT(mac_gte_store_g4_p012, 3)
|
|||||||
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3))
|
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3))
|
||||||
WORD_COUNT(mac_gte_store_g4_p3, 1)
|
WORD_COUNT(mac_gte_store_g4_p3, 1)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_gte_sqr_v3(r_sx, r_sy, r_sz, r_sq_x, r_sq_y, r_sq_z) \
|
|
||||||
mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop) \
|
|
||||||
, gte_mv_from_data_r(r_sq_x, C2_MAC1) \
|
|
||||||
, gte_mv_from_data_r(r_sq_y, C2_MAC2) \
|
|
||||||
, gte_mv_from_data_r(r_sq_z, C2_MAC3)
|
|
||||||
WORD_COUNT(mac_gte_sqr_v3, 8)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, delay_slot) \
|
|
||||||
gte_mv_to_data_r(r_sx, C2_IR1) \
|
|
||||||
, gte_mv_to_data_r(r_sy, C2_IR2) \
|
|
||||||
, gte_mv_to_data_r(r_sz, C2_IR3) \
|
|
||||||
, delay_slot \
|
|
||||||
, gte_cmdw_sqr
|
|
||||||
WORD_COUNT(mac_gte_sqr_v3s4, 5)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_gte_gpf_scale(r_sx, r_sy, r_sz, r_recip_est, r_shift, r_dx, r_dy, r_dz) \
|
|
||||||
gte_mv_to_data_r(r_recip_est, C2_IR0) \
|
|
||||||
, gte_mv_to_data_r(r_sx, C2_IR1) \
|
|
||||||
, gte_mv_to_data_r(r_sy, C2_IR2) \
|
|
||||||
, gte_mv_to_data_r(r_sz, C2_IR3) \
|
|
||||||
, GteDelay_ nop2 /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */ \
|
|
||||||
, gte_cmdw_gpf \
|
|
||||||
, gte_mv_from_data_r(r_dx, C2_MAC1) \
|
|
||||||
, gte_mv_from_data_r(r_dy, C2_MAC2) \
|
|
||||||
, gte_mv_from_data_r(r_dz, C2_MAC3) \
|
|
||||||
, shift_aright_var(r_dx, r_dx, r_shift) \
|
|
||||||
, shift_aright_var(r_dy, r_dy, r_shift) \
|
|
||||||
, shift_aright_var(r_dz, r_dz, r_shift)
|
|
||||||
WORD_COUNT(mac_gte_gpf_scale, 12)
|
|
||||||
|
|
||||||
#define mac_trans_mt3s3s4(r_mtx, r_off, r_t0, r_t1, r_t2) \
|
|
||||||
load_word( r_t0, r_off, O_(V3_S4,x)) \
|
|
||||||
, load_word( r_t1, r_off, O_(V3_S4,y)) \
|
|
||||||
, load_word( r_t2, r_off, O_(V3_S4,z)) \
|
|
||||||
, store_word(r_t0, r_mtx, O_(MT3_S2S4,t[0])) \
|
|
||||||
, store_word(r_t1, r_mtx, O_(MT3_S2S4,t[1])) \
|
|
||||||
, store_word(r_t2, r_mtx, O_(MT3_S2S4,t[2]))
|
|
||||||
WORD_COUNT(mac_trans_mt3s3s4, 6)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_lzcr_round_even_half_shift(r_shift, r_mag_sq, r_mag_sq_copy) \
|
|
||||||
and_i(r_shift, r_shift, gte_lzcr_even_mask) \
|
|
||||||
, or_u(r_mag_sq_copy, r_mag_sq, 0) \
|
|
||||||
, li_s( r_mag_sq, 31) \
|
|
||||||
, sub_s( r_mag_sq, r_mag_sq, r_shift) \
|
|
||||||
, shift_aright(r_mag_sq, r_mag_sq, 1)
|
|
||||||
WORD_COUNT(mac_lzcr_round_even_half_shift, 5)
|
|
||||||
|
|
||||||
#define mac_gte_general_purpose_interopolation(to_ir0, to_ir1, to_ir2, to_ir3, fr_mac1, fr_mac2, fr_mac3, nop_slot1, nop_slot2) \
|
|
||||||
gte_mv_to_data_r(to_ir0, C2_IR0) \
|
|
||||||
, gte_mv_to_data_r(to_ir1, C2_IR1) /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */ \
|
|
||||||
, gte_mv_to_data_r(to_ir2, C2_IR2) \
|
|
||||||
, gte_mv_to_data_r(to_ir3, C2_IR3) /* IR3 = src.z (reloaded) */ \
|
|
||||||
, GteDelay_ nop_slot1 \
|
|
||||||
, GteDelay_ nop_slot2 \
|
|
||||||
, gte_cmdw_gpf \
|
|
||||||
, gte_mv_from_data_r(fr_mac1, C2_MAC1) \
|
|
||||||
, gte_mv_from_data_r(fr_mac2, C2_MAC2) \
|
|
||||||
, gte_mv_from_data_r(fr_mac3, C2_MAC3)
|
|
||||||
WORD_COUNT(mac_gte_general_purpose_interopolation, 10)
|
|
||||||
|
|
||||||
#define mac_gte_mv_from_data_r_mac123(fr_mac1, fr_mac2, fr_mac3) \
|
|
||||||
gte_mv_from_data_r(fr_mac1, C2_MAC1) \
|
|
||||||
, gte_mv_from_data_r(fr_mac2, C2_MAC2) \
|
|
||||||
, gte_mv_from_data_r(fr_mac3, C2_MAC3)
|
|
||||||
WORD_COUNT(mac_gte_mv_from_data_r_mac123, 3)
|
|
||||||
|
|
||||||
#define mac_gte_mv_from_mac123_v3s4(v) \
|
|
||||||
mac_gte_mv_from_data_r_mac123(v.x, v.y, v.z)
|
|
||||||
WORD_COUNT(mac_gte_mv_from_mac123_v3s4, 3)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \
|
#define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \
|
||||||
mac_load_word_imm(reg_transfer, cmd) \
|
load_upper_i(reg_transfer, cmd >> 16) \
|
||||||
, store_word( reg_transfer, reg_base, port)
|
, or_i_self( reg_transfer, cmd & 0xFFFF) \
|
||||||
|
, store_word( reg_transfer, reg_base, port)
|
||||||
WORD_COUNT(mac_gcmd_push, 3)
|
WORD_COUNT(mac_gcmd_push, 3)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
@@ -359,7 +149,6 @@ WORD_COUNT(mac_pack_color_word, 3)
|
|||||||
mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b)
|
mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b)
|
||||||
WORD_COUNT(mac_format_f3_color, 3)
|
WORD_COUNT(mac_format_f3_color, 3)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_format_g4_color(r_prim_cursor, r0, g0, b0, r1, g1, b1, r2, g2, b2, r3, g3, b3) \
|
#define mac_format_g4_color(r_prim_cursor, r0, g0, b0, r1, g1, b1, r2, g2, b2, r3, g3, b3) \
|
||||||
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0) \
|
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0) \
|
||||||
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1) \
|
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1) \
|
||||||
@@ -367,41 +156,29 @@ WORD_COUNT(mac_format_f3_color, 3)
|
|||||||
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3)
|
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3)
|
||||||
WORD_COUNT(mac_format_g4_color, 12)
|
WORD_COUNT(mac_format_g4_color, 12)
|
||||||
|
|
||||||
#define mac_insert_ot_tag(r_ot_base, r_prim_cursor, poly_size) \
|
#define mac_insert_ot_tag_f3(r_ot_base, r_prim_cursor) \
|
||||||
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
|
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
|
||||||
, add_u_self( R_T1, r_ot_base) /* T1 = & OrderingTable[OTZ] */ \
|
, add_u_self( R_T1, r_ot_base) /* T1 = & OrderingTable[OTZ] */ \
|
||||||
, load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \
|
, load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \
|
||||||
, load_upper_i(R_V0, (poly_size/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) \
|
, load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) /* V0 = (5 - 1) << 24 = 4 << 24 */ \
|
||||||
, mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
|
, mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
|
||||||
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
|
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
|
||||||
, store_word( R_AT, r_prim_cursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \
|
, store_word( R_AT, r_prim_cursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \
|
||||||
, shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
|
, shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
|
||||||
, shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \
|
, shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \
|
||||||
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
|
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
|
||||||
WORD_COUNT(mac_insert_ot_tag, 11)
|
WORD_COUNT(mac_insert_ot_tag_f3, 11)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
#define mac_insert_ot_tag_g4(r_ot_base, r_prim_cursor) \
|
||||||
#define mac_pad_set_centered_axes(state, scratch) \
|
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
|
||||||
load_upper_i(scratch, (PadAxis_Centered >> 16) & 0xFFFF) \
|
, add_u_self( R_T1, r_ot_base) /* T1 = & OrderingTable[OTZ] */ \
|
||||||
, or_i_self( scratch, PadAxis_Centered & 0xFFFF) /* mac_load_word_imm(scratch, PadAxis_Centered), */ \
|
, load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \
|
||||||
, store_word( scratch, state, O_(PadState,axes))
|
, load_upper_i(R_V0, (S_(Poly_G4)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) /* V0 = (9 - 1) << 24 = 8 << 24 */ \
|
||||||
WORD_COUNT(mac_pad_set_centered_axes, 3)
|
, mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
|
||||||
|
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
|
||||||
/* atom_dbg_skip */
|
, store_word( R_AT, r_prim_cursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \
|
||||||
#define mac_pad_set_id_byte(state, r_id, id_value) \
|
, shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
|
||||||
add_ui( r_id, R_0, id_value) \
|
, shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \
|
||||||
, store_byte(r_id, state, O_(PadState,id))
|
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
|
||||||
WORD_COUNT(mac_pad_set_id_byte, 2)
|
WORD_COUNT(mac_insert_ot_tag_g4, 11)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_pad_set_status(r_tmp, r_state, pad_status) \
|
|
||||||
add_ui( r_tmp, R_0, pad_status) \
|
|
||||||
, store_word(r_tmp, r_state, O_(PadState,status))
|
|
||||||
WORD_COUNT(mac_pad_set_status, 2)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_pad_store_inverted_buttons(r_buttons, r_pad_state) \
|
|
||||||
nor_u( r_buttons, r_buttons, R_0) \
|
|
||||||
, store_half(r_buttons, r_pad_state, O_(PadState,buttons))
|
|
||||||
WORD_COUNT(mac_pad_store_inverted_buttons, 2)
|
|
||||||
|
|
||||||
|
|||||||
+11
-31
@@ -11,10 +11,8 @@
|
|||||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h
|
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
|
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
|
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
|
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
|
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.h
|
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
|
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
|
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
|
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
|
||||||
@@ -25,35 +23,17 @@
|
|||||||
#pragma region duffle
|
#pragma region duffle
|
||||||
|
|
||||||
|
|
||||||
// --- atom: example_atom_proc (10 words) ---
|
// --- atom: pad_bios_snapshot (78 words) ---
|
||||||
|
|
||||||
#define _atom_offset_example_atom_proc_skip 2
|
#define _atom_offset_snap_root_skip_disconnected 8
|
||||||
|
#define _atom_offset_disconnected_snap_end 61
|
||||||
enum {
|
#define _atom_offset_case_2_id_dispatch 8
|
||||||
atom_offset_example_atom_proc_skip = _atom_offset_example_atom_proc_skip,
|
#define _atom_offset_pending_snap_end 51
|
||||||
};
|
#define _atom_offset_id_dispatch_try_analog_stick 11
|
||||||
|
#define _atom_offset_id_dispatch_snap_end 38
|
||||||
// --- atom: build_normalize_v3s4 (67 words) ---
|
#define _atom_offset_try_analog_stick_try_analog_pad 12
|
||||||
|
#define _atom_offset_analog_stick_snap_end 24
|
||||||
#define _atom_offset_aligned_done_srav_path 3
|
#define _atom_offset_try_analog_pad_try_unsupported 11
|
||||||
#define _atom_offset_srav_path_aligned_done 4
|
|
||||||
|
|
||||||
enum {
|
|
||||||
atom_offset_aligned_done_srav_path = _atom_offset_aligned_done_srav_path,
|
|
||||||
atom_offset_srav_path_aligned_done = _atom_offset_srav_path_aligned_done,
|
|
||||||
};
|
|
||||||
|
|
||||||
// --- atom: pad_bios_snapshot (84 words) ---
|
|
||||||
|
|
||||||
#define _atom_offset_snap_root_skip_disconnected 10
|
|
||||||
#define _atom_offset_disconnected_snap_end 65
|
|
||||||
#define _atom_offset_case_2_id_dispatch 9
|
|
||||||
#define _atom_offset_pending_snap_end 54
|
|
||||||
#define _atom_offset_id_dispatch_try_analog_stick 12
|
|
||||||
#define _atom_offset_id_dispatch_snap_end 40
|
|
||||||
#define _atom_offset_try_analog_stick_try_analog_pad 13
|
|
||||||
#define _atom_offset_analog_stick_snap_end 25
|
|
||||||
#define _atom_offset_try_analog_pad_try_unsupported 12
|
|
||||||
#define _atom_offset_analog_pad_snap_end 10
|
#define _atom_offset_analog_pad_snap_end 10
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
|
|||||||
+37
-16
@@ -8,48 +8,54 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(gp_atom_c);
|
|||||||
|
|
||||||
#pragma region MACs (Mips Atom Components)
|
#pragma region MACs (Mips Atom Components)
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_gcmd_push(AtomBuilder_R ab, U4 cmd, U4 reg_transfer, U4 reg_base, U2 port)
|
FI_ Slice_MipsCode ac_gcmd_push(U4 cmd, U4 reg_transfer, U4 reg_base, U2 port)
|
||||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
MipsAtomComp_Proc_(ac_gcmd_push, {
|
||||||
mac_load_word_imm(reg_transfer, cmd),
|
load_upper_i(reg_transfer, cmd >> 16),
|
||||||
store_word( reg_transfer, reg_base, port),
|
or_i_self( reg_transfer, cmd & 0xFFFF),
|
||||||
|
store_word( reg_transfer, reg_base, port),
|
||||||
})
|
})
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_store_rgb8(AtomBuilder_R ab, U1 rr, U1 rg, U1 rb, U4 base, U4 offset)
|
FI_ Slice_MipsCode ac_store_rgb8(U1 rr, U1 rg, U1 rb, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rgb8, {
|
||||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
store_byte(rr, base, offset + O_(RGB8,r)),
|
store_byte(rr, base, offset + O_(RGB8,r)),
|
||||||
store_byte(rg, base, offset + O_(RGB8,g)),
|
store_byte(rg, base, offset + O_(RGB8,g)),
|
||||||
store_byte(rb, base, offset + O_(RGB8,b)),
|
store_byte(rb, base, offset + O_(RGB8,b)),
|
||||||
})
|
})
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_pack_color_word(AtomBuilder_R ab, U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b)
|
/* Words: 3; Emits one (cmd|color) word to R_PrimCursor at the given
|
||||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
* byte offset. Internal helper used by the *_format_*_color macros. */
|
||||||
|
FI_ Slice_MipsCode ac_pack_color_word(U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, {
|
||||||
load_upper_i(R_AT, (cmd) << 8 | (b)),
|
load_upper_i(R_AT, (cmd) << 8 | (b)),
|
||||||
or_i_self( R_AT, ((g) << 8) | (r)),
|
or_i_self( R_AT, ((g) << 8) | (r)),
|
||||||
store_word( R_AT, r_base, (off)),
|
store_word( R_AT, r_base, (off)),
|
||||||
})
|
})
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_format_f3_color(AtomBuilder_R ab, U4 r_base, U1 r, U1 g, U1 b)
|
/* Words: 3; Emits the F3 command+color word (cmd byte | BLUE | GREEN | RED)
|
||||||
atom_dbg_skip MipsAtomComp_Proc_(ab, { mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
|
* Args: _r, _g, _b are 8-bit RGB byte values (not raw 16-bit fields). */
|
||||||
|
FI_ Slice_MipsCode ac_format_f3_color(U4 r_base, U1 r, U1 g, U1 b)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, { mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_format_g4_color(AtomBuilder_R ab, U4 r_prim_cursor,
|
/* Words: 12; Emits the four (code|color) words of a Poly_G4.
|
||||||
|
* Args: rN,gN,bN are 8-bit RGB byte values for each of the 4 vertices. */
|
||||||
|
FI_ Slice_MipsCode ac_format_g4_color(U4 r_prim_cursor,
|
||||||
U1 r0, U1 g0, U1 b0,
|
U1 r0, U1 g0, U1 b0,
|
||||||
U1 r1, U1 g1, U1 b1,
|
U1 r1, U1 g1, U1 b1,
|
||||||
U1 r2, U1 g2, U1 b2,
|
U1 r2, U1 g2, U1 b2,
|
||||||
U1 r3, U1 g3, U1 b3)
|
U1 r3, U1 g3, U1 b3)
|
||||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
MipsAtomComp_Proc_(ac_format_g4_color, {
|
||||||
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0),
|
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0),
|
||||||
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1),
|
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1),
|
||||||
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c2), 0, r2,g2,b2),
|
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c2), 0, r2,g2,b2),
|
||||||
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3),
|
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3),
|
||||||
})
|
})
|
||||||
|
|
||||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list. */
|
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
|
||||||
// TODO(Ed): Expose R_T1 as a r_t0, r_V0 as r_t2
|
* Hardcoded for Poly_F3 (5 words). For Poly_G4, use ac_insert_ot_tag_g4. */
|
||||||
I_ Slice_MipsCode ac_insert_ot_tag(AtomBuilder_R ab, Reg r_ot_base, Reg r_prim_cursor, U2 poly_size) MipsAtomComp_Proc_(ab, {
|
I_ Slice_MipsCode ac_insert_ot_tag_f3(U4 r_ot_base, U4 r_prim_cursor) MipsAtomComp_Proc_(ac_insert_ot_tag_f3, {
|
||||||
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
|
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
|
||||||
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
|
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
|
||||||
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
|
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
|
||||||
load_upper_i(R_V0, (poly_size/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits),
|
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits), // V0 = (5 - 1) << 24 = 4 << 24
|
||||||
mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
|
mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
|
||||||
or_u( R_AT, R_AT, R_V0), // Merge length
|
or_u( R_AT, R_AT, R_V0), // Merge length
|
||||||
store_word( R_AT, r_prim_cursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr)
|
store_word( R_AT, r_prim_cursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr)
|
||||||
@@ -58,4 +64,19 @@ I_ Slice_MipsCode ac_insert_ot_tag(AtomBuilder_R ab, Reg r_ot_base, Reg r_prim_c
|
|||||||
store_word( R_AT, R_T1, O_(PolyTag,code)), // OrderingTable[OTZ] = PrimCursor
|
store_word( R_AT, R_T1, O_(PolyTag,code)), // OrderingTable[OTZ] = PrimCursor
|
||||||
})
|
})
|
||||||
|
|
||||||
|
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
|
||||||
|
* Hardcoded for Poly_G4 (9 words). For Poly_F3, use ac_insert_ot_tag_f3. */
|
||||||
|
I_ Slice_MipsCode ac_insert_ot_tag_g4(U4 r_ot_base, U4 r_prim_cursor) MipsAtomComp_Proc_(ac_insert_ot_tag_g4, {
|
||||||
|
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
|
||||||
|
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
|
||||||
|
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
|
||||||
|
load_upper_i(R_V0, (S_(Poly_G4)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits), // V0 = (9 - 1) << 24 = 8 << 24
|
||||||
|
mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
|
||||||
|
or_u( R_AT, R_AT, R_V0), // Merge length
|
||||||
|
store_word( R_AT, r_prim_cursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr)
|
||||||
|
shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)), // AT = (prim_length << 24) | old_addr
|
||||||
|
shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
|
||||||
|
store_word( R_AT, R_T1, O_(PolyTag,code)), // OrderingTable[OTZ] = PrimCursor
|
||||||
|
})
|
||||||
|
|
||||||
#pragma endregion MACs (Mips Atom Components)
|
#pragma endregion MACs (Mips Atom Components)
|
||||||
|
|||||||
+61
-58
@@ -21,7 +21,7 @@
|
|||||||
* 4. Semantic encoders gp0_word_poly_f3(r,g,b)
|
* 4. Semantic encoders gp0_word_poly_f3(r,g,b)
|
||||||
* 3. Composite encoders enc_color_word(cmd, r, g, b)
|
* 3. Composite encoders enc_color_word(cmd, r, g, b)
|
||||||
* 2. Per-field encoders enc_gp0_color_r(r), enc_gp0_color_g(g), ...
|
* 2. Per-field encoders enc_gp0_color_r(r), enc_gp0_color_g(g), ...
|
||||||
* 1. Bitfield layout consts gp0_color_red_shift = 0, gp0_color_red_width = 8
|
* 1. Bitfield layout consts gp0_color_red_shift = 0, gp0_color_red_mask = 0xFF
|
||||||
* 0. Opcode IDs gp0_cmd_poly_f3 = 0x20
|
* 0. Opcode IDs gp0_cmd_poly_f3 = 0x20
|
||||||
*
|
*
|
||||||
* Vendor mnemonics (gte_mtc2, gte_mfc2, etc.) are NOT in this header.
|
* Vendor mnemonics (gte_mtc2, gte_mfc2, etc.) are NOT in this header.
|
||||||
@@ -68,14 +68,13 @@ enum {
|
|||||||
|
|
||||||
#define gp0_send(word) (HW_GP0[0] = (word))
|
#define gp0_send(word) (HW_GP0[0] = (word))
|
||||||
#define gp1_send(word) (HW_GP1[0] = (word))
|
#define gp1_send(word) (HW_GP1[0] = (word))
|
||||||
#define DmaSlot_ // Annotate an instruction as filling a CPU <-> Command DMA delay slot/s
|
|
||||||
|
|
||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
* GP0 command byte constants + Layer 1 (GPU bitfield shifts)
|
* GP0 command byte constants + Layer 1 (GPU bitfield shifts)
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
* 8-bit GP0 opcodes (the upper byte of a primitive's first word). These are the BYTE only.
|
* 8-bit GP0 opcodes (the upper byte of a primitive's first word). These are the BYTE only.
|
||||||
* NO macro body past this point uses a raw shift or raw mask.
|
* NO macro body past this point uses a raw shift or raw mask.
|
||||||
* Mirrors the OPCODE_SHIFT / RS_SHIFT convention from mips.h.
|
* Mirrors the OPCODE_SHIFT / RS_SHIFT / REG_MASK convention from mips.h.
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
enum {
|
enum {
|
||||||
gp0_cmd_Nop = 0x00,
|
gp0_cmd_Nop = 0x00,
|
||||||
@@ -117,20 +116,21 @@ enum {
|
|||||||
gp0_cmd_SetDrawOffset = 0xE5,
|
gp0_cmd_SetDrawOffset = 0xE5,
|
||||||
gp0_cmd_SetMaskBit = 0xE6,
|
gp0_cmd_SetMaskBit = 0xE6,
|
||||||
|
|
||||||
/* bitfield shifts / widths ----
|
/* bitfield shifts / widths / masks ----
|
||||||
* Generic GP0/GP1 command byte (upper 8 bits of every word sent to either port). */
|
* Generic GP0/GP1 command byte (upper 8 bits of every word sent to either port). */
|
||||||
gp0_cmd_shift = 24,
|
gp0_cmd_shift = 24,
|
||||||
gp0_cmd_width = 8,
|
gp0_cmd_width = 8,
|
||||||
|
gp0_cmd_mask = 0xFF,
|
||||||
|
|
||||||
/* Color word layout (lives in Poly_F3.color, Poly_G4.c0..c3, etc.):
|
/* Color word layout (lives in Poly_F3.color, Poly_G4.c0..c3, etc.):
|
||||||
* bits 31..24 = command byte
|
* bits 31..24 = command byte
|
||||||
* bits 23..16 = BLUE
|
* bits 23..16 = BLUE
|
||||||
* bits 15..08 = GREEN
|
* bits 15..08 = GREEN
|
||||||
* bits 07..00 = RED (PSX GPU is BGR, NOT RGB) */
|
* bits 07..00 = RED (PSX GPU is BGR, NOT RGB) */
|
||||||
gp0_color_cmd_shift = 24, gp0_color_cmd_width = 8,
|
gp0_color_cmd_shift = 24, gp0_color_cmd_width = 8, gp0_color_cmd_mask = 0xFF,
|
||||||
gp0_color_blue_shift = 16, gp0_color_blue_width = 8,
|
gp0_color_blue_shift = 16, gp0_color_blue_width = 8, gp0_color_blue_mask = 0xFF,
|
||||||
gp0_color_green_shift = 8, gp0_color_green_width = 8,
|
gp0_color_green_shift = 8, gp0_color_green_width = 8, gp0_color_green_mask = 0xFF,
|
||||||
gp0_color_red_shift = 0, gp0_color_red_width = 8,
|
gp0_color_red_shift = 0, gp0_color_red_width = 8, gp0_color_red_mask = 0xFF,
|
||||||
};
|
};
|
||||||
|
|
||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
@@ -143,12 +143,12 @@ enum {
|
|||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
|
|
||||||
/* ---- Layer 1.5: per-field encoders ---- */
|
/* ---- Layer 1.5: per-field encoders ---- */
|
||||||
#define enc_gp0_cmd(cmd) ((cmd) << gp0_cmd_shift)
|
#define enc_gp0_cmd(cmd) (((cmd) & gp0_cmd_mask) << gp0_cmd_shift)
|
||||||
|
|
||||||
#define enc_gp0_color_cmd(cmd) ((cmd) << gp0_color_cmd_shift)
|
#define enc_gp0_color_cmd(cmd) (((cmd) & gp0_color_cmd_mask) << gp0_color_cmd_shift)
|
||||||
#define enc_gp0_color_r(r) ((r) << gp0_color_red_shift)
|
#define enc_gp0_color_r(r) (((r) & gp0_color_red_mask) << gp0_color_red_shift)
|
||||||
#define enc_gp0_color_g(g) ((g) << gp0_color_green_shift)
|
#define enc_gp0_color_g(g) (((g) & gp0_color_green_mask) << gp0_color_green_shift)
|
||||||
#define enc_gp0_color_b(b) ((b) << gp0_color_blue_shift)
|
#define enc_gp0_color_b(b) (((b) & gp0_color_blue_mask) << gp0_color_blue_shift)
|
||||||
|
|
||||||
/* ---- Layer 2: composite encoders ---- */
|
/* ---- Layer 2: composite encoders ---- */
|
||||||
#define enc_color_word(cmd, r, g, b) (enc_gp0_color_cmd(cmd) | enc_gp0_color_r(r) | enc_gp0_color_g(g) | enc_gp0_color_b(b))
|
#define enc_color_word(cmd, r, g, b) (enc_gp0_color_cmd(cmd) | enc_gp0_color_r(r) | enc_gp0_color_g(g) | enc_gp0_color_b(b))
|
||||||
@@ -211,38 +211,38 @@ enum {
|
|||||||
gp1_disp_Color24 = 0x1,
|
gp1_disp_Color24 = 0x1,
|
||||||
gp1_disp_VInterlace = 0x1,
|
gp1_disp_VInterlace = 0x1,
|
||||||
|
|
||||||
/* ---- Layer 1: GP1 display-mode + range + draw-area shifts/widths ---- */
|
/* ---- Layer 1: GP1 display-mode + range + draw-area shifts/masks ---- */
|
||||||
gp1_disp_hres_shift = 0, gp1_disp_hres_width = 2,
|
gp1_disp_hres_shift = 0, gp1_disp_hres_width = 2, gp1_disp_hres_mask = 0x3,
|
||||||
gp1_disp_vres_shift = 2, gp1_disp_vres_width = 1,
|
gp1_disp_vres_shift = 2, gp1_disp_vres_width = 1, gp1_disp_vres_mask = 0x1,
|
||||||
gp1_disp_color_shift = 4, gp1_disp_color_width = 1,
|
gp1_disp_color_shift = 4, gp1_disp_color_width = 1, gp1_disp_color_mask = 0x1,
|
||||||
gp1_disp_interlace_shift = 5, gp1_disp_interlace_width = 1,
|
gp1_disp_interlace_shift = 5, gp1_disp_interlace_width = 1, gp1_disp_interlace_mask = 0x1,
|
||||||
|
|
||||||
/* GP1 horizontal display range: bits 0..11 = X2, bits 12..23 = X1 */
|
/* GP1 horizontal display range: bits 0..11 = X2, bits 12..23 = X1 */
|
||||||
gp1_hrange_x1_shift = 12, gp1_hrange_x1_width = 12,
|
gp1_hrange_x1_shift = 12, gp1_hrange_x1_width = 12, gp1_hrange_x1_mask = 0xFFF,
|
||||||
gp1_hrange_x2_shift = 0, gp1_hrange_x2_width = 12,
|
gp1_hrange_x2_shift = 0, gp1_hrange_x2_width = 12, gp1_hrange_x2_mask = 0xFFF,
|
||||||
|
|
||||||
/* GP1 vertical display range: bits 0..9 = Y2, bits 10..19 = Y1 */
|
/* GP1 vertical display range: bits 0..9 = Y2, bits 10..19 = Y1 */
|
||||||
gp1_vrange_y1_shift = 10, gp1_vrange_y1_width = 10,
|
gp1_vrange_y1_shift = 10, gp1_vrange_y1_width = 10, gp1_vrange_y1_mask = 0x3FF,
|
||||||
gp1_vrange_y2_shift = 0, gp1_vrange_y2_width = 10,
|
gp1_vrange_y2_shift = 0, gp1_vrange_y2_width = 10, gp1_vrange_y2_mask = 0x3FF,
|
||||||
|
|
||||||
/* GP1 draw area (top-left or bottom-right): bits 0..9 = X, bits 10..19 = Y
|
/* GP1 draw area (top-left or bottom-right): bits 0..9 = X, bits 10..19 = Y
|
||||||
* (10-bit signed — caller pre-signs) */
|
* (10-bit signed — caller pre-signs and masks with the named mask) */
|
||||||
gp1_draw_x_shift = 0, gp1_draw_x_width = 10,
|
gp1_draw_x_shift = 0, gp1_draw_x_width = 10, gp1_draw_x_mask = 0x3FF,
|
||||||
gp1_draw_y_shift = 10, gp1_draw_y_width = 10,
|
gp1_draw_y_shift = 10, gp1_draw_y_width = 10, gp1_draw_y_mask = 0x3FF,
|
||||||
};
|
};
|
||||||
|
|
||||||
/* ---- Layer 1.5: GP1 per-field encoders ---- */
|
/* ---- Layer 1.5: GP1 per-field encoders ---- */
|
||||||
#define enc_gp1_disp_hres(h) ((h) << gp1_disp_hres_shift)
|
#define enc_gp1_disp_hres(h) (((h) & gp1_disp_hres_mask) << gp1_disp_hres_shift)
|
||||||
#define enc_gp1_disp_vres(v) ((v) << gp1_disp_vres_shift)
|
#define enc_gp1_disp_vres(v) (((v) & gp1_disp_vres_mask) << gp1_disp_vres_shift)
|
||||||
#define enc_gp1_disp_color(c) ((c) << gp1_disp_color_shift)
|
#define enc_gp1_disp_color(c) (((c) & gp1_disp_color_mask) << gp1_disp_color_shift)
|
||||||
#define enc_gp1_disp_interlace(i) ((i) << gp1_disp_interlace_shift)
|
#define enc_gp1_disp_interlace(i) (((i) & gp1_disp_interlace_mask) << gp1_disp_interlace_shift)
|
||||||
|
|
||||||
#define enc_gp1_hrange_x1(x1) ((x1) << gp1_hrange_x1_shift)
|
#define enc_gp1_hrange_x1(x1) (((x1) & gp1_hrange_x1_mask) << gp1_hrange_x1_shift)
|
||||||
#define enc_gp1_hrange_x2(x2) ((x2) << gp1_hrange_x2_shift)
|
#define enc_gp1_hrange_x2(x2) (((x2) & gp1_hrange_x2_mask) << gp1_hrange_x2_shift)
|
||||||
#define enc_gp1_vrange_y1(y1) ((y1) << gp1_vrange_y1_shift)
|
#define enc_gp1_vrange_y1(y1) (((y1) & gp1_vrange_y1_mask) << gp1_vrange_y1_shift)
|
||||||
#define enc_gp1_vrange_y2(y2) ((y2) << gp1_vrange_y2_shift)
|
#define enc_gp1_vrange_y2(y2) (((y2) & gp1_vrange_y2_mask) << gp1_vrange_y2_shift)
|
||||||
#define enc_gp1_draw_x(x) ((x) << gp1_draw_x_shift)
|
#define enc_gp1_draw_x(x) (((x) & gp1_draw_x_mask) << gp1_draw_x_shift)
|
||||||
#define enc_gp1_draw_y(y) ((y) << gp1_draw_y_shift)
|
#define enc_gp1_draw_y(y) (((y) & gp1_draw_y_mask) << gp1_draw_y_shift)
|
||||||
|
|
||||||
/* ---- Layer 2: GP1 composite encoders ---- */
|
/* ---- Layer 2: GP1 composite encoders ---- */
|
||||||
#define enc_gp1_disp_mode_word(h, v, c, i) (enc_gp0_cmd(gp1_cmd_DisplayMode) | enc_gp1_disp_hres(h) | enc_gp1_disp_vres(v) | enc_gp1_disp_color(c) | enc_gp1_disp_interlace(i))
|
#define enc_gp1_disp_mode_word(h, v, c, i) (enc_gp0_cmd(gp1_cmd_DisplayMode) | enc_gp1_disp_hres(h) | enc_gp1_disp_vres(v) | enc_gp1_disp_color(c) | enc_gp1_disp_interlace(i))
|
||||||
@@ -419,11 +419,14 @@ typedef Struct_(PolyTag) {
|
|||||||
};
|
};
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/* DSL cast convention: every cast uses `C_()`, every pointer qualifier is `R_` (restrict) or `V_` (volatile).
|
||||||
|
* No raw C-style casts. RHS values are assumed to be `U4` — caller passes a `U4` directly. */
|
||||||
#define set_len(tag,v) (C_(PolyTag_R,tag)->len = u4_(v))
|
#define set_len(tag,v) (C_(PolyTag_R,tag)->len = u4_(v))
|
||||||
#define set_addr(tag,v) (C_(PolyTag_R,tag)->addr = u4_(v))
|
#define set_addr(tag,v) (C_(PolyTag_R,tag)->addr = u4_(v))
|
||||||
/* `set_code` is no longer in the new PolyTag design
|
/* `set_code` is no longer in the new PolyTag design — the code byte lives in the primitive body
|
||||||
* (e.g. `((Poly_F3*)(p))->code`), not in the tag.
|
* (e.g. `((Poly_F3*)(p))->code`), not in the tag.
|
||||||
* Use the typed primitive structs (Poly_F3, Poly_G4, etc.) and the `set_poly_*` setters, which set both the tag's length and the code. */
|
* Use the typed primitive structs (Poly_F3, Poly_G4, etc.) and the `set_poly_*` setters,
|
||||||
|
* which set both the tag's length and the code. */
|
||||||
#define get_len(tag) C_(U4,C_(PolyTag_R,tag)->len)
|
#define get_len(tag) C_(U4,C_(PolyTag_R,tag)->len)
|
||||||
#define get_addr(tag) C_(U4,C_(PolyTag_R,tag)->addr)
|
#define get_addr(tag) C_(U4,C_(PolyTag_R,tag)->addr)
|
||||||
|
|
||||||
@@ -552,14 +555,14 @@ typedef Struct_(Poly_GT4) {
|
|||||||
* bits 12..31 = reserved (zero)
|
* bits 12..31 = reserved (zero)
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
enum {
|
enum {
|
||||||
/* ---- Layer 1: TPage bitfield shifts / widths ---- */
|
/* ---- Layer 1: TPage bitfield shifts / widths / masks ---- */
|
||||||
gp0_tpage_x_shift = 0, gp0_tpage_x_width = 4,
|
gp0_tpage_x_shift = 0, gp0_tpage_x_width = 4, gp0_tpage_x_mask = 0xF,
|
||||||
gp0_tpage_y_shift = 4, gp0_tpage_y_width = 1,
|
gp0_tpage_y_shift = 4, gp0_tpage_y_width = 1, gp0_tpage_y_mask = 0x1,
|
||||||
gp0_tpage_semi_trans_shift = 5, gp0_tpage_semi_trans_width = 2,
|
gp0_tpage_semi_trans_shift = 5, gp0_tpage_semi_trans_width = 2, gp0_tpage_semi_trans_mask = 0x3,
|
||||||
gp0_tpage_color_depth_shift = 7, gp0_tpage_color_depth_width = 2,
|
gp0_tpage_color_depth_shift = 7, gp0_tpage_color_depth_width = 2, gp0_tpage_color_depth_mask = 0x3,
|
||||||
gp0_tpage_dither_shift = 9, gp0_tpage_dither_width = 1,
|
gp0_tpage_dither_shift = 9, gp0_tpage_dither_width = 1, gp0_tpage_dither_mask = 0x1,
|
||||||
gp0_tpage_draw_to_disp_shift = 10, gp0_tpage_draw_to_disp_width = 1,
|
gp0_tpage_draw_to_disp_shift = 10, gp0_tpage_draw_to_disp_width = 1, gp0_tpage_draw_to_disp_mask = 0x1,
|
||||||
gp0_tpage_tex_disable_shift = 11, gp0_tpage_tex_disable_width = 1,
|
gp0_tpage_tex_disable_shift = 11, gp0_tpage_tex_disable_width = 1, gp0_tpage_tex_disable_mask = 0x1,
|
||||||
|
|
||||||
/* TPage color-depth payload values (NOT bit positions — these go in
|
/* TPage color-depth payload values (NOT bit positions — these go in
|
||||||
* the 2-bit field at gp0_tpage_color_depth_shift). */
|
* the 2-bit field at gp0_tpage_color_depth_shift). */
|
||||||
@@ -570,7 +573,7 @@ enum {
|
|||||||
/* Default TPage value libpsyx's SetDefDrawEnv writes (matches the `li v1, 10; sh v1, 20(v0)` sequence at C11_only.elf:0x8001273C). */
|
/* Default TPage value libpsyx's SetDefDrawEnv writes (matches the `li v1, 10; sh v1, 20(v0)` sequence at C11_only.elf:0x8001273C). */
|
||||||
gp0_tpage_default = 10,
|
gp0_tpage_default = 10,
|
||||||
|
|
||||||
/* TPage semi-transparency mode payload values. */
|
/* TPage semi-transparency mode payload values (NOT bit positions). */
|
||||||
gp0_tpage_semi_trans_none = 0x0,
|
gp0_tpage_semi_trans_none = 0x0,
|
||||||
gp0_tpage_semi_trans_alpha = 0x1,
|
gp0_tpage_semi_trans_alpha = 0x1,
|
||||||
gp0_tpage_semi_trans_add = 0x2,
|
gp0_tpage_semi_trans_add = 0x2,
|
||||||
@@ -578,13 +581,13 @@ enum {
|
|||||||
};
|
};
|
||||||
|
|
||||||
/* ---- Layer 1.5: TPage per-field encoders. Mirrors enc_gte_sf/mx/v in gte.h. ---- */
|
/* ---- Layer 1.5: TPage per-field encoders. Mirrors enc_gte_sf/mx/v in gte.h. ---- */
|
||||||
#define enc_gp0_tpage_x(x) ((x) << gp0_tpage_x_shift)
|
#define enc_gp0_tpage_x(x) (((x) & gp0_tpage_x_mask) << gp0_tpage_x_shift)
|
||||||
#define enc_gp0_tpage_y(y) ((y) << gp0_tpage_y_shift)
|
#define enc_gp0_tpage_y(y) (((y) & gp0_tpage_y_mask) << gp0_tpage_y_shift)
|
||||||
#define enc_gp0_tpage_semi_trans(s) ((s) << gp0_tpage_semi_trans_shift)
|
#define enc_gp0_tpage_semi_trans(s) (((s) & gp0_tpage_semi_trans_mask) << gp0_tpage_semi_trans_shift)
|
||||||
#define enc_gp0_tpage_color_depth(c) ((c) << gp0_tpage_color_depth_shift)
|
#define enc_gp0_tpage_color_depth(c) (((c) & gp0_tpage_color_depth_mask) << gp0_tpage_color_depth_shift)
|
||||||
#define enc_gp0_tpage_dither(d) ((d) << gp0_tpage_dither_shift)
|
#define enc_gp0_tpage_dither(d) (((d) & gp0_tpage_dither_mask) << gp0_tpage_dither_shift)
|
||||||
#define enc_gp0_tpage_draw_to_disp(d) ((d) << gp0_tpage_draw_to_disp_shift)
|
#define enc_gp0_tpage_draw_to_disp(d) (((d) & gp0_tpage_draw_to_disp_mask) << gp0_tpage_draw_to_disp_shift)
|
||||||
#define enc_gp0_tpage_tex_disable(t) ((t) << gp0_tpage_tex_disable_shift)
|
#define enc_gp0_tpage_tex_disable(t) (((t) & gp0_tpage_tex_disable_mask) << gp0_tpage_tex_disable_shift)
|
||||||
|
|
||||||
/* ---- Layer 2: TPage composite encoder. Mirrors enc_gte_cmdw in gte.h ---- */
|
/* ---- Layer 2: TPage composite encoder. Mirrors enc_gte_cmdw in gte.h ---- */
|
||||||
#define enc_gp0_tpage_word(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
|
#define enc_gp0_tpage_word(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
|
||||||
@@ -614,17 +617,17 @@ typedef Struct_(TexturePage) { U4 raw; };
|
|||||||
* bits 24..31 = command byte — 0x20 (4bpp load) or 0x25 (8bpp load)
|
* bits 24..31 = command byte — 0x20 (4bpp load) or 0x25 (8bpp load)
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
enum {
|
enum {
|
||||||
/* ---- Layer 1: CLUT bitfield shifts / widths ---- */
|
/* ---- Layer 1: CLUT bitfield shifts / widths / masks ---- */
|
||||||
gp0_clut_y_shift = 0, gp0_clut_y_width = 6,
|
gp0_clut_y_shift = 0, gp0_clut_y_width = 6, gp0_clut_y_mask = 0x3F,
|
||||||
gp0_clut_x_shift = 6, gp0_clut_x_width = 9,
|
gp0_clut_x_shift = 6, gp0_clut_x_width = 9, gp0_clut_x_mask = 0x1FF,
|
||||||
/* CLUT-load cmd-byte variants — the upper byte of the GP0 word. */
|
/* CLUT-load cmd-byte variants — the upper byte of the GP0 word. */
|
||||||
gp0_clut_cmd_Load4bpp = 0x20,
|
gp0_clut_cmd_Load4bpp = 0x20,
|
||||||
gp0_clut_cmd_Load8bpp = 0x25,
|
gp0_clut_cmd_Load8bpp = 0x25,
|
||||||
};
|
};
|
||||||
|
|
||||||
/* ---- Layer 1.5: CLUT per-field encoders ---- */
|
/* ---- Layer 1.5: CLUT per-field encoders ---- */
|
||||||
#define enc_gp0_clut_x(x) ((x) << gp0_clut_x_shift)
|
#define enc_gp0_clut_x(x) (((x) & gp0_clut_x_mask) << gp0_clut_x_shift)
|
||||||
#define enc_gp0_clut_y(y) ((y) << gp0_clut_y_shift)
|
#define enc_gp0_clut_y(y) (((y) & gp0_clut_y_mask) << gp0_clut_y_shift)
|
||||||
|
|
||||||
/* ---- Layer 2: CLUT composite encoder ---- */
|
/* ---- Layer 2: CLUT composite encoder ---- */
|
||||||
#define enc_gp0_clut_word(cmd, x, y) (enc_gp0_cmd(cmd) | enc_gp0_clut_x(x) | enc_gp0_clut_y(y))
|
#define enc_gp0_clut_word(cmd, x, y) (enc_gp0_cmd(cmd) | enc_gp0_clut_x(x) | enc_gp0_clut_y(y))
|
||||||
|
|||||||
+21
-363
@@ -11,62 +11,25 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(gte_atom_c);
|
|||||||
#pragma region MACs (Mips Atom Components)
|
#pragma region MACs (Mips Atom Components)
|
||||||
|
|
||||||
/* Words: 3; Loads 3 S2 indices from the face array */
|
/* Words: 3; Loads 3 S2 indices from the face array */
|
||||||
FI_ Slice_MipsCode ac_load_tri_indices(AtomBuilder_R ab, U4 r_face_cusor, U4 r_i0, U4 r_i1, U4 r_i2)
|
FI_ Slice_MipsCode ac_load_tri_indices(U4 r_face_cusor, U4 r_i0, U4 r_i1, U4 r_i2) atom_dbg_skip MipsAtomComp_Proc_(ac_load_tri_indices, {
|
||||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)),
|
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)),
|
||||||
load_half_u(r_i1, r_face_cusor, 1 * S_(S2)),
|
load_half_u(r_i1, r_face_cusor, 1 * S_(S2)),
|
||||||
load_half_u(r_i2, r_face_cusor, 2 * S_(S2)),
|
load_half_u(r_i2, r_face_cusor, 2 * S_(S2)),
|
||||||
})
|
})
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_gte_mv_to_cr_diag_v3s4(AtomBuilder_R ab, Reg_(V3_S4) v) MipsAtomComp_Proc_(ab, {
|
|
||||||
gte_mv_to_ctrl_r(v.y, gte_cr_RT13),
|
|
||||||
gte_mv_to_ctrl_r(v.z, gte_cr_RT22),
|
|
||||||
gte_mv_to_ctrl_r(v.x, gte_cr_RT11),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_gte_ld_ir123_v3s4(AtomBuilder_R ab, Reg_(V3_S4) v) MipsAtomComp_Proc_(ab, {
|
|
||||||
gte_mv_to_data_r(v.x, C2_IR1),
|
|
||||||
gte_mv_to_data_r(v.y, C2_IR2),
|
|
||||||
gte_mv_to_data_r(v.z, C2_IR3),
|
|
||||||
})
|
|
||||||
|
|
||||||
/* ─── GTE OP cross product (a × b → a) ───
|
|
||||||
* Sets up RT diagonal from a.xyz, IR1/2/3 from b.xyz, fires OP,
|
|
||||||
* reads MAC1/2/3, shifts right 12 (S12.20 → S12.0 OuterProduct12), writes back to a.xyz.
|
|
||||||
* Composes the three sub-primitives (RT-load, IR-load, OP, MAC-read, shift)
|
|
||||||
* into one component for use by atoms that need the cross product inline.
|
|
||||||
*
|
|
||||||
* Output gpr (a) aliases source-A gpr; MAC read clobbers source-A's load targets,
|
|
||||||
* but by that point the RT load is complete and source A is dead.
|
|
||||||
* Pipeline: clobbers IR1..3, MAC1..3, RT11..33.
|
|
||||||
*
|
|
||||||
* The CPU→COP2 transfer chains (3 ctc2, 3 mtc2) require a 2-slot retirement gap,
|
|
||||||
* and the MFC2→GPR chain (3 mfc2) requires a 1-slot retirement gap, before the GPR can be read.
|
|
||||||
* The hazard nops are inlined below — same convention as ac_gte_gpf_scale — so any atom body inlining this component inherits them.
|
|
||||||
*
|
|
||||||
* Words: 18 (3 ctc2 + 2 nop + 3 mtc2 + 2 nop + 1 op + 3 mfc2 + 1 nop + 3 sra).
|
|
||||||
*/
|
|
||||||
FI_ Slice_MipsCode ac_gte_op_cross_v3s4(AtomBuilder_R ab, Reg_(V3_S4) a, Reg_(V3_S4) b) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
mac_gte_mv_to_cr_diag_v3s4(a), GteDelay_ /* RT diagonal: D1 = a.x, D2 = a.y, D3 = a.z */
|
|
||||||
mac_gte_ld_ir123_v3s4(b), GteDelay_ /* IR: second operand (b.xyz) */
|
|
||||||
gte_cmdw_cross, /* OP: MAC1/2/3 = a × b (S12.20) */
|
|
||||||
mac_gte_mv_from_mac123_v3s4(a), GteDelay_ /* Read MAC1/2/3 → a.xyz (overwrites source-A's load targets) */
|
|
||||||
mac_shift_aright_v3s4_self(a, 12), /* Right-shift MAC by 12 (S12.20 → S12.0 OuterProduct12) */
|
|
||||||
})
|
|
||||||
|
|
||||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
|
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
|
||||||
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
|
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
|
||||||
FI_ Slice_MipsCode ac_gte_store_f3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
FI_ Slice_MipsCode ac_gte_store_f3(U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_f3, {
|
||||||
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)),
|
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)),
|
||||||
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_F3,p1)),
|
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_F3,p1)),
|
||||||
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_F3,p2)),
|
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_F3,p2)),
|
||||||
})
|
})
|
||||||
|
|
||||||
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
|
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
|
||||||
I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
I_ Slice_MipsCode ac_gte_load_tri_verts(U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_load_tri_verts, {
|
||||||
shift_lleft(R_AT, r_v0, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
shift_lleft(R_AT, r_v0, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||||
shift_lleft(R_AT, r_v1, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
shift_lleft(R_AT, r_v1, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
||||||
shift_lleft(R_AT, r_v2, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
shift_lleft(R_AT, r_v2, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
||||||
})
|
})
|
||||||
|
|
||||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
|
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
|
||||||
@@ -74,7 +37,7 @@ I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, U4 r_vert_base, U4 r_v
|
|||||||
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
|
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
|
||||||
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
|
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
|
||||||
* (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
|
* (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
|
||||||
FI_ Slice_MipsCode ac_gte_store_g4_p012(AtomBuilder_R ab, Reg r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
FI_ Slice_MipsCode ac_gte_store_g4_p012(U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p012, {
|
||||||
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)),
|
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)),
|
||||||
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)),
|
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)),
|
||||||
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2)),
|
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2)),
|
||||||
@@ -84,334 +47,29 @@ FI_ Slice_MipsCode ac_gte_store_g4_p012(AtomBuilder_R ab, Reg r_primitive_cursor
|
|||||||
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2;
|
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2;
|
||||||
* SXY0 still holds v0.screen from the earlier RTPT.
|
* SXY0 still holds v0.screen from the earlier RTPT.
|
||||||
*/
|
*/
|
||||||
FI_ Slice_MipsCode ac_gte_store_g4_p3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) })
|
FI_ Slice_MipsCode ac_gte_store_g4_p3(U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p3, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) })
|
||||||
|
|
||||||
/* ─── STAGE 1 of normalize: SQR + mfc2 MAC1/2/3 ───
|
|
||||||
* Emits squared magnitude per component (in MAC1/2/3) into caller-provided scratch regs. */
|
|
||||||
FI_ Slice_MipsCode ac_gte_sqr_v3(AtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_sq_x, U4 r_sq_y, U4 r_sq_z) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop),
|
|
||||||
gte_mv_from_data_r(r_sq_x, C2_MAC1),
|
|
||||||
gte_mv_from_data_r(r_sq_y, C2_MAC2),
|
|
||||||
gte_mv_from_data_r(r_sq_z, C2_MAC3),
|
|
||||||
})
|
|
||||||
|
|
||||||
/* ─── SQR FIRE — mtc2 3 GPRs into IR1/IR2/IR3, then fire SQR. ─── */
|
|
||||||
FI_ Slice_MipsCode ac_gte_sqr_v3s4(AtomBuilder_R ab, Reg r_sx, Reg r_sy, Reg r_sz, MipsCode delay_slot)
|
|
||||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
gte_mv_to_data_r(r_sx, C2_IR1),
|
|
||||||
gte_mv_to_data_r(r_sy, C2_IR2),
|
|
||||||
gte_mv_to_data_r(r_sz, C2_IR3),
|
|
||||||
delay_slot, gte_cmdw_sqr,
|
|
||||||
})
|
|
||||||
|
|
||||||
/* ─── STAGE 4 of normalize: mtc2 IR0..3 + GPF + mfc2 MAC + srav finalize ───
|
|
||||||
* Reusable standalone — given an IR0 = 1/|v| estimate (typically from a sqrtbl lookup) and a shift count
|
|
||||||
* (typically (31 - LZCR)/2), multiplies IR0*IR[i] via GPF and shifts right to produce the normalized output.
|
|
||||||
* Used standalone for "scale vector by scalar".
|
|
||||||
* Words: 11. Clobbers: IR0..3, MAC1..3. Uses gte_cmdw_gpf (sf=0, lm=0). */
|
|
||||||
FI_ Slice_MipsCode ac_gte_gpf_scale(AtomBuilder_R ab,
|
|
||||||
U4 r_sx, U4 r_sy, U4 r_sz,
|
|
||||||
U4 r_recip_est, U4 r_shift,
|
|
||||||
U4 r_dx, U4 r_dy, U4 r_dz)
|
|
||||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
gte_mv_to_data_r(r_recip_est, C2_IR0),
|
|
||||||
gte_mv_to_data_r(r_sx, C2_IR1),
|
|
||||||
gte_mv_to_data_r(r_sy, C2_IR2),
|
|
||||||
gte_mv_to_data_r(r_sz, C2_IR3),
|
|
||||||
GteDelay_ nop2, /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */
|
|
||||||
gte_cmdw_gpf,
|
|
||||||
gte_mv_from_data_r(r_dx, C2_MAC1),
|
|
||||||
gte_mv_from_data_r(r_dy, C2_MAC2),
|
|
||||||
gte_mv_from_data_r(r_dz, C2_MAC3),
|
|
||||||
shift_aright_var(r_dx, r_dx, r_shift),
|
|
||||||
shift_aright_var(r_dy, r_dy, r_shift),
|
|
||||||
shift_aright_var(r_dz, r_dz, r_shift),
|
|
||||||
})
|
|
||||||
|
|
||||||
/* ─── TRANS MATRIX (libgte TransMatrix port) ───
|
|
||||||
* Atom component — auto-generates mac_trans_matrix Mac composer macro.
|
|
||||||
* m->t = v (struct copy; libgte's TransMatrix at 0x8001a540 is just 3 store_words, no GTE, no add).
|
|
||||||
* Uses 1 GPR (r_t1 = off value) per axis; per-axis load-delay-slot pattern.
|
|
||||||
* Words: 9. Clobbers: r_t1. */
|
|
||||||
FI_ Slice_MipsCode ac_trans_mt3s3s4(AtomBuilder_R ab
|
|
||||||
, U4 r_mtx, U4 r_off
|
|
||||||
, U4 r_t0, U4 r_t1, U4 r_t2
|
|
||||||
) MipsAtomComp_Proc_(ab, {
|
|
||||||
load_word( r_t0, r_off, O_(V3_S4,x)),
|
|
||||||
load_word( r_t1, r_off, O_(V3_S4,y)),
|
|
||||||
load_word( r_t2, r_off, O_(V3_S4,z)),
|
|
||||||
store_word(r_t0, r_mtx, O_(MT3_S2S4,t[0])),
|
|
||||||
store_word(r_t1, r_mtx, O_(MT3_S2S4,t[1])),
|
|
||||||
store_word(r_t2, r_mtx, O_(MT3_S2S4,t[2])),
|
|
||||||
})
|
|
||||||
|
|
||||||
/* ─── LZCR ROUND EVEN + HALF-SHIFT ───
|
|
||||||
* Takes the raw LZCR leading-zero/ones count (from mfc2 C2_LZCR, range 1..32 per PSX-SPX cop2r31) and the |v|² sum (in r_mag_sq from the MAC1+MAC2+MAC3 add).
|
|
||||||
* Produces:
|
|
||||||
* r_shift ← LZCR rounded down to even (clear bit 0)
|
|
||||||
* r_mag_sq_copy ← |v|² sum (moved out of r_mag_sq before it's overwritten)
|
|
||||||
* r_mag_sq ← (31 - even_LZCR) / 2 = the final srav/GPF shift amount
|
|
||||||
*
|
|
||||||
* Rounding to even ensures (31 - LZCR) is always odd, so the >> 1 division is consistent — no 0.5 loss.
|
|
||||||
* The caller branches on LZCR < 24 to decide left-shift vs right-shift of r_mag_sq_copy, then saves the shift count.
|
|
||||||
*
|
|
||||||
* Note: C2_LZCR (cop2r31) is a fixed read-only C2 data register — the caller must read it via mfc2 from C2_LZCR;
|
|
||||||
* there is no register choice at the hardware level. Only the GPR that holds the result is caller-determined. */
|
|
||||||
FI_ Slice_MipsCode ac_lzcr_round_even_half_shift(AtomBuilder_R ab,
|
|
||||||
U4 r_shift,
|
|
||||||
U4 r_mag_sq,
|
|
||||||
U4 r_mag_sq_copy)
|
|
||||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
and_i(r_shift, r_shift, gte_lzcr_even_mask),
|
|
||||||
or_u(r_mag_sq_copy, r_mag_sq, 0),
|
|
||||||
li_s( r_mag_sq, 31),
|
|
||||||
sub_s( r_mag_sq, r_mag_sq, r_shift),
|
|
||||||
shift_aright(r_mag_sq, r_mag_sq, 1),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_gte_general_purpose_interopolation(AtomBuilder_R ab
|
|
||||||
, Reg to_ir0, Reg to_ir1, Reg to_ir2, Reg to_ir3
|
|
||||||
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3
|
|
||||||
, MipsCode nop_slot1, MipsCode nop_slot2)
|
|
||||||
MipsAtomComp_Proc_(ab, {
|
|
||||||
gte_mv_to_data_r(to_ir0, C2_IR0),
|
|
||||||
gte_mv_to_data_r(to_ir1, C2_IR1), /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */
|
|
||||||
gte_mv_to_data_r(to_ir2, C2_IR2),
|
|
||||||
gte_mv_to_data_r(to_ir3, C2_IR3), /* IR3 = src.z (reloaded) */
|
|
||||||
GteDelay_ nop_slot1,
|
|
||||||
GteDelay_ nop_slot2,
|
|
||||||
gte_cmdw_gpf,
|
|
||||||
gte_mv_from_data_r(fr_mac1, C2_MAC1),
|
|
||||||
gte_mv_from_data_r(fr_mac2, C2_MAC2),
|
|
||||||
gte_mv_from_data_r(fr_mac3, C2_MAC3),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_gte_mv_from_data_r_mac123(AtomBuilder_R ab
|
|
||||||
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3)
|
|
||||||
MipsAtomComp_Proc_(ab, {
|
|
||||||
gte_mv_from_data_r(fr_mac1, C2_MAC1),
|
|
||||||
gte_mv_from_data_r(fr_mac2, C2_MAC2),
|
|
||||||
gte_mv_from_data_r(fr_mac3, C2_MAC3),
|
|
||||||
})
|
|
||||||
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_gte_mv_from_mac123_v3s4(AtomBuilder_R ab, Reg_(V3_S4) v) MipsAtomComp_ProcMap_(ab, mac_gte_mv_from_data_r_mac123(v.x, v.y, v.z))
|
|
||||||
|
|
||||||
#pragma endregion MACs (Mips Atom Components)
|
#pragma endregion MACs (Mips Atom Components)
|
||||||
|
|
||||||
#pragma region Atom Procs
|
#pragma region Bsked Atoms
|
||||||
|
|
||||||
/* ─── Local copy of PSYQ's sqrtbl (1/sqrt lookup table for VectorNormal). ───
|
typedef Struct_(Binds_SetGteWorld) {
|
||||||
* Source: PSYQ 4.7 libgte sqrtbl at 0x800185B4 in hello_camera.elf.
|
M3_S2* transform;
|
||||||
* objdump -s --start-address=0x800185B4 --stop-address=0x800185F4 hello_camera.elf → 192 entries × 16-bit signed, in 1.12 fixed-point (max value 0x1000 = 1.0).
|
|
||||||
*
|
|
||||||
* Data is identical to the libgte original (byte-for-byte verified).
|
|
||||||
*
|
|
||||||
* ─── Per-entry semantics (decoded from libgte msc02 VectorNormal) ───
|
|
||||||
* Each entry is `1/sqrt(x)` in 1.12 fixed point (value / 4096).
|
|
||||||
* The 192 entries span 4 octaves of the input magnitude, with 48 entries per octave:
|
|
||||||
* Octave 0 (entries 0- 47): mantissa in [0x8000, 0x10000) output ~[1.000, 0.707]
|
|
||||||
* Octave 1 (entries 48- 95): mantissa in [0x10000, 0x20000) output ~[0.707, 0.500]
|
|
||||||
* Octave 2 (entries 96-143): mantissa in [0x20000, 0x40000) output ~[0.500, 0.354]
|
|
||||||
* Octave 3 (entries144-191): mantissa in [0x40000, 0x80000) output ~[0.354, 0.251]
|
|
||||||
* Within each octave, 8 sub-entries interpolate over the 8 fractional bits of the mantissa
|
|
||||||
* (the byte `(0x80 | (i mod 8))` for the lower-byte of the aligned value).
|
|
||||||
* Sampling the first value of each octave:
|
|
||||||
* [0] 0x1000 = 1.0000 ; 1 / sqrt(1.0000)
|
|
||||||
* [48] 0x0e4f = 0.8940 ; 1 / sqrt(1.2500)
|
|
||||||
* [96] 0x0d10 = 0.8164 ; 1 / sqrt(1.5000)
|
|
||||||
* [144] 0x0c0a = 0.7520 ; 1 / sqrt(1.7500)
|
|
||||||
* And representative sub-entries within octave 0 (mantissa in [0x8000, 0x8100)):
|
|
||||||
* [0] 0x1000 = 1.0000 ; 1 / sqrt(0x8000)
|
|
||||||
* [1] 0x0fe0 = 0.9922 ; 1 / sqrt(0x8100)
|
|
||||||
* [2] 0x0fc1 = 0.9846 ; 1 / sqrt(0x8200)
|
|
||||||
* [3] 0x0fa3 = 0.9773 ; 1 / sqrt(0x8300)
|
|
||||||
* [4] 0x0f85 = 0.9700 ; 1 / sqrt(0x8400)
|
|
||||||
* [5] 0x0f68 = 0.9629 ; 1 / sqrt(0x8500)
|
|
||||||
* [6] 0x0f4c = 0.9561 ; 1 / sqrt(0x8600)
|
|
||||||
* [7] 0x0f30 = 0.9492 ; 1 / sqrt(0x8700)
|
|
||||||
*
|
|
||||||
* The algorithm's `addi -64 / sll 1 / lh` selects the entry at `(aligned - 64) * 2` for the case where `aligned` has its top bit at bit 24.
|
|
||||||
* After the sllv/srav pair, `aligned` always lands in `[0x80, 0x100)`
|
|
||||||
* (with top bit at bit 24 → after `sub $aligned - 64`, the index sits in `[0x40, 0x80) * 2 = [0x80, 0x100)` bytes = entries [64, 128) within the sqrtbl).
|
|
||||||
* The earlier 64 entries (octave 0) are reached when the magnitude after shifting puts the top bit below bit 24 (the `sllv` branch),
|
|
||||||
* and the load upper_halves of the table bracket the input range.
|
|
||||||
* The later 64 entries (octaves 2-3) are the `srav` branch when the magnitude's top bit is well above bit 24.
|
|
||||||
*
|
|
||||||
* Reproduced verbatim from libgte (verified against libpsn00b/psxgte/vector.s:100-123 — 24 rows × 8 halfwords, last entry 0x0804).
|
|
||||||
* */
|
|
||||||
internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
|
|
||||||
0x1000, 0x0fe0, 0x0fc1, 0x0fa3, 0x0f85, 0x0f68, 0x0f4c, 0x0f30,
|
|
||||||
0x0f15, 0x0efb, 0x0ee1, 0x0ec7, 0x0eae, 0x0e96, 0x0e7e, 0x0e66,
|
|
||||||
0x0e4f, 0x0e38, 0x0e22, 0x0e0c, 0x0df7, 0x0de2, 0x0dcd, 0x0db9,
|
|
||||||
0x0da5, 0x0d91, 0x0d7e, 0x0d6b, 0x0d58, 0x0d45, 0x0d33, 0x0d21,
|
|
||||||
0x0d10, 0x0cff, 0x0cee, 0x0cdd, 0x0ccc, 0x0cbc, 0x0cac, 0x0c9c,
|
|
||||||
0x0c8d, 0x0c7d, 0x0c6e, 0x0c5f, 0x0c51, 0x0c42, 0x0c34, 0x0c26,
|
|
||||||
0x0c18, 0x0c0a, 0x0bfd, 0x0bef, 0x0be2, 0x0bd5, 0x0bc8, 0x0bbb,
|
|
||||||
0x0baf, 0x0ba2, 0x0b96, 0x0b8a, 0x0b7e, 0x0b72, 0x0b67, 0x0b5b,
|
|
||||||
0x0b50, 0x0b45, 0x0b39, 0x0b2e, 0x0b24, 0x0b19, 0x0b0e, 0x0b04,
|
|
||||||
0x0af9, 0x0aef, 0x0ae5, 0x0adb, 0x0ad1, 0x0ac7, 0x0abd, 0x0ab4,
|
|
||||||
0x0aaa, 0x0aa1, 0x0a97, 0x0a8e, 0x0a85, 0x0a7c, 0x0a73, 0x0a6a,
|
|
||||||
0x0a61, 0x0a59, 0x0a50, 0x0a47, 0x0a3f, 0x0a37, 0x0a2e, 0x0a26,
|
|
||||||
0x0a1e, 0x0a16, 0x0a0e, 0x0a06, 0x09fe, 0x09f6, 0x09ef, 0x09e7,
|
|
||||||
0x09e0, 0x09d8, 0x09d1, 0x09c9, 0x09c2, 0x09bb, 0x09b4, 0x09ad,
|
|
||||||
0x09a5, 0x099e, 0x0998, 0x0991, 0x098a, 0x0983, 0x097c, 0x0976,
|
|
||||||
0x096f, 0x0969, 0x0962, 0x095c, 0x0955, 0x094f, 0x0949, 0x0943,
|
|
||||||
0x093c, 0x0936, 0x0930, 0x092a, 0x0924, 0x091e, 0x0918, 0x0912,
|
|
||||||
0x090d, 0x0907, 0x0901, 0x08fb, 0x08f6, 0x08f0, 0x08eb, 0x08e5,
|
|
||||||
0x08e0, 0x08da, 0x08d5, 0x08cf, 0x08ca, 0x08c5, 0x08bf, 0x08ba,
|
|
||||||
0x08b5, 0x08b0, 0x08ab, 0x08a6, 0x08a1, 0x089c, 0x0897, 0x0892,
|
|
||||||
0x088d, 0x0888, 0x0883, 0x087e, 0x087a, 0x0875, 0x0870, 0x086b,
|
|
||||||
0x0867, 0x0862, 0x085e, 0x0859, 0x0855, 0x0850, 0x084c, 0x0847,
|
|
||||||
0x0843, 0x083e, 0x083a, 0x0836, 0x0831, 0x082d, 0x0829, 0x0824,
|
|
||||||
0x0820, 0x081c, 0x0818, 0x0814, 0x0810, 0x080c, 0x0808, 0x0804,
|
|
||||||
};
|
};
|
||||||
|
internal MipsAtom_(set_gte_world) atom_info(
|
||||||
typedef Struct_(Binds_NormalizeV3S4) {
|
atom_bind(Binds_SetGteWorld)
|
||||||
U2 src_offset; /* offset of src V3_S4 within the BIOS scratchpad */
|
|
||||||
U2 dst_offset; /* offset of dst V3_S4 within the BIOS scratchpad */
|
|
||||||
};
|
|
||||||
typedef Struct_(RegUse_build_normalize_v3s4) {
|
|
||||||
Reg scratch; /* scratchpad base; loaded via load_word_imm below. */
|
|
||||||
Reg src_ptr;
|
|
||||||
Reg dst_ptr;
|
|
||||||
Reg recip_est; /* |v|² sum + shift-input + sqrtbl[index] */
|
|
||||||
Reg norm; Reg shift;
|
|
||||||
Reg src_x;
|
|
||||||
union { Reg mac1_scratch, dst_offset; } t3;
|
|
||||||
union { Reg mac2_scratch; } t4;
|
|
||||||
union { Reg btarget, shift_count, lookup_addr, src_z, src_offset; } t5;
|
|
||||||
};
|
|
||||||
/* ─── Full normalize (all 4 stages inline) ───
|
|
||||||
* Generic 4-stage GTE normalize (SQR → sum+LZCR → align+sqrtbl → GPF+srav). */
|
|
||||||
internal MipsAtom* build_normalize_v3s4(AtomArena_R aa, RegUse_build_normalize_v3s4 r)
|
|
||||||
MipsAtom_Proc_(aa, {
|
|
||||||
/* Load scratch base via immediate (always Scratchpad_Loc = 0x1F800000 — the BIOS
|
|
||||||
* scratchpad, aliased by every consumer's ResolveLookAtScratch struct). */
|
|
||||||
mac_load_word_imm(r.scratch, Scratchpad_Loc),
|
|
||||||
/* Tape pop: src_offset, dst_offset = 4 bytes (packed into 1 U4: low16=src, high16=dst).
|
|
||||||
* Loads back-to-back fill each other's load-delay slots; the subsequent add_u
|
|
||||||
* (2 cycles after the matching load) sees a valid value. */
|
|
||||||
load_half(r.t5.src_offset, R_TapePtr, O_(Binds_NormalizeV3S4, src_offset)),
|
|
||||||
load_half(r.t3.dst_offset, R_TapePtr, O_(Binds_NormalizeV3S4, dst_offset)),
|
|
||||||
LdSlot_ add_u(r.src_ptr, r.scratch, r.t5.src_offset),
|
|
||||||
LdSlot_ add_u(r.dst_ptr, r.scratch, r.t3.dst_offset),
|
|
||||||
LdSlot_ add_ui_self(R_TapePtr, S_(Binds_NormalizeV3S4)),
|
|
||||||
|
|
||||||
/* Load src.x/y/z from r_src_ptr (caller-determined address) into r_tmp/r_recip_est/r_branch_tmp.
|
|
||||||
* r.rt1_src_x holds src.x throughout stages 1-2 — r_mac2_scratch is clobbered to MAC2 in stage 1.5 (line below).
|
|
||||||
* t5.src_offset/dst_offset are dead by here; t5 is reused for src.z in the mac_load_word_v3 below. */
|
|
||||||
mac_load_word_v3(r.src_x, r.recip_est, r.t5.src_z, r.src_ptr, 0),
|
|
||||||
|
|
||||||
/* Stage 1: mtc2 src → IR1/2/3, SQR fires. */
|
|
||||||
LdSlot_ mac_gte_sqr_v3s4(r.src_x, r.recip_est, r.t5.src_z, LdSlot_ nop),
|
|
||||||
|
|
||||||
/* Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS. */
|
|
||||||
mac_gte_mv_from_data_r_mac123(r.t3.mac1_scratch, r.t4.mac2_scratch, r.norm), LdSlot_ nop,
|
|
||||||
add_u_self( r.norm, r.t3.mac1_scratch),
|
|
||||||
add_u_self( r.norm, r.t4.mac2_scratch),
|
|
||||||
gte_mv_to_data_r( r.norm, C2_LZCS), GteDelay_ nop2,
|
|
||||||
gte_mv_from_data_r(r.shift, C2_LZCR), GteDelay_ nop,
|
|
||||||
|
|
||||||
/* Stage 3: round LZCR to even, compute half-shift, align |v|² to bit 24.
|
|
||||||
* r_norm holds |v|² sum; r_shift holds the LZCR count from mfc2.
|
|
||||||
* After the component: r_shift = even(LZCR), r_norm = half-shift, r_mac1_scratch = |v|². */
|
|
||||||
mac_lzcr_round_even_half_shift(r.shift, r.norm, r.t3.mac1_scratch),
|
|
||||||
/* r_branch_tmp = LZCR - 24 (overwrites r_branch_tmp; src.z no longer needed after SQR) */
|
|
||||||
add_si( r.t5.btarget, r.shift, -24),
|
|
||||||
branch_lt_zero(r.t5.btarget, atom_offset(aligned_done, srav_path)), BdSlot_ nop, /* bltz → srav_path (LZCR < 24 path) */
|
|
||||||
jump_rel(atom_offset(srav_path, aligned_done)), /* b → aligned_done (LZCR >= 24 path) */
|
|
||||||
BdSlot_ shift_lleft_var(r.t3.mac1_scratch, r.t3.mac1_scratch, r.t5.btarget), /* src=sum (r_mac1_scratch), dst=same */
|
|
||||||
atom_label(srav_path)
|
|
||||||
li_s( r.t5.shift_count, 24),
|
|
||||||
sub_s(r.t5.shift_count, r.t5.shift_count, r.shift),
|
|
||||||
shift_aright_var(r.t3.mac1_scratch, r.t3.mac1_scratch, r.t5.shift_count), /* src=sum (r_mac1_scratch), dst=same */
|
|
||||||
atom_label(aligned_done)
|
|
||||||
// Save the shift count to r_shift before the next 5 instructions overwrite r_norm (the sqrtbl lookup loads 1/|v| into r_norm, which becomes IR0 in stage 4).
|
|
||||||
or_u(r.shift, r.norm, 0), /* r_shift ← shift count (preserved through stage 4) */
|
|
||||||
/* r_mac1_scratch holds |v|² aligned (top bit at bit 7). */
|
|
||||||
add_si( r.t3.mac1_scratch, r.t3.mac1_scratch, -64),
|
|
||||||
shift_lleft(r.t3.mac1_scratch, r.t3.mac1_scratch, 1),
|
|
||||||
mac_load_word_imm(r.t5.lookup_addr, & gte_normalize_sqr_tbl), add_u_self(r.t5.lookup_addr, r.t3.mac1_scratch),
|
|
||||||
load_half(r.norm, r.t5.lookup_addr, 0), /* r_norm = sqrtbl[aligned-64] = 1/|v| (IR0 in stage 4) */
|
|
||||||
|
|
||||||
/* r_branch_tmp held the sqrtbl base+index, NOT src.z. Reload src.z from scratch now that r_branch_tmp is free. */
|
|
||||||
LdSlot_ load_word(r.t5.src_z, r.src_ptr, O_(V3_S4,z)), /* r_branch_tmp = src.z (for IR3 in stage 4) */
|
|
||||||
|
|
||||||
/* Stage 4: GPF + srav finalize (r_shift = shift count, r_norm = 1/|v|). */
|
|
||||||
LdSlot_ mac_gte_general_purpose_interopolation(
|
|
||||||
r.norm,
|
|
||||||
r.src_x, /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */
|
|
||||||
r.recip_est,
|
|
||||||
r.t5.src_z, /* IR3 = src.z (reloaded) */
|
|
||||||
r.t4.mac2_scratch, r.recip_est, r.t5.src_z,
|
|
||||||
GteDelay_ nop,
|
|
||||||
GteDelay_ nop
|
|
||||||
),
|
|
||||||
/* sra by r_shift = (31-LZCR)/2 (saved before sqrtbl lookup) */
|
|
||||||
mac_shift_aright_var_v3_self(r.t4.mac2_scratch, r.recip_est, r.t5.src_z, r.shift),
|
|
||||||
/* Store result.x/y/z to r_dst_ptr (caller-determined dst address). */
|
|
||||||
mac_store_word_v3(r.t4.mac2_scratch, r.recip_est, r.t5.src_z, r.dst_ptr, 0),
|
|
||||||
|
|
||||||
mac_yield()
|
|
||||||
})
|
|
||||||
|
|
||||||
/* ─── GTE OP cross product (a × b → out) ───
|
|
||||||
* Generalized V3_S4 cross product via GTE OP (OuterProduct12 libpsyx convention).
|
|
||||||
* The >> 12 shift converts S12.20 → S12.0 OuterProduct12. */
|
|
||||||
typedef Struct_(Binds_gte_cross_v3s4) { V3_S4* src_a; V3_S4* src_b; V3_S4* out; };
|
|
||||||
typedef Struct_(RegUse_gte_cross_v3s4) {
|
|
||||||
Reg_(V3_S4) a;
|
|
||||||
Reg_(V3_S4) b;
|
|
||||||
union { Reg out, t0; } x;
|
|
||||||
union { Reg src_a, t1, rt11; } y;
|
|
||||||
union { Reg src_b, t2, rt22; } z;
|
|
||||||
};
|
|
||||||
internal MipsAtom* gte_cross_v3s4(AtomArena_R aa, RegUse_gte_cross_v3s4 r)
|
|
||||||
atom_info(atom_bind(Binds_gte_cross_v3s4)) MipsAtom_Proc_(aa, {
|
|
||||||
load_word(r.y.src_a, R_TapePtr, O_(Binds_gte_cross_v3s4,src_a)),
|
|
||||||
load_word(r.z.src_b, R_TapePtr, O_(Binds_gte_cross_v3s4,src_b)),
|
|
||||||
load_word(r.x.out, R_TapePtr, O_(Binds_gte_cross_v3s4,out)),
|
|
||||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_gte_cross_v3s4)),
|
|
||||||
|
|
||||||
mac_load_v3s4(r.a, r.y.src_a, 0), LdSlot_
|
|
||||||
mac_load_v3s4(r.b, r.z.src_b, 0), LdSlot_
|
|
||||||
mac_gte_op_cross_v3s4(r.a, r.b), /* RT diagonal + IR + OP + MAC read + shift */
|
|
||||||
mac_store_v3s4(r.a, r.x.out, 0),
|
|
||||||
|
|
||||||
mac_yield()
|
|
||||||
})
|
|
||||||
#pragma endregion Atom Procs
|
|
||||||
|
|
||||||
#pragma region Baked Atoms
|
|
||||||
|
|
||||||
typedef Struct_(Binds_SetGteMT3S2S4) {
|
|
||||||
MT3_S2S4* transform;
|
|
||||||
};
|
|
||||||
internal MipsAtom_(set_gte_mt3s2s4) atom_info(
|
|
||||||
atom_bind(Binds_SetGteMT3S2S4)
|
|
||||||
, atom_reads(R_TapePtr)
|
, atom_reads(R_TapePtr)
|
||||||
){
|
){
|
||||||
/* Pop matrix address from tape into R_T3 ($11) */
|
/* Pop matrix address from tape into R_T3 ($11) */
|
||||||
load_word(R_T3, R_TapePtr, O_(Binds_SetGteMT3S2S4,transform)),
|
load_word(R_T3, R_TapePtr, O_(Binds_SetGteWorld,transform)),
|
||||||
add_ui_self( R_TapePtr, S_(Binds_SetGteMT3S2S4)),
|
add_ui_self( R_TapePtr, S_(Binds_SetGteWorld)),
|
||||||
/* Load 3x3 Rotation + 3x1 Translation from R_T3 into GTE CONTROL Regs (ctc2) */
|
/* Load 3x3 Rotation + 3x1 Translation from R_T3 into GTE CONTROL Regs (ctc2) */
|
||||||
load_word(R_T0, R_T3, 0),
|
load_word(R_T0, R_T3, 0), load_word(R_T1, R_T3, 4),
|
||||||
load_word(R_T1, R_T3, 4),
|
gte_mv_to_ctrl_r(R_T0, gte_cr_RT11), gte_mv_to_ctrl_r(R_T1, gte_cr_RT12),
|
||||||
gte_mv_to_ctrl_r(R_T0, gte_cr_RT11),
|
load_word(R_T0, R_T3, 8), load_word(R_T1, R_T3, 12), load_word(R_T2, R_T3, 16),
|
||||||
gte_mv_to_ctrl_r(R_T1, gte_cr_RT12),
|
gte_mv_to_ctrl_r(R_T0, gte_cr_RT13), gte_mv_to_ctrl_r(R_T1, gte_cr_RT21), gte_mv_to_ctrl_r(R_T2, gte_cr_RT22),
|
||||||
load_word(R_T0, R_T3, 8),
|
load_word(R_T0, R_T3, 20), load_word(R_T1, R_T3, 24), load_word(R_T2, R_T3, 28),
|
||||||
load_word(R_T1, R_T3, 12),
|
gte_mv_to_ctrl_r(R_T0, gte_cr_TRX), gte_mv_to_ctrl_r(R_T1, gte_cr_TRY), gte_mv_to_ctrl_r(R_T2, gte_cr_TRZ),
|
||||||
load_word(R_T2, R_T3, 16),
|
|
||||||
gte_mv_to_ctrl_r(R_T0, gte_cr_RT13),
|
|
||||||
gte_mv_to_ctrl_r(R_T1, gte_cr_RT21),
|
|
||||||
gte_mv_to_ctrl_r(R_T2, gte_cr_RT22),
|
|
||||||
load_word(R_T0, R_T3, 20),
|
|
||||||
load_word(R_T1, R_T3, 24),
|
|
||||||
load_word(R_T2, R_T3, 28),
|
|
||||||
gte_mv_to_ctrl_r(R_T0, gte_cr_TRX),
|
|
||||||
gte_mv_to_ctrl_r(R_T1, gte_cr_TRY),
|
|
||||||
gte_mv_to_ctrl_r(R_T2, gte_cr_TRZ),
|
|
||||||
mac_yield()
|
mac_yield()
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|||||||
+53
-151
@@ -16,6 +16,9 @@
|
|||||||
* gte_mv_to_data_r (gte + mv + to + data + register)
|
* gte_mv_to_data_r (gte + mv + to + data + register)
|
||||||
* gte_lw_v0_xy(base) (gte + lw + v0 + xy)
|
* gte_lw_v0_xy(base) (gte + lw + v0 + xy)
|
||||||
* load_upper_i (load-upper + immediate, unique verb)
|
* load_upper_i (load-upper + immediate, unique verb)
|
||||||
|
*
|
||||||
|
* Vendor mnemonics (gte_mtc2, gte_mfc2, gte_lwc2, gte_swc2, etc.) are NOT in this header.
|
||||||
|
* They are in the opt-in `gte_vendor_sym.h` for users who prefer the textbook MIPS assembly mnemonics.
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
|
|
||||||
#ifdef INTELLISENSE_DIRECTIVES
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
@@ -30,7 +33,7 @@
|
|||||||
* gte.h — Geometry Transformation Engine (COP2) for the PS1
|
* gte.h — Geometry Transformation Engine (COP2) for the PS1
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
*
|
*
|
||||||
* Hand-rolled DSL for emitting GTE/MIPS instruction words from C.
|
* Hand-rolled DSL for emitting GTE/MIPS instruction words as raw `.word` constants from C.
|
||||||
* No GCC inline-assembly string syntax in the code body.
|
* No GCC inline-assembly string syntax in the code body.
|
||||||
*
|
*
|
||||||
* STYLE NOTES
|
* STYLE NOTES
|
||||||
@@ -98,20 +101,20 @@ enum {
|
|||||||
|
|
||||||
/* Semantic Aliases for GTE Data Registers */
|
/* Semantic Aliases for GTE Data Registers */
|
||||||
enum {
|
enum {
|
||||||
C2_InV0_XY = C2_VXY0, /* Input Vector 0 (X, Y) */
|
gte_in_v0_xy = C2_VXY0, /* Input Vector 0 (X, Y) */
|
||||||
C2_InV0_Z = C2_VZ0, /* Input Vector 0 (Z) */
|
gte_in_v0_z = C2_VZ0, /* Input Vector 0 (Z) */
|
||||||
C2_InV1_XY = C2_VXY1, /* Input Vector 1 (X, Y) */
|
gte_in_v1_xy = C2_VXY1, /* Input Vector 1 (X, Y) */
|
||||||
C2_InV1_Z = C2_VZ1, /* Input Vector 1 (Z) */
|
gte_in_v1_z = C2_VZ1, /* Input Vector 1 (Z) */
|
||||||
C2_InV2_XY = C2_VXY2, /* Input Vector 2 (X, Y) */
|
gte_in_v2_xy = C2_VXY2, /* Input Vector 2 (X, Y) */
|
||||||
C2_InV2_Z = C2_VZ2, /* Input Vector 2 (Z) */
|
gte_in_v2_z = C2_VZ2, /* Input Vector 2 (Z) */
|
||||||
C2_In_RGB = C2_RGB, /* Input Color (R, G, B, MipsCode) */
|
gte_in_rgb = C2_RGB, /* Input Color (R, G, B, MipsCode) */
|
||||||
C2_OutSrc_XY0 = C2_SXY0, /* Output Screen Coord 0 (X, Y) */
|
gte_out_scr_xy0 = C2_SXY0, /* Output Screen Coord 0 (X, Y) */
|
||||||
C2_OutSrc_XY1 = C2_SXY1, /* Output Screen Coord 1 (X, Y) */
|
gte_out_scr_xy1 = C2_SXY1, /* Output Screen Coord 1 (X, Y) */
|
||||||
C2_OutSrc_XY2 = C2_SXY2, /* Output Screen Coord 2 (X, Y) */
|
gte_out_scr_xy2 = C2_SXY2, /* Output Screen Coord 2 (X, Y) */
|
||||||
C2_OutDepth = C2_OTZ, /* Output Ordering Table Z (Depth) */
|
gte_out_depth = C2_OTZ, /* Output Ordering Table Z (Depth) */
|
||||||
C2_MathAccu0 = C2_MAC0, /* Math Accumulator 0 */
|
gte_math_accum0 = C2_MAC0, /* Math Accumulator 0 */
|
||||||
C2_MathAccu1 = C2_MAC1, /* Math Accumulator 1 */
|
gte_math_accum1 = C2_MAC1, /* Math Accumulator 1 */
|
||||||
C2_MathAccu2 = C2_MAC2, /* Math Accumulator 2 */
|
gte_math_accum2 = C2_MAC2, /* Math Accumulator 2 */
|
||||||
};
|
};
|
||||||
|
|
||||||
/* --- GTE Command Semantics (The Bitfield Meanings) ---
|
/* --- GTE Command Semantics (The Bitfield Meanings) ---
|
||||||
@@ -158,8 +161,6 @@ enum {
|
|||||||
gte_cmd_nclip = 0x06, /* Normal Clipping (Backface culling) */
|
gte_cmd_nclip = 0x06, /* Normal Clipping (Backface culling) */
|
||||||
gte_cmd_op = 0x0C, /* Outer Product */
|
gte_cmd_op = 0x0C, /* Outer Product */
|
||||||
gte_cmd_mvmva = 0x12, /* Matrix Vector Multiply & Add (Custom math) */
|
gte_cmd_mvmva = 0x12, /* Matrix Vector Multiply & Add (Custom math) */
|
||||||
gte_cmd_sqr = 0x28, /* Square vector — MAC[i] = IR[i]²; IR[i] ← MAC[i] saturated */
|
|
||||||
gte_cmd_gpf = 0x3D, /* General-purpose Interpolation — MAC[i] = IR0 * IR[i] */
|
|
||||||
|
|
||||||
/* --- GTE Command Bit-Field Layout ---
|
/* --- GTE Command Bit-Field Layout ---
|
||||||
* A GTE command word (sent to COP2 with RS=1) is laid out as:
|
* A GTE command word (sent to COP2 with RS=1) is laid out as:
|
||||||
@@ -170,42 +171,19 @@ enum {
|
|||||||
* +------------+--+-----+------+------+------+------+---+--------+----------+
|
* +------------+--+-----+------+------+------+------+---+--------+----------+
|
||||||
* \_____ GTE_PAYLOAD _____/ \__ GTE_CMD __/
|
* \_____ GTE_PAYLOAD _____/ \__ GTE_CMD __/
|
||||||
*
|
*
|
||||||
* Shifts/masks below are the *bit positions* and *bit widths* of each configurable field, used by the ENC_GTE_CMD encoder.
|
* Shifts/masks below are the *bit positions* and *bit widths* of each
|
||||||
|
* configurable field, used by the ENC_GTE_CMD encoder.
|
||||||
* Mirrors the OPCODE_SHIFT / RS_SHIFT convention used in mips.h.
|
* Mirrors the OPCODE_SHIFT / RS_SHIFT convention used in mips.h.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
gte_shift_sf = 19, gte_width_sf = 1,
|
gte_shift_sf = 19, gte_width_sf = 1, gte_mask_sf = 0x1,
|
||||||
gte_shift_mx = 17, gte_width_mx = 2,
|
gte_shift_mx = 17, gte_width_mx = 2, gte_mask_mx = 0x3,
|
||||||
gte_shift_v = 15, gte_width_v = 2,
|
gte_shift_v = 15, gte_width_v = 2, gte_mask_v = 0x3,
|
||||||
gte_shift_cv = 13, gte_width_cv = 2,
|
gte_shift_cv = 13, gte_width_cv = 2, gte_mask_cv = 0x3,
|
||||||
gte_shift_lm = 10, gte_width_lm = 1,
|
gte_shift_lm = 10, gte_width_lm = 1, gte_mask_lm = 0x1,
|
||||||
gte_shift_cmd = 0, gte_width_cmd = 6,
|
gte_shift_cmd = 0, gte_width_cmd = 6, gte_mask_cmd = 0x3F,
|
||||||
|
|
||||||
/* Fake command number (bits 24-20) — IGNORED by the GTE hardware per PSX-SPX `geometrytransformationenginegte.md` line 48.
|
|
||||||
* libgte's compiler emits non-zero values in this field as a disassembly signature. */
|
|
||||||
gte_shift_fake_cmd = 20,
|
|
||||||
gte_width_fake_cmd = 5,
|
|
||||||
};
|
};
|
||||||
|
|
||||||
/* --- GTE Control Register Aliases (Pitfall 1) ---
|
|
||||||
* Three pairs of aliases map to the C2 control-register slot:
|
|
||||||
* C2[24] = gte_cr_RBK (background R) | gte_cr_OFX (screen offset X)
|
|
||||||
* C2[25] = gte_cr_GBK (background G) | gte_cr_OFY (screen offset Y)
|
|
||||||
* C2[26] = gte_cr_BBK (background B) | gte_cr_H (projection plane distance H)
|
|
||||||
* Cross-alias writes inside one atom body, or across the wave-context boundary, silently clobber each other.
|
|
||||||
* The metaprogram's check_gte_cr_alias_writes (CHECK_RULES row) warns about each pair per source.
|
|
||||||
* See psx-spx docs/gte_reference.md §"Control-register alias table" for the silicon rationale and the libgte outer-product convention.
|
|
||||||
*/
|
|
||||||
|
|
||||||
/* --- RT-matrix packed-slot convention (Pitfall 4) ---
|
|
||||||
* The silicon packs two 16-bit RT elements per 32-bit C2 slot:
|
|
||||||
* C2[2] = (RT22 << 16) | RT13 (gte_cr_RT13 writes the low half, gte_cr_RT22 writes the high half)
|
|
||||||
* C2[4] = (RT33 << 16) | RT22 (gte_cr_RT22 writes the low half — clobbers prior RT22 value if RT13 was also written)
|
|
||||||
* OP and MVMVA read D1/D2/D3 from these packed slots.
|
|
||||||
* The libgte outer-product convention (see ac_apply_matrix_lv at gte.atom.c:108-122) writes C2[2] then C2[4] in sequence;
|
|
||||||
* the SECOND write's low half is RT22, not RT13.
|
|
||||||
*/
|
|
||||||
|
|
||||||
/* --- GTE Control Register Indices (for ctc2/cfc2) ---
|
/* --- GTE Control Register Indices (for ctc2/cfc2) ---
|
||||||
* Preprocessor-visible integer ids for the COP2 control register file.
|
* Preprocessor-visible integer ids for the COP2 control register file.
|
||||||
* Each enum value is bound to a parallel `_Code` `#define` so the preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
* Each enum value is bound to a parallel `_Code` `#define` so the preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
||||||
@@ -265,10 +243,10 @@ enum { _C2_OPS_ = 0
|
|||||||
* bit 1 (0x02): register class — 0 = data, 1 = control
|
* bit 1 (0x02): register class — 0 = data, 1 = control
|
||||||
* bit 2 (0x04): direction — 0 = read, 1 = write
|
* bit 2 (0x04): direction — 0 = read, 1 = write
|
||||||
*
|
*
|
||||||
* The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit numbers as general MIPS `cop_mf` / `cop_mt` defined in mips.h
|
* The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit numbers as the general MIPS `cop_mf` / `cop_mt` defined in mips.h
|
||||||
* (which target the data register file on any coprocessor).
|
* (which target the data register file on any coprocessor).
|
||||||
* They are re-aliased here so the four-way table reads like the spec mnemonics (MFC2 / CFC2 / MTC2 / CTC2)
|
* They are re-aliased here so the four-way table reads like the spec mnemonics (MFC2 / CFC2 / MTC2 / CTC2)
|
||||||
* and so the encoding is next to its only consumer (this header).
|
* and so the encoding lives next to its only consumer (this header).
|
||||||
*
|
*
|
||||||
* Vendor mnemonic aliases (gte_mfc2 / gte_mtc2 / gte_cfc2 / gte_ctc2) live in gte_vendor_sym.h. */
|
* Vendor mnemonic aliases (gte_mfc2 / gte_mtc2 / gte_cfc2 / gte_ctc2) live in gte_vendor_sym.h. */
|
||||||
enum { _C2_TX_SUBS_ = 0
|
enum { _C2_TX_SUBS_ = 0
|
||||||
@@ -292,7 +270,8 @@ enum { _C2_TX_SUBS_ = 0
|
|||||||
// #define gte_mv_from_data_r(rt, rd) enc_gte_tx(cop_mf, (rt), (rd)) /* Move GTE Control Register (rd) to GPR (rt) */
|
// #define gte_mv_from_data_r(rt, rd) enc_gte_tx(cop_mf, (rt), (rd)) /* Move GTE Control Register (rd) to GPR (rt) */
|
||||||
|
|
||||||
/* GTE Data vs Control Register Transfers
|
/* GTE Data vs Control Register Transfers
|
||||||
* Each macro emits a single instruction for one of MFC2/CFC2/MTC2/CTC2.
|
*
|
||||||
|
* Each macro emits a single .word constant for one of MFC2/CFC2/MTC2/CTC2.
|
||||||
*
|
*
|
||||||
* `rd` is the C2 register index in the file the sub-opcode names:
|
* `rd` is the C2 register index in the file the sub-opcode names:
|
||||||
* gte_mv_from_data_r / gte_mv_to_data_r → C2 data register file
|
* gte_mv_from_data_r / gte_mv_to_data_r → C2 data register file
|
||||||
@@ -307,14 +286,14 @@ enum { _C2_TX_SUBS_ = 0
|
|||||||
#define gte_mv_from_ctrl_r(rt, rd) enc_gte_tx(sub_cfc2, (rt), (rd)) /* Copy From ctrl reg */
|
#define gte_mv_from_ctrl_r(rt, rd) enc_gte_tx(sub_cfc2, (rt), (rd)) /* Copy From ctrl reg */
|
||||||
#define gte_mv_to_data_r(rt, rd) enc_gte_tx(sub_mtc2, (rt), (rd)) /* Move To data reg */
|
#define gte_mv_to_data_r(rt, rd) enc_gte_tx(sub_mtc2, (rt), (rd)) /* Move To data reg */
|
||||||
#define gte_mv_to_ctrl_r(rt, rd) enc_gte_tx(sub_ctc2, (rt), (rd)) /* Copy To ctrl reg */
|
#define gte_mv_to_ctrl_r(rt, rd) enc_gte_tx(sub_ctc2, (rt), (rd)) /* Copy To ctrl reg */
|
||||||
#define GteDelay_ // Annotate an instruction as filling a CPU <-> GTE DMA delay slot/s
|
|
||||||
|
|
||||||
/* COP2 Data Load (lwc2): `lwc2 rt, off(rs)`
|
/* COP2 Data Load (lwc2): `lwc2 rt, off(rs)`
|
||||||
* Layout: [op_lwc2:6][rs:5][rt:5][imm:16]
|
* Layout: [op_lwc2:6][rs:5][rt:5][imm:16]
|
||||||
* - rs: GPR base address
|
* - rs: GPR base address
|
||||||
* - rt: COP2 data register index (0..31)
|
* - rt: COP2 data register index (0..31)
|
||||||
* - imm: signed 16-bit offset
|
* - imm: signed 16-bit offset
|
||||||
* NOTE: When `rs` is a runtime register, the encoding cannot be pre-baked into a .word — use the string-style `gte_load_v0` macro below instead. */
|
* NOTE: When `rs` is a runtime register, the encoding cannot be pre-baked
|
||||||
|
* into a .word — use the string-style `gte_load_v0` macro below instead. */
|
||||||
#define enc_gte_lw(rt, base, off) enc_i(op_lwc2, (base), (rt), (off))
|
#define enc_gte_lw(rt, base, off) enc_i(op_lwc2, (base), (rt), (off))
|
||||||
/* Store Word */
|
/* Store Word */
|
||||||
#define enc_gte_sw(rt, base, off) enc_i(op_swc2, (base), (rt), (off))
|
#define enc_gte_sw(rt, base, off) enc_i(op_swc2, (base), (rt), (off))
|
||||||
@@ -323,30 +302,30 @@ enum { _C2_TX_SUBS_ = 0
|
|||||||
* `swc2` is redundant when we're already inside the `gte_` namespace.
|
* `swc2` is redundant when we're already inside the `gte_` namespace.
|
||||||
* gte_lw rt, base, off → lwc2 rt, off(base)
|
* gte_lw rt, base, off → lwc2 rt, off(base)
|
||||||
* gte_sw rt, base, off → swc2 rt, off(base)
|
* gte_sw rt, base, off → swc2 rt, off(base)
|
||||||
* For the typical user-facing vector-level load (xy + z as two instructions), use the higher-level `gte_load_vN` macros below. */
|
* For the typical user-facing vector-level load (xy + z as two instructions),
|
||||||
|
* use the higher-level `gte_load_vN` macros below. */
|
||||||
#define gte_lw(rt, base, off) enc_gte_lw(rt, base, off)
|
#define gte_lw(rt, base, off) enc_gte_lw(rt, base, off)
|
||||||
#define gte_sw(rt, base, off) enc_gte_sw(rt, base, off)
|
#define gte_sw(rt, base, off) enc_gte_sw(rt, base, off)
|
||||||
|
|
||||||
/* GTE Command Format
|
/* GTE Command Format
|
||||||
* Opcode is always MIPS_OP_COP2, RS is always 1 (CO).
|
* Opcode is always MIPS_OP_COP2, RS is always 1 (CO).
|
||||||
* Lower 25 bits are GTE-specific command payload.
|
* The lower 25 bits are the GTE-specific command payload.
|
||||||
*
|
*
|
||||||
* The `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs` pattern in mips.h:
|
* The granular `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs` pattern in mips.h:
|
||||||
* Each one self-masks and shifts its own field, so a caller can build up a GTE command piece by piece
|
* Each one self-masks and shifts its own field, so a caller can build up a GTE command piece by piece
|
||||||
* (handy for state-driven MVMVA emitters that vary one field at a time).
|
* (handy for state-driven MVMVA emitters that vary one field at a time).
|
||||||
*
|
*
|
||||||
* `ENC_GTE_CMD` is an all-in-one convenience for emitting a full command word.
|
* `ENC_GTE_CMD` is the all-in-one convenience for emitting a full command word in one go.
|
||||||
* It just ORs the per-field encoders together. */
|
* It just ORs the per-field encoders together. */
|
||||||
#define gte_cmd_base (enc_op(op_cop2) | (1 << 25))
|
#define gte_cmd_base (enc_op(op_cop2) | (1 << 25))
|
||||||
|
|
||||||
/* Per-field encoders. Each one does (value & mask) << shift on its own. */
|
/* Per-field encoders. Each one does (value & mask) << shift on its own. */
|
||||||
#define enc_gte_sf(sf) ((sf) << gte_shift_sf )
|
#define enc_gte_sf(sf) (((sf) & gte_mask_sf ) << gte_shift_sf )
|
||||||
#define enc_gte_mx(mx) ((mx) << gte_shift_mx )
|
#define enc_gte_mx(mx) (((mx) & gte_mask_mx ) << gte_shift_mx )
|
||||||
#define enc_gte_v(v) ((v) << gte_shift_v )
|
#define enc_gte_v(v) (((v) & gte_mask_v ) << gte_shift_v )
|
||||||
#define enc_gte_cv(cv) ((cv) << gte_shift_cv )
|
#define enc_gte_cv(cv) (((cv) & gte_mask_cv ) << gte_shift_cv )
|
||||||
#define enc_gte_lm(lm) ((lm) << gte_shift_lm )
|
#define enc_gte_lm(lm) (((lm) & gte_mask_lm ) << gte_shift_lm )
|
||||||
#define enc_gte_cmd(cmd) ((cmd) << gte_shift_cmd )
|
#define enc_gte_cmd(cmd) (((cmd) & gte_mask_cmd) << gte_shift_cmd)
|
||||||
#define enc_gte_fake_cmd(x) ((x) << gte_shift_fake_cmd)
|
|
||||||
|
|
||||||
/* Composite: all six GTE fields + the COP2/CO base. */
|
/* Composite: all six GTE fields + the COP2/CO base. */
|
||||||
#define enc_gte_cmdw(sf, mx, v, cv, lm, cmd) ( \
|
#define enc_gte_cmdw(sf, mx, v, cv, lm, cmd) ( \
|
||||||
@@ -384,11 +363,11 @@ enum { _C2_TX_SUBS_ = 0
|
|||||||
* (the perspective divide happens regardless of `sf`).
|
* (the perspective divide happens regardless of `sf`).
|
||||||
*
|
*
|
||||||
* If we emit a strictly-spec-compliant word (`sf=0`, reserved bits clear),
|
* If we emit a strictly-spec-compliant word (`sf=0`, reserved bits clear),
|
||||||
* PCSX-Redux's GTE checks those bits more strictly than the silicon does and RTPT silently no-ops.
|
* PCSX-Redux's GTE checks those bits more strictly than the silicon does and RTPT silently no-ops —
|
||||||
* The floor's screen coordinates come out as raw projection-of-rotation (Z never divided),
|
* the floor's screen coordinates come out as raw projection-of-rotation (Z never divided),
|
||||||
* `nclip` ends up wrong, and the triangle is culled.
|
* `nclip` ends up wrong, and the triangle is culled.
|
||||||
*
|
*
|
||||||
* So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern.
|
* So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern everyone has shipped for 25 years.
|
||||||
* NCLIP / OP / MVMVA stay spec-clean — their reserved bits really are zero in the original PsyQ source.
|
* NCLIP / OP / MVMVA stay spec-clean — their reserved bits really are zero in the original PsyQ source.
|
||||||
* --------------------------------------------------------------------------
|
* --------------------------------------------------------------------------
|
||||||
*/
|
*/
|
||||||
@@ -398,91 +377,12 @@ enum { _C2_TX_SUBS_ = 0
|
|||||||
#define gte_cmdw_rtpt (gte_cmd_base | enc_gte_cmd(gte_cmd_rtpt ) | gte_cmdw_psyq_compat)
|
#define gte_cmdw_rtpt (gte_cmd_base | enc_gte_cmd(gte_cmd_rtpt ) | gte_cmdw_psyq_compat)
|
||||||
#define gte_cmdw_nclip (gte_cmd_base | enc_gte_cmd(gte_cmd_nclip))
|
#define gte_cmdw_nclip (gte_cmd_base | enc_gte_cmd(gte_cmd_nclip))
|
||||||
#define gte_cmdw_op (gte_cmd_base | enc_gte_cmd(gte_cmd_op ))
|
#define gte_cmdw_op (gte_cmd_base | enc_gte_cmd(gte_cmd_op ))
|
||||||
#define gte_cmdw_outer_product gte_cmdw_op /* "outer product" -- PSY-Q terminology */
|
#define gte_cmdw_outer_product gte_cmdw_op /* "outer product" -- NOCASH/Sdk terminology */
|
||||||
#define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology. */
|
#define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology */
|
||||||
#define gte_cmdw_cross gte_cmdw_op /* "cross product" -- geometric-algebra terminology.
|
|
||||||
* RGA(Lengyel): The GTE OP is a 3D signed-16-bit D x IR cross, not a generic RGA exterior product.
|
|
||||||
* The wedge alias is a 3D complement interpretation of the same 3 scalars (MAC1..MAC3). */
|
|
||||||
#define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva))
|
#define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
/* MVMVA with sf=0 (no shift, full-integer), cv=3 (no translation), v=3 (IR vector input).
|
|
||||||
* Reads input from IR1/2/3 (loaded via mtc2 rt, C2_IRx). MAC1/2/3 = RT row · IR (full product, no >>12).
|
|
||||||
* Per PSX-SPX: SAR (sf*12) with sf=0 = SAR 0 = no shift. */
|
|
||||||
#define gte_cmdw_mvmva_sf0_ir (gte_cmd_base | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_cmd(gte_cmd_mvmva))
|
|
||||||
|
|
||||||
/* MVMVA with sf=1 (>>12 shift, 4.12 fixed-point), cv=3 (no translation), v=3 (IR): for ApplyMatrixLV.
|
|
||||||
* Reads input from IR1/2/3 (loaded via mtc2 rt, C2_IRx). MAC1/2/3 = (RT row · IR) >> 12.
|
|
||||||
* Per PSX-SPX: SAR (sf*12) with sf=1 = SAR 12 = arithmetic right-shift by 12.
|
|
||||||
* This matches the libgte C-side ApplyMatrixLV output (R*pos >> 12). */
|
|
||||||
#define gte_cmdw_mvmva_ir (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_cmd(gte_cmd_mvmva))
|
|
||||||
|
|
||||||
/* MVMVA: sf=0, mx=3 (Light matrix), v=3 (IR), cv=3 (no TR).
|
|
||||||
* For pass1 of the C11 two-pass decomposition. Reads L matrix.
|
|
||||||
* Since L matrix is typically zero, pass1 contributes 0 to the combine. */
|
|
||||||
#define gte_cmdw_mvmva_sf0_mx3_v3_cv3 (gte_cmd_base | enc_gte_sf(0) | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
|
|
||||||
|
|
||||||
/* MVMVA: sf=1 (>>12), mx=3 (Light matrix), v=2 (V0), cv=0 (with TR).
|
|
||||||
* Matches the C11 ApplyMatrixLV pass 2 command word (0x49E012) exactly.
|
|
||||||
* The combine is (pass1 << 3) + pass2. */
|
|
||||||
#define gte_cmdw_mvmva_pass2_c11 (gte_cmd_base | enc_gte_sf(1) | enc_gte_v(2) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
|
|
||||||
|
|
||||||
/* MVMVA: sf=0, mx=3, v=2, cv=0. Matches the C11 pass 1 command. */
|
|
||||||
#define gte_cmdw_mvmva_pass1_c11 (gte_cmd_base | enc_gte_v(2) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
|
|
||||||
#define gte_cmdw_mvmva_no_tr gte_cmdw_mvmva_ir
|
|
||||||
|
|
||||||
/* MVMVA pass 2 — C11 ApplyMatrixLV command.
|
|
||||||
* Decoded: op_cop2 | CO | fake_cmd=4 | sf=1 (>>12) | mx=0 (RT matrix) | v=3 (IR) | cv=3 (no translation) | lm=0 | cmd=MVMVA.
|
|
||||||
* Reads (RT row · IR) >> 12 into MAC1/2/3. Per-field composition (no opaque literal)
|
|
||||||
* keeps the bit layout visible at the call site + matches the libgte C-side byte-exact. */
|
|
||||||
#define gte_cmdw_mvmva_c11_pass2 (gte_cmd_base | enc_gte_fake_cmd(4) | enc_gte_sf(1) | enc_gte_v(3) | enc_gte_mx(0) | enc_gte_cv(3) | enc_gte_cmd(gte_cmd_mvmva))
|
|
||||||
|
|
||||||
/* MVMVA: sf=1 (>>12), mx=0 (RT matrix), v=0 (V0), cv=3 (no TR). */
|
|
||||||
#define gte_cmdw_mvmva_sf1_mx0_v0_cv3 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_v(0) | enc_gte_mx(0) | enc_gte_cmd(gte_cmd_mvmva))
|
|
||||||
|
|
||||||
/* RTPS with sf=1 (12-bit shift, no translation): matches the output of libgte's ApplyMatrixLV when the GTE pipeline expects R*pos >> 12.
|
|
||||||
* The shift produces values like (-270, 710, 1713) which match the C11 reference path. */
|
|
||||||
#define gte_cmdw_rtps_sf1 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_cmd(gte_cmd_rtps))
|
|
||||||
|
|
||||||
/* SQR / GPF cosmetic-bits compat helpers.
|
|
||||||
* Each command's `_compat` macro ORs in the `fake_cmd` field value libgte happens to emit.
|
|
||||||
* The hardware ignores these bits (per PSX-SPX line 48). */
|
|
||||||
#define gte_cmdw_sqr_fake_sig enc_gte_fake_cmd(0x0A)
|
|
||||||
#define gte_cmdw_gpf_fake_sig enc_gte_fake_cmd(0x19)
|
|
||||||
|
|
||||||
/* SQR — Square Vector.
|
|
||||||
* PSX-SPX `geometrytransformationenginegte.md` §"SQR":
|
|
||||||
* [MAC1,MAC2,MAC3] = [IR1*IR1, IR2*IR2, IR3*IR3] SHR (sf*12)
|
|
||||||
* [IR1,IR2,IR3] = [MAC1,MAC2,MAC3] (saturated to 0x7FFF when lm=1)
|
|
||||||
* Sourced verbatim from libgte msc02 VectorNormal disassembly at 0x800160b0:
|
|
||||||
* 0x4AA00428 = gte_cmd_base | gte_cmdw_sqr_compat | enc_gte_lm(1) | enc_gte_cmd(0x28)
|
|
||||||
* bit 19 sf=0
|
|
||||||
* bit 10 lm=1
|
|
||||||
* bits 5-0 cmd=0x28=SQR
|
|
||||||
* bits 24-20 = 0x0A (libgte "nonsense SDK command number" signature) */
|
|
||||||
#define gte_cmdw_sqr (gte_cmd_base | enc_gte_cmd(gte_cmd_sqr) | enc_gte_lm(1) | gte_cmdw_sqr_fake_sig)
|
|
||||||
|
|
||||||
/* GPF — General-purpose Interpolation.
|
|
||||||
* PSX-SPX `geometrytransformationenginegte.md` §"GPF":
|
|
||||||
* [MAC1,MAC2,MAC3] = (([IR1,IR2,IR3] * IR0) + [MAC1,MAC2,MAC3]) SAR (sf * 12)
|
|
||||||
* [IR1,IR2,IR3] = [MAC1,MAC2,MAC3]
|
|
||||||
* Sourced verbatim from libgte msc02 VectorNormal disassembly at 0x8001613c:
|
|
||||||
* 0x4B90003D = gte_cmd_base | gte_cmdw_gpf_compat | enc_gte_cmd(0x3D)
|
|
||||||
* bit 19 sf = 0
|
|
||||||
* bit 10 lm = 0
|
|
||||||
* bits 5-0 cmd = 0x3D = GPF
|
|
||||||
* bits 24-20 = 0x19 (libgte "nonsense SDK command number" signature) */
|
|
||||||
#define gte_cmdw_gpf (gte_cmd_base | enc_gte_cmd(gte_cmd_gpf) | gte_cmdw_gpf_fake_sig)
|
|
||||||
|
|
||||||
/* Mask to round LZCR (leading-zero/ones count, range 1..32 per PSX-SPX cop2r31) down to even.
|
|
||||||
* The normalize_v3s4 half-shift logic computes (31 - LZCR) >> 1; clearing bit 0 ensures the subtraction result is always odd, so the >> 1 division is consistent (no 0.5 loss). */
|
|
||||||
enum {
|
|
||||||
gte_lzcr_even_mask = 0xFFFE, /* all bits except bit 0 */
|
|
||||||
};
|
|
||||||
|
|
||||||
#define gte_cmdw_rotate_translate_perspective_single gte_cmdw_rtps
|
#define gte_cmdw_rotate_translate_perspective_single gte_cmdw_rtps
|
||||||
#define gte_cmdw_rotate_translate_perspective_triple gte_cmdw_rtpt
|
#define gte_cmdw_rotate_translate_perspective_triple gte_cmdw_rtpt
|
||||||
/* RGA(Lengyel): RTPS/RTPT consume the matrix expansion of a rigid transformation (rotation matrix + translation vector) loaded into the RT/TR control registers.
|
|
||||||
* For unitized points the same result equals the motor antiproduct; the GTE executes the LA form, not a symbolic antiproduct. */
|
|
||||||
|
|
||||||
/* PsyQ compatibility bits for AVSZ3 (Bits 20, 22, 24 must be set) */
|
/* PsyQ compatibility bits for AVSZ3 (Bits 20, 22, 24 must be set) */
|
||||||
#define gte_cmdw_psyq_avsz3_compat (0x15 << 20)
|
#define gte_cmdw_psyq_avsz3_compat (0x15 << 20)
|
||||||
@@ -533,6 +433,7 @@ enum {
|
|||||||
#define gte_lw_v2_z(base) enc_gte_lw(gte_in_v2_z, (base), GTE_Z_Offset)
|
#define gte_lw_v2_z(base) enc_gte_lw(gte_in_v2_z, (base), GTE_Z_Offset)
|
||||||
|
|
||||||
/* gte_load_vN(r_ptr, base) — placeholder-punned lwc2 loaders
|
/* gte_load_vN(r_ptr, base) — placeholder-punned lwc2 loaders
|
||||||
|
*
|
||||||
* Emits `.word` constants encoding `lwc2 $N, off(<base>)` for the chosen GTE vector register, where `<base>` is the GPR number you pass in
|
* Emits `.word` constants encoding `lwc2 $N, off(<base>)` for the chosen GTE vector register, where `<base>` is the GPR number you pass in
|
||||||
* (typically one of R_T4..R_T9 for the standard "3-pointer" pattern).
|
* (typically one of R_T4..R_T9 for the standard "3-pointer" pattern).
|
||||||
*
|
*
|
||||||
@@ -581,8 +482,8 @@ enum {
|
|||||||
|
|
||||||
/* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — prelude to gte_cmd_rtpt.
|
/* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — prelude to gte_cmd_rtpt.
|
||||||
*
|
*
|
||||||
* Loads all three GTE input vectors (6 words) from three separate pointers, one per GTE vector register, each loaded from its own base GPR.
|
* Loads all three GTE input vectors (6 words) from three separate pointers, one per GTE vector register,
|
||||||
* Caller must bind each `pN` to `bN` via a register variable.
|
* each loaded from its own base GPR. Caller must bind each `pN` to `bN` via a register variable.
|
||||||
* register V3_S2* p0 rgcc(R_T4) = verts[0].ptr; // → __asm__("$12")
|
* register V3_S2* p0 rgcc(R_T4) = verts[0].ptr; // → __asm__("$12")
|
||||||
* register V3_S2* p1 rgcc(R_T5) = verts[1].ptr; // → __asm__("$13")
|
* register V3_S2* p1 rgcc(R_T5) = verts[1].ptr; // → __asm__("$13")
|
||||||
* register V3_S2* p2 rgcc(R_T6) = verts[2].ptr; // → __asm__("$14")
|
* register V3_S2* p2 rgcc(R_T6) = verts[2].ptr; // → __asm__("$14")
|
||||||
@@ -676,7 +577,8 @@ enum {
|
|||||||
* Loads the 3x3 rotation matrix at `r0` into the GTE's rotation-matrix control registers (RT11..RT22, indices 0..4) via ctc2.
|
* Loads the 3x3 rotation matrix at `r0` into the GTE's rotation-matrix control registers (RT11..RT22, indices 0..4) via ctc2.
|
||||||
*
|
*
|
||||||
* Memory layout at r0: five contiguous 32-bit words (offsets 0..16), each holding two packed 16-bit matrix elements.
|
* Memory layout at r0: five contiguous 32-bit words (offsets 0..16), each holding two packed 16-bit matrix elements.
|
||||||
* The first 1.5 rows of a standard PSX SDK MATRIX struct (where each row is laid out as [RT_xx, RT_xy] | [RT_xz, pad] | ...).
|
* The first 1.5 rows of a standard PSX SDK MATRIX struct (where each row is laid out as
|
||||||
|
* [RT_xx, RT_xy] | [RT_xz, pad] | ...).
|
||||||
*
|
*
|
||||||
* Generated MIPS (mirrors the source macro):
|
* Generated MIPS (mirrors the source macro):
|
||||||
* lw $12, 0( %0 ) ; word 0
|
* lw $12, 0( %0 ) ; word 0
|
||||||
|
|||||||
+162
-346
@@ -12,240 +12,191 @@
|
|||||||
#endif
|
#endif
|
||||||
|
|
||||||
#pragma region Tape Drive
|
#pragma region Tape Drive
|
||||||
/* -----------------------------------------------------------------------------------------------------------
|
/* -----------------------------------------------------------------------------
|
||||||
* TAPE DRIVE ABI
|
* TAPE DRIVE ABI
|
||||||
* -----------------------------------------------------------------------------------------------------------
|
* -----------------------------------------------------------------------------
|
||||||
* Note(Ed): One of the main purposes of this codebase is to help me learn this,
|
* Note(Ed): One of the main purposes of this codebase is to help me
|
||||||
* as such the information below may not* be entirely realized or finalized conceptually.
|
* learn this, as such the information below may be entirely realized
|
||||||
* -----------------------------------------------------------------------------------------------------------
|
* or finalized conceptually.
|
||||||
* This ABI and its associated legos were directly inspired by researching the work of
|
* -----------------------------------------------------------------------------
|
||||||
* Timothy Lottes and Onat Türkçüoğlu; along with many others. It's the simplest bootstrap of a
|
* This ABI and its associated legos were directly inspired by researching
|
||||||
* directly executed chain of assemby arrays (Atoms) that terminate with a yield sequence to the next atom.
|
* the work of Timothy Lottes and Onat Türkçüoğlu; along with many others.
|
||||||
* These eventually lead to a terminal atom for the tape which is defined below as "tape_exit".
|
* It's the simplest bootstrap of a a directly executed chain of assemby
|
||||||
|
* arrays (Atoms) that terminate with a yield sequence to the next atom.
|
||||||
|
* These eventually lead to a terminal atom for the tape which is defined
|
||||||
|
* below as "tape_exit".
|
||||||
*
|
*
|
||||||
* It behaves as one of the simplest runtime harnesses ontop of a host-enviornment's execution engine
|
* This behaves as one of the simplest runtime harnesses ontop of a
|
||||||
* to author and compose programs with. From here various conventions can be further applied.
|
* host-enviornment's execution engine to author and compose programs with.
|
||||||
* To make things easier to understand it may be better to focus on what this ABI does not have.
|
* From here various conventions can be further applied.
|
||||||
* It does not have have any branching within the tape but relative branches within atoms or between atoms.
|
* To make things easier to understand it may be better to focus on what this
|
||||||
* Branching nearly is always downstream. Automatic stack usage is non-existent.
|
* ABI does not have. It does not have have any branching within the tape but
|
||||||
* Push/Pop, FIFO, or Arena/Bump data structures are used by atoms explicitly.
|
* relative branches between atoms. Branching nearly is always downstream.
|
||||||
* In it's current form with the C11 macro DSL, the user also has fullfill manual register allocation per atom.
|
* Stack usage is non-existent. Push/Pop, FIFO, or Arena/Bump data structures
|
||||||
|
* are used by atoms explicitly. In it's current form withe C11 macro dsl,
|
||||||
|
* the user also has to do manual register allocation per atom.
|
||||||
*
|
*
|
||||||
* One of the remarkable things about utilizing this ABI is its essentially interopable with CPUs, GPUs, FPGA,
|
* One of the remarkable things about utilizing this abi is its essentially
|
||||||
* or, basically anything from the 5th generation consoles and onward.
|
* interopable with CPUs, GPUs, FPGA, or, basically anything
|
||||||
* The ABI directly reflects how all computational hardware must be architected in order to execute
|
* from the 5th generation consoles and onward.
|
||||||
* digital logic effectively on current era tech.
|
* The ABI directly reflects how all computational hardware must be architected
|
||||||
* On the PS1 we don't have access to a few features like multi-threading, speculative execution, or L3 cache;
|
* in order to execute digital logic effectively on current era tech.
|
||||||
* but, we can set the foundation for legoing whats required for eventually expanding this ABI's paradigm
|
* On the PS1 we don't have access to a few features like multi-threading,
|
||||||
* and core atoms to take those newer hardware features into account. For example, you can easily expand
|
* speculative execution, or L3 cache; but, we can set the foundation for legoing
|
||||||
* this to support wave-based execution model on a PS2 or PS3. Not having a stack or
|
* whats required baseline wise for eventually expanding the harness and core atoms
|
||||||
* automatic register allocation means the user cannot ignore excessive argument shuffle across workload or
|
* to take those newer hardware features into account. For example, you can easily
|
||||||
* waves and thier phases. Crossing ABI boundaries to other runtimes that do has obviouss penalties.
|
* expand this to support wave-based execution model on a PS2 or PS3.
|
||||||
|
* Not having a stack or automatic register allocation means the user can't ignore
|
||||||
|
* excessive argument shuffle across workload or waves and thier phases.
|
||||||
|
* Crossing ABI boundaries to other runtimes that do has an obviouss penalties.
|
||||||
*
|
*
|
||||||
* Learning data-oriented code becomes a natural progression. Your not fighting a stack-based procedural
|
* Learning data-oreinted code becomes a natural progression. Your not fighting
|
||||||
* paradigm that wants to argument shuffle. There is no ambiguity due to the lack of constraints, for example,
|
* a stack-based procedural paradigm that wants to argument shuffle on the stack
|
||||||
* on how the user may "call" a procedure in traditional random dispatch runtimes. The user does have to
|
* by lack of constraints on how the user may "call" a procedure. The user doesn't
|
||||||
* hammer down "rules" or patterns for massaging the compiler to dissolve those call frames; just to get
|
* have to hammer down "rules" or patterns to know how to massage the compiler
|
||||||
* the asesmbly into its desired form. The form is obvious, and once the user gets to author these compoonents
|
* to get the asesmbly into its natural form. The form is obvious, and once
|
||||||
* it becomes a game of tetris.
|
* the user gets to author their compoonents it becomes a game of tetris.
|
||||||
*
|
*
|
||||||
* Another feature is this ABI is very compatible with bootstrapping and developing simple toolchains built off
|
* Another feature is this ABI is very compatible with bootstrapping and developing
|
||||||
* of bit-packed annotated command streams the user can directly author, maintatain, and immediately execute.
|
* simple toolchains built off of bit-packed annotated command streams the user can
|
||||||
* That being like a color forth, or maybe something more familar like an immediate mode library
|
* directly author, maintatain, and immediately execute. That being a color forth.
|
||||||
* for various systems such as GUIs. This can make the tetris less of a chore with some helpful policy
|
* This can make the tetris less of a chore with some helpful policy generation for
|
||||||
* generation for allocation of registers, helping to choose resuable components, designing DSL on the fly, etc.
|
* allocation of registers, helping to choose resuable components, designing DSL on
|
||||||
* -----------------------------------------------------------------------------------------------------------
|
* the fly, etc.
|
||||||
* TODO(Ed): We need pretty ascii diagrams and proper guides, articles, etc.
|
* -----------------------------------------------------------------------------
|
||||||
* -----------------------------------------------------------------------------------------------------------
|
* TODO(Ed): We ned pretty ascii diagrams and proper guides, articles, etc.
|
||||||
* For now this ideation has just started functioning. I'm abusing C11 & a lua metaprogram to help establish
|
* -----------------------------------------------------------------------------
|
||||||
* a hybrid toolchain to ideate on a traditional text-based authoring UX for this paradigm.
|
* For now this thing is just functioning and I'm abusing C11 + a lua metaprogram
|
||||||
* If pcsx-redux provides viable hot-reload and persistent data storage beyond save-states
|
* to help establish a hybrid toolchain to ideate on a traditional text-based
|
||||||
* (just copying ram to filesystem), I can author a color forth to mess around with.
|
* authoring UX for this paradigm.
|
||||||
* With either an editor in-emulator or on the actual machine itself. Assembly is tedius,
|
* If pcsx-redux gets me viable hot-reload and persistent data storage beyond
|
||||||
* but I think this codebase most likely has a pretty ergonomic flavor worst case...
|
* save-states (just copying ram to filesystem). I can author a color forth to
|
||||||
|
* mess around with, with an editor in-emulator or on the actual machine itself.
|
||||||
|
* Assembly is tedius, but I think this codebase most likely has some of the most,
|
||||||
|
* ergonomic you can come across..
|
||||||
* */
|
* */
|
||||||
/* Register Allocation Info */
|
/* Register Allocation Info */
|
||||||
enum {
|
enum {
|
||||||
R_ScratchBase = R_SP atom_reg, /* Scratchpad base address (host frame top) */
|
R_AtomJmp = R_T8 atom_reg, /* debug-visible; tape yield handshake scratch */
|
||||||
R_AtomJmp = R_FP atom_reg, /* Next atom target (yield handshake scratch) */
|
R_TapePtr = R_T9 atom_reg, /* The Instruction Stream Pointer */
|
||||||
R_TapePtr = R_RA atom_reg, /* The Instruction Stream Pointer */
|
|
||||||
/* Stringification codes for the GCC inline assembler clobber lists. */
|
/* Stringification codes for the GCC inline assembler clobber lists. */
|
||||||
#define R_ScratchBase_Code R_SP_Code
|
#define R_AtomJmp_Code R_T8_Code
|
||||||
#define R_AtomJmp_Code R_FP_Code
|
#define R_TapePtr_Code R_T9_Code
|
||||||
#define R_TapePtr_Code R_RA_Code
|
|
||||||
|
|
||||||
// R_InCursor = R_T4,
|
// R_InCursor = R_T4,
|
||||||
// #define R_InCursor_Code R_T4_Code
|
// #define R_InCursor_Code R_T4_Code
|
||||||
|
|
||||||
// Reserved Registers (Callee-saved across the host ABI transition):
|
// Reserved Registers (Callee-saved):
|
||||||
// - R_SP: Holds the scratchpad base while tape code executes.
|
// - R_T9: Holds the Tape Ptr which we need to increment
|
||||||
// - R_FP: Holds the next atom target.
|
// If we hit a wall with register allocations we can clobber V0 & V1 (return values), defering as opt-in by user.
|
||||||
// - R_RA: Holds the tape cursor.
|
// - R_RA: Not sure??
|
||||||
// All atom-body allocations must stay out of these.
|
// Needed by ac_yield but can be used as atom scratch:
|
||||||
// Atom bodies may freely use R2-R25.
|
// - R_T8: Will be used as the atom jump register.
|
||||||
|
|
||||||
// All allocatable registers for atom bodies (R2-R25, 24 registers):
|
// All allocatable registers for mips atoms:
|
||||||
|
R_TScratchVolatile = R_AT, // This one is reserved for psuedo instructions, but you can technically use it.
|
||||||
R_PsuedoVolatile = R_AT, // Assembler temporary; never allocate.
|
R_TScratch0 = R_T0,
|
||||||
|
R_TScratch1 = R_T1,
|
||||||
// Atom Allocation Pool
|
R_TScratch2 = R_T2,
|
||||||
R_Atom0 = R_T0,
|
R_TScratch3 = R_T3,
|
||||||
R_Atom1 = R_T1,
|
R_TScratch4 = R_T4,
|
||||||
R_Atom2 = R_T2,
|
R_TScratch5 = R_T5,
|
||||||
R_Atom3 = R_T3,
|
R_TScratch6 = R_T6,
|
||||||
R_Atom4 = R_T4,
|
R_TScratch7 = R_T7,
|
||||||
R_Atom5 = R_T5,
|
R_TScratch8 = R_T8,
|
||||||
R_Atom6 = R_T6,
|
R_TScratch10 = R_V0, // Tend to be used with gte DMAs
|
||||||
R_Atom7 = R_T7,
|
R_TScratch11 = R_V1, // Tend to be used with gte DMAs
|
||||||
R_Atom8 = R_T8,
|
// Note(Ed): We can technically clobber these, but don't unless we hit a bottleneck.
|
||||||
R_Atom9 = R_T9,
|
// A 0-2
|
||||||
R_Atom10 = R_V0, // Tend to be used with gte DMAs
|
// S 0-7
|
||||||
R_Atom11 = R_V1, // Tend to be used with gte DMAs
|
|
||||||
R_Atom12 = R_A0,
|
|
||||||
R_Atom13 = R_A1,
|
|
||||||
R_Atom14 = R_A2,
|
|
||||||
R_Atom15 = R_A3,
|
|
||||||
R_Atom16 = R_S0,
|
|
||||||
R_Atom17 = R_S1,
|
|
||||||
R_Atom18 = R_S2,
|
|
||||||
R_Atom19 = R_S3,
|
|
||||||
R_Atom20 = R_S4,
|
|
||||||
R_Atom21 = R_S5,
|
|
||||||
R_Atom22 = R_S6,
|
|
||||||
R_Atom23 = R_S7,
|
|
||||||
};
|
};
|
||||||
|
|
||||||
typedef U2 Reg; // Register parameter used with atom or atom component procedures
|
|
||||||
#define Reg_(type) tmpl(Reg,type) // Just a way to template register allocations of C-struct types.
|
|
||||||
|
|
||||||
typedef U4 const MipsCode; // Underlying type to mips asm words.
|
typedef U4 const MipsCode; // Underlying type to mips asm words.
|
||||||
typedef Slice_(MipsCode);
|
typedef Slice_(MipsCode);
|
||||||
|
|
||||||
typedef U4 const MipsAtom; // Underlying type to a mips atom defnition
|
typedef U4 const MipsAtom; // Underlying type to an array of mips asm words that must terminate with an ac_yield.
|
||||||
typedef Slice_(MipsAtom);
|
|
||||||
|
|
||||||
// Sometimes a user will define a bundle of atoms that represent a procedure of work as:
|
|
||||||
// MipsAtom* <identifier>[...];
|
|
||||||
// Unfortuantely if using slice_from_array it will make the slice's pointer: MipsAtom** so this enforce its defined as MipsAtom*
|
|
||||||
// TODO(Ed): Alternatively we can make the MipsAtom an opaque pointer to the atom... so that the proc returns 'MipsAtom'.
|
|
||||||
#define atombundle_from_array(array) (Slice_MipsAtom){.ptr=array[0],.len=Array_len(array)}
|
|
||||||
|
|
||||||
// Underlying type to an ptr to an array of mips asm words that must terminate with an ac_yield.
|
|
||||||
#define MipsAtom_(sym) MipsCode sym [] align_(4) =
|
#define MipsAtom_(sym) MipsCode sym [] align_(4) =
|
||||||
|
|
||||||
// Used for atoms with value-args
|
|
||||||
// internal MipsAtom* X_proc(AtomArena_R aa, args) MipsAtom_Proc_(X, aa, { body })
|
|
||||||
// expands to:
|
|
||||||
// internal MipsAtom* X_proc(AtomArena_R aa, args) { MipsCode atom_comp_code[] align_(4) = { body }; return atomarena_push(aa, slice_from_array(MipsCode, atom_comp_code)); }
|
|
||||||
// The atom name is derived by the Lua metaprogram from the preceding
|
|
||||||
// `MipsAtom* X_proc(...)` declaration (backward walk from the macro site,
|
|
||||||
// strips the `_proc` suffix).
|
|
||||||
#define MipsAtom_Proc_(aa, ...) { MipsCode atom_comp_code[] align_(4) = __VA_ARGS__; return atomarena_push(aa, slice_from_array(MipsCode, atom_comp_code)); }
|
|
||||||
|
|
||||||
// Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names).
|
// Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names).
|
||||||
// MipsAtomComp_(ac_X) { body }
|
// MipsAtomComp_(ac_X) { body }
|
||||||
// expands to:
|
// expands to:
|
||||||
// MipsCode ac_X[] align_(4) = { body };
|
// MipsCode ac_X[] align_(4) = { body };
|
||||||
#define MipsAtomComp_(sym) MipsCode sym [] align_(4) =
|
#define MipsAtomComp_(sym) MipsCode sym [] align_(4) =
|
||||||
|
|
||||||
// Used for components with value-args (mandatory `ab` (atom-builder) arg).
|
// Used for components with value-args (e.g., ac_format_f3_color).
|
||||||
// FI_ void ac_X(MipsAtomBuilder_R ab, args) MipsAtomComp_Proc_(ab, { body })
|
// FI_ Slice_MipsCode ac_X(args) MipsAtomComp_Proc_(ac_X, { body })
|
||||||
// expands to:
|
// expands to:
|
||||||
// FI_ void ac_X(MipsAtomBuilder_R ab, args) {
|
// FI_ Slice_MipsCode ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return slice_from_array(MipsCode, ac_X); }
|
||||||
// MipsCode atom_comp_code[] align_(4) = { body };
|
#define MipsAtomComp_Proc_(sym, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; return slice_from_array(MipsCode, sym); }
|
||||||
// atombuilder_push(ab, slice_from_array(MipsCode, atom_comp_code));
|
|
||||||
// }
|
|
||||||
// The body must NOT include mac_yield() (the parent atom yields).
|
|
||||||
// The component name is derived by the Lua metaprogram from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration (backward walk from the macro site).
|
|
||||||
// Inline-only callers (the generated `mac_<name>` aliases) skip the `ab` arg via metaprogram filtering; escape callers (ac_<name> invoked as a function) pass a long-lived builder.
|
|
||||||
#define MipsAtomComp_Proc_(ab, ...) { MipsCode atom_comp_code[] align_(4) = __VA_ARGS__; atombuilder_push(ab, slice_from_array(MipsCode, atom_comp_code)); }
|
|
||||||
|
|
||||||
// Used for trivial mappings from one atom component proc to the command of a more baser (meant for type-mapping)
|
/* Line-table anchor: gcc only adds a file to the .debug_line file table when the
|
||||||
#define MipsAtomComp_ProcMap_(ab, base_command) atom_dbg_skip MipsAtomComp_Proc_(ab, {base_command })
|
file contains line-numbered content. Files containing only:
|
||||||
|
- `MipsAtomComp_` static-array declarations, or
|
||||||
|
- `MipsAtomComp_Proc_` (force-inline) function bodies whose line info gets
|
||||||
|
attributed to the call site at the include point are otherwise omitted from the file table,
|
||||||
|
which breaks the DWARF injection when it tries to resolve atom-component provenance paths.
|
||||||
|
|
||||||
/* Line-table anchor: gcc only adds a file to the .debug_line file table when the contains line-numbered content.
|
|
||||||
Files containing only atoms and atom components.
|
|
||||||
Place `ATOM_FILE_LINE_MARKER();` once at file scope in any `.atom.c` that defines atoms.
|
Place `ATOM_FILE_LINE_MARKER();` once at file scope in any `.atom.c` that defines atoms.
|
||||||
Macro expands to a file-scope `internal U4 const` declaration keeps the file in the line table.
|
The macro expands to a file-scope `internal U4 const` declaration keeps the file in the line table.
|
||||||
The constant is in `.rodata` so the linker may eliminate it. */
|
The constant is in `.rodata` and unreferenced; the linker may eliminate it.
|
||||||
|
The two-level concat + `__LINE__` suffix makes the identifier unique per call site
|
||||||
|
(the identifier embeds the source line, so duplicates across `#include`d files don't collide). */
|
||||||
#define ATOM_FILE_DEBUGGER_LINE_MARKER(file_name) internal U4 const tmpl(atom_file_debugger_line_marker,file_name) = 0
|
#define ATOM_FILE_DEBUGGER_LINE_MARKER(file_name) internal U4 const tmpl(atom_file_debugger_line_marker,file_name) = 0
|
||||||
|
|
||||||
typedef Slice_MipsAtom Tape;
|
typedef Slice_(MipsAtom); typedef Slice_MipsAtom Tape;
|
||||||
|
|
||||||
typedef Struct_(TapeHostFrame) {
|
/* The 'Exit' Atom */
|
||||||
U4 s0;
|
atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(rret_addr), nop };
|
||||||
U4 s1;
|
|
||||||
U4 s2;
|
|
||||||
U4 s3;
|
|
||||||
U4 s4;
|
|
||||||
U4 s5;
|
|
||||||
U4 s6;
|
|
||||||
U4 s7;
|
|
||||||
U4 fp;
|
|
||||||
U4 sp;
|
|
||||||
U4 ra;
|
|
||||||
};
|
|
||||||
|
|
||||||
enum {
|
// TODO(Ed): When we have a substantial workload/throughput, profile each of these to see impact at ABI boundaries.
|
||||||
TapeHostFrame_Loc = Scratchpad_End - S_(TapeHostFrame),
|
|
||||||
TapeScratch_Len = TapeHostFrame_Loc - Scratchpad_Loc,
|
|
||||||
};
|
|
||||||
static_assert(S_(TapeHostFrame) == 11 * S_(U4));
|
|
||||||
static_assert(TapeHostFrame_Loc == 0x1F8003D4);
|
|
||||||
|
|
||||||
atom_dbg_skip MipsAtom_(tape_enter) {
|
/* Tape Runner (Default) */
|
||||||
mac_load_word_imm(R_V0, u4_(TapeHostFrame_Loc)),
|
FI_ void tape_run(Tape tape) { register U4* tape_ptr rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile(
|
||||||
store_word(R_S0, R_V0, O_(TapeHostFrame,s0)),
|
asm_words(
|
||||||
store_word(R_S1, R_V0, O_(TapeHostFrame,s1)),
|
load_word( R_AtomJmp, R_TapePtr, 0) /* Bootstrap the first jump */
|
||||||
store_word(R_S2, R_V0, O_(TapeHostFrame,s2)),
|
, add_ui_self(R_TapePtr, S_(MipsAtom)) /* Advance tape */
|
||||||
store_word(R_S3, R_V0, O_(TapeHostFrame,s3)),
|
, call_reg( R_AtomJmp) /* jalr $t9 */
|
||||||
store_word(R_S4, R_V0, O_(TapeHostFrame,s4)),
|
, nop /* Branch delay slot */
|
||||||
store_word(R_S5, R_V0, O_(TapeHostFrame,s5)),
|
)
|
||||||
store_word(R_S6, R_V0, O_(TapeHostFrame,s6)),
|
asm_rpins, r_use(tape_ptr)
|
||||||
store_word(R_S7, R_V0, O_(TapeHostFrame,s7)),
|
asm_clobber:
|
||||||
store_word(R_FP, R_V0, O_(TapeHostFrame,fp)),
|
rlit(R_AT),
|
||||||
store_word(R_SP, R_V0, O_(TapeHostFrame,sp)),
|
rlit(R_V0), rlit(R_V1), // We clobber these for GTE ACs (that don't expose register selection, might expose them in the future...)
|
||||||
store_word(R_RA, R_V0, O_(TapeHostFrame,ra)),
|
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||||
add_ui(R_TapePtr, R_A0, 0),
|
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8),
|
||||||
load_upper_i(R_ScratchBase, u4_hi(Scratchpad_Loc)),
|
clb_mem_drain
|
||||||
load_word(R_AtomJmp, R_TapePtr, 0),
|
); }
|
||||||
add_ui_self( R_TapePtr, S_(MipsAtom)),
|
|
||||||
jump_reg(R_AtomJmp), BdSlot_ nop,
|
|
||||||
};
|
|
||||||
|
|
||||||
atom_dbg_skip MipsAtom_(tape_exit) {
|
/* Tape Runner (Static and Arg Clobbers) */
|
||||||
mac_load_word_imm(R_V0, u4_(TapeHostFrame_Loc)),
|
FI_ void tape_run_a02_s07(Tape tape) { register U4* tape_ptr rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile(
|
||||||
load_word(R_S0, R_V0, O_(TapeHostFrame,s0)),
|
asm_words(
|
||||||
load_word(R_S1, R_V0, O_(TapeHostFrame,s1)),
|
load_word( R_AtomJmp, R_TapePtr, 0) /* Bootstrap the first jump */
|
||||||
load_word(R_S2, R_V0, O_(TapeHostFrame,s2)),
|
, add_ui_self(R_TapePtr, S_(MipsAtom)) /* Advance tape */
|
||||||
load_word(R_S3, R_V0, O_(TapeHostFrame,s3)),
|
, call_reg( R_AtomJmp) /* jalr $t9 */
|
||||||
load_word(R_S4, R_V0, O_(TapeHostFrame,s4)),
|
, nop /* Branch delay slot */
|
||||||
load_word(R_S5, R_V0, O_(TapeHostFrame,s5)),
|
)
|
||||||
load_word(R_S6, R_V0, O_(TapeHostFrame,s6)),
|
asm_rpins, r_use(tape_ptr)
|
||||||
load_word(R_S7, R_V0, O_(TapeHostFrame,s7)),
|
asm_clobber:
|
||||||
load_word(R_RA, R_V0, O_(TapeHostFrame,ra)),
|
rlit(R_AT),
|
||||||
load_word(R_FP, R_V0, O_(TapeHostFrame,fp)),
|
rlit(R_V0), rlit(R_V1), rlit(R_A0), rlit(R_A1), rlit(R_A2),
|
||||||
load_word(R_SP, R_V0, O_(TapeHostFrame,sp)),
|
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||||
jump_reg(R_RA), BdSlot_ nop,
|
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8),
|
||||||
};
|
rlit(R_S0), rlit(R_S1), rlit(R_S2), rlit(R_S3), rlit(R_S4),
|
||||||
|
rlit(R_S5), rlit(R_S6), rlit(R_S7),
|
||||||
typedef void Proc_(TapeEntryFn)(MipsAtom* tape_ptr);
|
clb_mem_drain
|
||||||
|
); }
|
||||||
FI_ void tape_run(Tape tape) { C_(TapeEntryFn*, tape_enter)(tape.ptr); }
|
|
||||||
|
|
||||||
// Procedural authoring of tapes:
|
// Procedural authoring of tapes:
|
||||||
typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; };
|
typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; };
|
||||||
FI_ void tb_init(TapeBuilder* tb, FArena* arena) { tb->ptr = arena->start; tb->used = 0; }
|
FI_ void tb_init(TapeBuilder* tb, FArena* arena) { tb->ptr = arena->start; tb->used = 0; }
|
||||||
FI_ TapeBuilder tb_make_old( FArena* arena) { return (TapeBuilder){ arena->start, 0 }; }
|
FI_ TapeBuilder tb_make_old( FArena* arena) { return (TapeBuilder){ arena->start, 0 }; }
|
||||||
FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ u4_(mem.ptr), mem.len, 0 }; } /* capacity in elements (matches used units) */
|
FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ mem.ptr, mem.len, 0 }; }
|
||||||
|
|
||||||
FI_ void tb_emit(TapeBuilder* tb, MipsAtom* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
|
FI_ void tb_emit(TapeBuilder* tb, MipsCode* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
|
||||||
FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; }
|
FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; }
|
||||||
#define tb_emit_(atom) tb_emit(& tb, atom)
|
#define tb_emit_(atom) tb_emit(& tb, atom)
|
||||||
#define tb_data_(field, data) tb_data(& tb, u4_(data))
|
#define tb_data_(field, data) tb_data(& tb, u4_(data))
|
||||||
|
|
||||||
FI_ void tb_emit_bundle(TapeBuilder_R tb, Slice_MipsAtom atoms) { mem_copy(u4_(tb->ptr), u4_(atoms.ptr), S_slice(atoms)); tb->used += atoms.len; }
|
|
||||||
|
|
||||||
FI_ Tape tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Tape){ C_(U4*,tb->ptr), tb->used }; }
|
FI_ Tape tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Tape){ C_(U4*,tb->ptr), tb->used }; }
|
||||||
FI_ Tape tb_slice(TapeBuilder tb) { return (Tape){ C_(U4*,tb.ptr), tb.used }; }
|
FI_ Tape tb_slice(TapeBuilder tb) { return (Tape){ C_(U4*,tb.ptr), tb.used }; }
|
||||||
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit))
|
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit))
|
||||||
@@ -261,11 +212,15 @@ FI_ void tb_scope_run_end(TapeBuilder* tb) { tb_emit(tb,tape_exit); tape_run(tb_
|
|||||||
* ---------------------------------------------------------------------------*/
|
* ---------------------------------------------------------------------------*/
|
||||||
|
|
||||||
// The 'Yield' sequence for Tape Atoms (mac_yield).
|
// The 'Yield' sequence for Tape Atoms (mac_yield).
|
||||||
|
// - mac_yield() is the safe default for atom-endings: 4 words, BD-slot of jr is mandatory nop.
|
||||||
|
// - mac_yield_load() + mac_yield_tail():
|
||||||
|
// - unconditional branch: mac_yield_load fills the branch's BD-slot (replaces a nop);
|
||||||
|
// - mac_yield_tail runs at the branch target (does NOT re-load R_AtomJmp).
|
||||||
|
|
||||||
atom_dbg_skip MipsAtomComp_(ac_yield) {
|
atom_dbg_skip MipsAtomComp_(ac_yield) {
|
||||||
load_word(R_AtomJmp, R_TapePtr, 0),
|
load_word(R_AtomJmp, R_TapePtr, 0),
|
||||||
add_ui_self( R_TapePtr, S_(MipsCode)),
|
add_ui_self( R_TapePtr, S_(MipsCode)),
|
||||||
jump_reg( R_AtomJmp), BdSlot_ nop,
|
jump_reg( R_AtomJmp), nop,
|
||||||
};
|
};
|
||||||
|
|
||||||
atom_dbg_skip MipsAtomComp_(ac_yield_load) {
|
atom_dbg_skip MipsAtomComp_(ac_yield_load) {
|
||||||
@@ -274,176 +229,37 @@ atom_dbg_skip MipsAtomComp_(ac_yield_load) {
|
|||||||
|
|
||||||
atom_dbg_skip MipsAtomComp_(ac_yield_tail) {
|
atom_dbg_skip MipsAtomComp_(ac_yield_tail) {
|
||||||
add_ui_self(R_TapePtr, S_(MipsCode)),
|
add_ui_self(R_TapePtr, S_(MipsCode)),
|
||||||
jump_reg( R_AtomJmp), BdSlot_ nop,
|
jump_reg( R_AtomJmp), nop,
|
||||||
};
|
};
|
||||||
|
|
||||||
#pragma endregion Macro Atom Components
|
#pragma endregion Macro Atom Components
|
||||||
|
|
||||||
#pragma region Atom Builder
|
#pragma region Mips Atom Builder
|
||||||
// This helps with runtime procedural authoring of mips atoms.
|
// This helps with runtime procedural authoring of mips atoms.
|
||||||
typedef Relative_(FArena) Struct_(AtomBuilder) { U4 start; U4 capacity; U4 used; };
|
|
||||||
|
|
||||||
// Usual way to resolve an atom after the bulder is done.
|
typedef Struct_(FMipsAtom512) { U4 data[512]; U4 used; };
|
||||||
#define atom_from_atombuilder(ab) C_(MipsAtom*, (ab).start)
|
|
||||||
|
|
||||||
FI_ void atombuilder_push(AtomBuilder_R ab, Slice_MipsCode code) {
|
// FArena Related
|
||||||
assert(ab->capacity - ab->used - code.len);
|
typedef Relative_(FArena) Struct_(MipsAtomBuilder) { U4 start; U4 capacity; U4 used; };
|
||||||
U4 dest = ab->start + ab->used * S_(MipsCode); U4 size = S_slice(code);
|
// Whatever the builder is writting to should most likely coresspond
|
||||||
mem_copy(dest, u4_(code.ptr), size); ab->used += size;
|
// to something that can fit within instruction cache?
|
||||||
|
|
||||||
|
FI_ void atombuilder_unroll(MipsAtomBuilder_R ab, Slice_MipsCode_R code) {
|
||||||
|
assert(ab->capacity - ab->used - code->len);
|
||||||
|
mem_copy(ab->start, u4_(code->ptr), code->len);
|
||||||
|
mem_bump(ab->start, ab->capacity, & ab->used, code->len);
|
||||||
}
|
}
|
||||||
#define atombuilder_push_mac(ab, mac) atombuilder_push(ab, slice_arg_from_array(Slice_MipsCode, mac))
|
#define atombuilder_unroll_mac(ab, mac) atombuilder_unroll(ab, slice_arg_from_array(Slice_MipsCode, mac))
|
||||||
|
|
||||||
// When done authoring, utilize this to cap-off the atom (if not utilizing a MipsAtom_Proc).
|
// When done authoring, utilize this to cap-off the atom
|
||||||
FI_ void atombuilder_end(AtomBuilder_R ab) { atombuilder_push(ab, slice_from_array(MipsCode, ac_yield)); }
|
FI_ void atombuilder_end(MipsAtomBuilder_R ab) {
|
||||||
|
mem_copy(ab->start, u4_(ac_yield), S_(ac_yield));
|
||||||
|
mem_bump(ab->start, ab->capacity, & ab->used, S_(ac_yield));
|
||||||
|
}
|
||||||
|
|
||||||
FI_ void tb_emit_atombuilder(TapeBuilder_R tb, AtomBuilder_R ab) { tb_emit(tb, atom_from_atombuilder(ab[0])); }
|
#define mipsatom_from_builder(ab) (Slice_MipsCode){ab.start, ab.used}
|
||||||
#pragma endregion Mips Atom Builder
|
#pragma endregion Mips Atom Builder
|
||||||
|
|
||||||
#pragma region Atom Arena
|
|
||||||
// Just a dedicated FArena that is meant to mem_copy and return atom definitions made with MipsAtom_Proc_
|
|
||||||
typedef Relative_(FArena) Struct_(AtomArena) { U4 start; U4 capacity; U4 used; };
|
|
||||||
|
|
||||||
#define atomarena_unused_start(ab) ((ab).start + (ab).used)
|
|
||||||
FI_ void atomarena_init(AtomArena_R arena, Slice mem) { assert(arena != nullptr);
|
|
||||||
arena->start = u4_(mem.ptr);
|
|
||||||
arena->capacity = mem.len;
|
|
||||||
arena->used = 0;
|
|
||||||
}
|
|
||||||
FI_ AtomArena atomarena_make(Slice mem) { AtomArena a; atomarena_init(& a, mem); return a; }
|
|
||||||
FI_ MipsAtom* atomarena_push(AtomArena_R aa, Slice_MipsCode code) {
|
|
||||||
assert(aa->capacity - aa->used - code.len);
|
|
||||||
U4 dest = atomarena_unused_start(aa[0]); U4 size = S_slice(code);
|
|
||||||
mem_copy(dest, u4_(code.ptr), size); aa->used += size;
|
|
||||||
return C_(MipsAtom*, dest);
|
|
||||||
}
|
|
||||||
FI_ void atomarena_reset(AtomArena_R aa) { aa->used = 0; }
|
|
||||||
#pragma endregion Atom Arena
|
|
||||||
|
|
||||||
#pragma region RegFile (Register File Allocator)
|
|
||||||
// A specialized allocator utilized to help the user track which registers are bound to values
|
|
||||||
// that must be preserved for the arena's bounds.
|
|
||||||
// TODO(Ed): Technically we can do this at comp-time with the metaprogram, but we may have namespace conflicts.
|
|
||||||
// Unless we follow a convention for #define <Scope_Prefix> or something per register allocation boundary.
|
|
||||||
|
|
||||||
/* ABI reserves that are never handed out by alloc.
|
|
||||||
* R_AT is the assembler temporary (per the MIPS O32 ABI).
|
|
||||||
* R_K0/K1 are kernel reserves.
|
|
||||||
* R_GP stays the host global pointer.
|
|
||||||
* R_SP/R_FP/R_RA are tape runtime carriers between tape_enter and tape_exit. */
|
|
||||||
U4 const regfile_abi_mask =
|
|
||||||
(1u << R_0) | (1u << R_AT) |
|
|
||||||
(1u << R_K0) | (1u << R_K1) |
|
|
||||||
(1u << R_GP) | (1u << R_SP) |
|
|
||||||
(1u << R_FP) | (1u << R_RA);
|
|
||||||
|
|
||||||
internal Reg const regfile_alloc_order[] = {
|
|
||||||
R_V0, R_V1,
|
|
||||||
R_A0, R_A1, R_A2, R_A3,
|
|
||||||
R_T0, R_T1, R_T2, R_T3, R_T4, R_T5, R_T6, R_T7,
|
|
||||||
R_S0, R_S1, R_S2, R_S3, R_S4, R_S5, R_S6, R_S7,
|
|
||||||
R_T8, R_T9,
|
|
||||||
};
|
|
||||||
|
|
||||||
typedef Struct_(RegFile) {
|
|
||||||
A2_U2 GPR;
|
|
||||||
A2_U2 GTE;
|
|
||||||
};
|
|
||||||
#define regfile(pin_mask) {.GPR={u4_lo(pin_mask), u4_hi(pin_mask)} }
|
|
||||||
FI_ void regfile_init(RegFile_R rf) {
|
|
||||||
/* pack the 32-bit ABI mask into the two U2s */
|
|
||||||
rf->GPR[0] = u4_lo(regfile_abi_mask);
|
|
||||||
rf->GPR[1] = u4_hi(regfile_abi_mask);
|
|
||||||
rf->GTE[0] = rf->GTE[1] = 0;
|
|
||||||
}
|
|
||||||
FI_ RegFile regfile_make(void) { RegFile rf; regfile_init(& rf); return rf; }
|
|
||||||
|
|
||||||
typedef Struct_(RegFile_RInfo) {
|
|
||||||
U2_R section;
|
|
||||||
U2 mask;
|
|
||||||
B2 occupied;
|
|
||||||
};
|
|
||||||
FI_ RegFile_RInfo regfile_rinfo(A2_U2 file, Reg r_id) {
|
|
||||||
U2 s_id = r_id >> 4;
|
|
||||||
U2_R section = & file[s_id];
|
|
||||||
U2 mask = u2_(1u << (r_id & 15));
|
|
||||||
B2 occupied = (section[0] & mask) != 0;
|
|
||||||
return (RegFile_RInfo){section, mask, occupied};
|
|
||||||
}
|
|
||||||
FI_ Reg regfile__alloc_helper(A2_U2 file, Reg r_id) {
|
|
||||||
Reg result = 0; RegFile_RInfo info = regfile_rinfo(file, r_id);
|
|
||||||
if (info.occupied == false) {
|
|
||||||
info.section[0] |= info.mask;
|
|
||||||
result = r_id;
|
|
||||||
}
|
|
||||||
return result;
|
|
||||||
}
|
|
||||||
/* regfile_alloc picks the next free GPR from regfile_alloc_order.
|
|
||||||
* The table is the first-fit allocation order: T0..T7, V0..V1, A0..A3,
|
|
||||||
* S0..S7, T8..T9. The 24 entries leave room for the tape program to use
|
|
||||||
* any of them while R0, R1, R26-R31 remain reserved. */
|
|
||||||
I_ Reg regfile_alloc(RegFile_R rf) {
|
|
||||||
Reg allocated = 0;
|
|
||||||
for index_iter(U4, r_id, R_V0, <, R_T9) {
|
|
||||||
allocated = regfile__alloc_helper(rf->GPR, r_id);
|
|
||||||
Jmp_nZero_(allocated,resolved);
|
|
||||||
}
|
|
||||||
assert(allocated != 0);
|
|
||||||
resolved: return allocated;
|
|
||||||
}
|
|
||||||
FI_ Reg regfile_pin(RegFile_R rf, Reg r_id) {
|
|
||||||
RegFile_RInfo info = regfile_rinfo(rf->GPR, r_id);
|
|
||||||
assert(info.occupied == false);
|
|
||||||
info.section[0] |= info.mask;
|
|
||||||
return r_id;
|
|
||||||
}
|
|
||||||
FI_ void regfile_pin_mask(RegFile_R rf, U4 mask) {
|
|
||||||
B4 occupied = u4_r(rf->GPR)[0] & mask;
|
|
||||||
assert(occupied == false);
|
|
||||||
u4_r(rf->GPR)[0] |= mask;
|
|
||||||
}
|
|
||||||
FI_ void regfile_free_mask(RegFile_R rf, U4 mask) {
|
|
||||||
if (regfile_abi_mask & mask) return;
|
|
||||||
u4_r(rf->GPR)[0] &= ~mask;
|
|
||||||
}
|
|
||||||
FI_ void regfile_free_reg(RegFile_R rf, Reg r_id) {
|
|
||||||
/* never free the ABI set */
|
|
||||||
if (regfile_abi_mask & (1u << r_id)) return;
|
|
||||||
RegFile_RInfo info = regfile_rinfo(rf->GPR, r_id);
|
|
||||||
info.section[0] &= ~info.mask;
|
|
||||||
}
|
|
||||||
FI_ void regfile_reset(RegFile_R rf) {
|
|
||||||
rf->GPR[0] = u4_lo(regfile_abi_mask);
|
|
||||||
rf->GPR[1] = u4_hi(regfile_abi_mask);
|
|
||||||
}
|
|
||||||
FI_ void regfile_reset_to_mask(RegFile_R rf, U4 mask) {
|
|
||||||
rf->GPR[0] = u4_lo(mask);
|
|
||||||
rf->GPR[1] = u4_hi(mask);
|
|
||||||
}
|
|
||||||
#pragma endregion RegFileArena (Register File Allocator)
|
|
||||||
|
|
||||||
#pragma region Mips Atom Procs
|
|
||||||
/* RegUse structs are a convention to organize register allocations for a mips atom procedure.
|
|
||||||
Unlike the usual enum-based declarations, they provide a namespaced scope and have view types via union declarations. */
|
|
||||||
#define RegUse_(proc_name) (tmpl(RegUse,proc_name))
|
|
||||||
|
|
||||||
typedef Struct_(RegUse_example_atom_proc) {
|
|
||||||
Reg const ro_register; // Scratch base carrier.
|
|
||||||
Reg usual_modifiable;
|
|
||||||
union { Reg view_1, view_2, view_3; } t1;
|
|
||||||
};
|
|
||||||
internal MipsAtom* example_atom_proc(AtomArena_R aa, U2 offset, RegUse_example_atom_proc r)
|
|
||||||
MipsAtom_Proc_(aa, {
|
|
||||||
add_si(r.usual_modifiable, r.ro_register, offset),
|
|
||||||
or_u(r.t1.view_1, r.ro_register, 0),
|
|
||||||
branch_lt_zero(r.t1.view_1, atom_offset(example_atom_proc, skip)), BdSlot_ nop,
|
|
||||||
li_s(r.t1.view_2, 100),
|
|
||||||
atom_label(skip)
|
|
||||||
add_si(r.t1.view_3, r.usual_modifiable, 10),
|
|
||||||
mac_yield(),
|
|
||||||
})
|
|
||||||
|
|
||||||
#pragma endregion Mips Atom Procs
|
|
||||||
|
|
||||||
#pragma region Baked Mips Atoms
|
#pragma region Baked Mips Atoms
|
||||||
// These atoms are resolved at compile time and are (usually) statically linked readonly data.
|
// These atoms are resolved at compile time and are (usually) statically linked readonly data.
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,29 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
# include "math.h"
|
||||||
|
# include "lottes_tape.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
ATOM_FILE_DEBUGGER_LINE_MARKER(math_atom_c);
|
||||||
|
|
||||||
|
#pragma region MACs (Mips Atom Component)
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_load_v2s2(U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v2s2, {
|
||||||
|
load_half( rs_x, r_base, O_(V3_S2,x)),
|
||||||
|
load_half( rs_y, r_base, O_(V3_S2,y)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_store_v2s2(U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v2s2, {
|
||||||
|
store_half(rt_x, base, offset + O_(V2_S2,x)),
|
||||||
|
store_half(rt_y, base, offset + O_(V2_S2,y)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_store_rects2(U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rects2, {
|
||||||
|
store_half(rt_x, base, offset + O_(Rect_S2,x)),
|
||||||
|
store_half(rt_y, base, offset + O_(Rect_S2,y)),
|
||||||
|
store_half(rt_width, base, offset + O_(Rect_S2,width)),
|
||||||
|
store_half(rt_height, base, offset + O_(Rect_S2,height)),
|
||||||
|
})
|
||||||
|
|
||||||
|
#pragma endregion MACs (Mips Atom Component)
|
||||||
@@ -1,96 +0,0 @@
|
|||||||
#ifdef INTELLISENSE_DIRECTIVES
|
|
||||||
# include "gen/macs.h"
|
|
||||||
# include "gen/offsets.h"
|
|
||||||
# include "math.h"
|
|
||||||
# include "lottes_tape.h"
|
|
||||||
#endif
|
|
||||||
|
|
||||||
ATOM_FILE_DEBUGGER_LINE_MARKER(math_atom_c);
|
|
||||||
|
|
||||||
#define v3s4_R_0() ((Reg_(V3_S4)){R_0,R_0,R_0})
|
|
||||||
|
|
||||||
typedef Struct_(Reg_V3_S2) { Reg x, y, z; };
|
|
||||||
typedef Struct_(Reg_V3_S4) { Reg x, y, z; }; // Register allocation of a V3_S4
|
|
||||||
typedef Struct_(Reg_P3_S4) { Reg x, y, z; }; // Register allocation of a P3_S4
|
|
||||||
|
|
||||||
#pragma region MACs (Mips Atom Component)
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_load_half_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
load_half(tx, base, offset + OA_(U2,[0])),
|
|
||||||
load_half(ty, base, offset + OA_(U2,[1])),
|
|
||||||
load_half(tz, base, offset + OA_(U2,[2])),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_load_v3s2(AtomBuilder_R ab, Reg_(V3_S2) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_half_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_load_v2s2(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
load_half(rs_x, r_base, offset + O_(V3_S2,x)),
|
|
||||||
load_half(rs_y, r_base, offset + O_(V3_S2,y)),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_store_v2s2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
store_half(rt_x, base, offset + O_(V2_S2,x)),
|
|
||||||
store_half(rt_y, base, offset + O_(V2_S2,y)),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_load_word_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
load_word(tx, base, offset + OA_(U4,[0])),
|
|
||||||
load_word(ty, base, offset + OA_(U4,[1])),
|
|
||||||
load_word(tz, base, offset + OA_(U4,[2])),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_load_v3s4(AtomBuilder_R ab, Reg_(V3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
|
||||||
FI_ Slice_MipsCode ac_load_p3s4(AtomBuilder_R ab, Reg_(P3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_store_half_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
store_half(tx, base, offset + OA_(U2,[0])),
|
|
||||||
store_half(ty, base, offset + OA_(U2,[1])),
|
|
||||||
store_half(tz, base, offset + OA_(U2,[2])),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_store_v3s2(AtomBuilder_R ab, Reg_(V3_S2) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_store_half_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_store_word_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
store_word(tx, base, offset + OA_(U4,[0])),
|
|
||||||
store_word(ty, base, offset + OA_(U4,[1])),
|
|
||||||
store_word(tz, base, offset + OA_(U4,[2])),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_store_v3s4(AtomBuilder_R ab, Reg_(V3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
|
||||||
FI_ Slice_MipsCode ac_store_p3s4(AtomBuilder_R ab, Reg_(P3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_add_si_v3s4(AtomBuilder_R ab, Reg rt_x, Reg rt_y, Reg rt_z, Reg base, U2 offset)
|
|
||||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
add_si(rt_x, base, O_(V3_S4,x)),
|
|
||||||
add_si(rt_y, base, O_(V3_S4,y)),
|
|
||||||
add_si(rt_z, base, O_(V3_S4,z)),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_sub_s_v3(AtomBuilder_R ab
|
|
||||||
, Reg dx, Reg dy, Reg dz
|
|
||||||
, Reg sx, Reg sy, Reg sz
|
|
||||||
, Reg tx, Reg ty, Reg tz
|
|
||||||
) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
sub_s(dx, sx, tx),
|
|
||||||
sub_s(dy, sy, ty),
|
|
||||||
sub_s(dz, sz, tz),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_sub_v3s4(AtomBuilder_R ab, Reg_(V3_S4) d, Reg_(V3_S4) s, Reg_(V3_S4) t) MipsAtomComp_ProcMap_(ab, mac_sub_s_v3(d.x, d.y, d.z, s.x, s.y, s.z, t.x, t.y, t.z))
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_sub_s_v3_self(AtomBuilder_R ab, Reg ds_x, Reg ds_y, Reg ds_z, Reg tx, Reg ty, Reg tz) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
sub_s(ds_x, ds_x, tx),
|
|
||||||
sub_s(ds_y, ds_y, ty),
|
|
||||||
sub_s(ds_z, ds_z, tz),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_sub_v3s4_self(AtomBuilder_R ab, Reg_(V3_S4) ds, Reg_(V3_S4) t) MipsAtomComp_ProcMap_(ab, mac_sub_s_v3_self(ds.x, ds.y, ds.z, t.x, t.y, t.z))
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_store_rects2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
store_half(rt_x, base, offset + O_(Rect_S2,x)),
|
|
||||||
store_half(rt_y, base, offset + O_(Rect_S2,y)),
|
|
||||||
store_half(rt_width, base, offset + O_(Rect_S2,width)),
|
|
||||||
store_half(rt_height, base, offset + O_(Rect_S2,height)),
|
|
||||||
})
|
|
||||||
|
|
||||||
#pragma endregion MACs (Mips Atom Component)
|
|
||||||
+7
-63
@@ -7,24 +7,11 @@
|
|||||||
#define max(A, B) (((A) > (B)) ? (A) : (B))
|
#define max(A, B) (((A) > (B)) ? (A) : (B))
|
||||||
#define clamp_bot(X, B) max(X, B)
|
#define clamp_bot(X, B) max(X, B)
|
||||||
|
|
||||||
/* Convention
|
|
||||||
<Type> ## <Width> _ <Component Type> ## <Component Width>
|
|
||||||
For types with compound data (Ex: Rotation Matrix & Translation):
|
|
||||||
<TypeA> ## <TypeB> ## <Width> _ <ComponentTypeA> ## <ComponentWidthA> ## <ComponentTypeB> ## <ComponentWidthB>
|
|
||||||
|
|
||||||
A: Array
|
|
||||||
V: Vector
|
|
||||||
R: Range
|
|
||||||
M: Matrix
|
|
||||||
T: Translation
|
|
||||||
*/
|
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
v3s2_byteoff = 3, // log2(8), used with shift_left_logical op for index via byte offset.
|
v3s2_byteoff = 3, // log2(8), used with shift_left_logical op for index via byte offset.
|
||||||
};
|
};
|
||||||
|
|
||||||
typedef Array_(U1, 2);
|
typedef Array_(U1, 2);
|
||||||
typedef Array_(U2, 2);
|
|
||||||
typedef Array_(U4, 2);
|
typedef Array_(U4, 2);
|
||||||
typedef Array_(S2, 2);
|
typedef Array_(S2, 2);
|
||||||
typedef Array_(S2, 3);
|
typedef Array_(S2, 3);
|
||||||
@@ -39,43 +26,23 @@ typedef Struct_(Extent2_S4) { S4 width; S4 height; };
|
|||||||
typedef Struct_(V2_U1) { U1 x; U1 y; };
|
typedef Struct_(V2_U1) { U1 x; U1 y; };
|
||||||
typedef Struct_(V2_S2) { S2 x; S2 y; };
|
typedef Struct_(V2_S2) { S2 x; S2 y; };
|
||||||
typedef Struct_(V2_S4) { S4 x; S4 y; };
|
typedef Struct_(V2_S4) { S4 x; S4 y; };
|
||||||
typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; }; // PSY-Q: SVECTOR
|
typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; };
|
||||||
typedef Struct_(V3_S4) { S4 x; S4 y; S4 z; S4 pad; }; // PSY-Q: VECTOR. RGA(Lengyel): Euclidean vector or direction. A zero-weight RGA point is stored as a V3_S4 with the implicit weight dropped.
|
typedef Struct_(V3_S4) { S4 x; S4 y; S4 z; S4 pad; };
|
||||||
typedef Struct_(V4_S2) { S2 x; S2 y; S2 z; S2 w; };
|
typedef Struct_(V4_S2) { S2 x; S2 y; S2 z; S2 w; };
|
||||||
typedef Struct_(V4_S4) { S4 x; S4 y; S4 z; S4 w; };
|
typedef Struct_(V4_S4) { S4 x; S4 y; S4 z; S4 w; };
|
||||||
|
|
||||||
// typedef Struct_(P3_S4) { S4 x; S4 y; S4 z; S4 w1; }; // RGA(Lengyel): Affine point with implicit weight one. Storage alias of V3_S4. Use P3_S4 when the value is a point.
|
typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; };
|
||||||
typedef V3_S4 P3_S4;
|
typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; };
|
||||||
|
|
||||||
typedef Struct_(R1_U2) { U2 p0; U2 p1; };
|
|
||||||
typedef Struct_(R1_S2) { S2 p0; S2 p1; };
|
|
||||||
|
|
||||||
typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; }; // Range-2 Signed 2-Byte (16-bit)
|
|
||||||
typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; }; // Range-2 Signed 4-Byte (32-bit)
|
|
||||||
|
|
||||||
typedef Struct_(Rect_S2) { S2 x; S2 y; S2 width; S2 height; };
|
typedef Struct_(Rect_S2) { S2 x; S2 y; S2 width; S2 height; };
|
||||||
typedef Struct_(Rect_S4) { S4 x; S4 y; S4 width; S4 height; };
|
typedef Struct_(Rect_S4) { S4 x; S4 y; S4 width; S4 height; };
|
||||||
|
|
||||||
typedef Struct_(MT3_S2S4) { A3x3_S2 m; A3_S4 t; }; // PSY-Q: MATRIX. RGA(Lengyel): Matrix expansion of a rigid transformation. GTE utilizes this representation; corresponding motor not constructed here.
|
typedef Struct_(M3_S2) { A3x3_S2 m; A3_S4 t; };
|
||||||
|
|
||||||
/* RGA(Lengyel) reserved names (deferred):
|
|
||||||
* P4_S4 - future flat point with explicit weight (Lengyel/TML FlatPoint3D analog).
|
|
||||||
* B3_S4 - future 3D bivector (callers store a Complement(Wedge(...)) as a V3_S4).
|
|
||||||
* Mo8_S4 - future motor. Not introduced until a course operation actually needs composition, interpolation, or inversion. */
|
|
||||||
|
|
||||||
typedef Array_(V2_U1, 2);
|
|
||||||
typedef Array_(V2_S2, 2);
|
typedef Array_(V2_S2, 2);
|
||||||
typedef Array_(V2_S2, 3);
|
typedef Array_(V2_S2, 3);
|
||||||
typedef Array_(V2_S2, 4);
|
typedef Array_(V2_S2, 4);
|
||||||
|
|
||||||
#define r1u2(p0,p1) (R1_U2){p0,p1}
|
|
||||||
|
|
||||||
enum {
|
|
||||||
fp_one = (1 << 12),
|
|
||||||
};
|
|
||||||
|
|
||||||
#define v3s4_fp_one() v3s4(fp_one, fp_one, fp_one)
|
|
||||||
|
|
||||||
#define v2s2(x,y) (V2_S2){x,y}
|
#define v2s2(x,y) (V2_S2){x,y}
|
||||||
#define v3s2(x,y,z) (V3_S2){x,y,z,0}
|
#define v3s2(x,y,z) (V3_S2){x,y,z,0}
|
||||||
#define v3s4(x,y,z) (V3_S4){x,y,z,0}
|
#define v3s4(x,y,z) (V3_S4){x,y,z,0}
|
||||||
@@ -94,28 +61,5 @@ FI_ void add_a3s4_fp(A3_S4_R out_a, A3_S4 b) {
|
|||||||
(out_a[0])[2] += b[2] >> 1;
|
(out_a[0])[2] += b[2] >> 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
FI_ void sub_a3s4(A3_S4_R out_a, A3_S4 b) {
|
FI_ void add_v3s4 (V3_S4_R out_a, V3_S4 b) { add_a3s4 (pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
|
||||||
(out_a[0])[0] -= b[0];
|
FI_ void add_v3s4_fp(V3_S4_R out_a, V3_S4 b) { add_a3s4_fp(pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
|
||||||
(out_a[0])[1] -= b[1];
|
|
||||||
(out_a[0])[2] -= b[2];
|
|
||||||
}
|
|
||||||
|
|
||||||
FI_ void sub_a3s4_fp(A3_S4_R out_a, A3_S4 b) {
|
|
||||||
(out_a[0])[0] -= b[0] >> 1;
|
|
||||||
(out_a[0])[1] -= b[1] >> 1;
|
|
||||||
(out_a[0])[2] -= b[2] >> 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
FI_ void mul_a3s4(A3_S4_R out_a, A3_S4 b) {
|
|
||||||
(out_a[0])[0] *= b[0];
|
|
||||||
(out_a[0])[1] *= b[1];
|
|
||||||
(out_a[0])[2] *= b[2];
|
|
||||||
}
|
|
||||||
|
|
||||||
FI_ void add_v3s4 (V3_S4_R out_a, V3_S4 b) { add_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
|
||||||
FI_ void add_v3s4_fp(V3_S4_R out_a, V3_S4 b) { add_a3s4_fp(C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
|
||||||
|
|
||||||
FI_ void sub_v3s4 (V3_S4_R out_a, V3_S4 b) { sub_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
|
||||||
FI_ void sub_v3s4_fp(V3_S4_R out_a, V3_S4 b) { sub_a3s4_fp(C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
|
||||||
|
|
||||||
FI_ void mul_v3s4 (V3_S4_R out_a, V3_S4 b) { mul_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
|
||||||
|
|||||||
+13
-34
@@ -18,7 +18,7 @@ I_ U4 align_pow2(U4 x, U4 b) {
|
|||||||
|
|
||||||
#define align_struct(type_width) ((U4)(((type_width) + 3) & ~3))
|
#define align_struct(type_width) ((U4)(((type_width) + 3) & ~3))
|
||||||
|
|
||||||
FI_ void mem_bump(U4 cap, U4*R_ used, U4 amount) {
|
FI_ void mem_bump(U4 start, U4 cap, U4*R_ used, U4 amount) {
|
||||||
assert(amount <= (cap - used[0]));
|
assert(amount <= (cap - used[0]));
|
||||||
used[0] += amount;
|
used[0] += amount;
|
||||||
}
|
}
|
||||||
@@ -58,13 +58,13 @@ typedef Struct_(Str8) { UTF8* ptr; U4 len; };
|
|||||||
typedef Struct_(Slice_Str8) { Str8* ptr; U4 len; };
|
typedef Struct_(Slice_Str8) { Str8* ptr; U4 len; };
|
||||||
#define slit(string_literal) (Str8){ (UTF8*) string_literal, S_(string_literal) - 1 }
|
#define slit(string_literal) (Str8){ (UTF8*) string_literal, S_(string_literal) - 1 }
|
||||||
|
|
||||||
typedef Struct_(Slice) { B1* ptr; U4 len; }; // Untyped Slice (byte-addressable; .len in elements)
|
typedef Struct_(Slice) { U4 ptr, len; }; // Untyped Slice
|
||||||
FI_ Slice slice_ut_(U4 ptr, U4 len) { return (Slice){(B1*)ptr, len}; }
|
FI_ Slice slice_ut_(U4 ptr, U4 len) { return (Slice){ptr, len}; }
|
||||||
|
|
||||||
#define Slice_(type) Struct_(tmpl(Slice,type)) { type* ptr; U4 len; }
|
#define Slice_(type) Struct_(tmpl(Slice,type)) { type* ptr; U4 len; }
|
||||||
typedef Slice_(B1);
|
typedef Slice_(B1);
|
||||||
#define slice_assert(s) do { assert((s).ptr != 0); assert((s).len > 0); } while(0)
|
#define slice_assert(s) do { assert((s).ptr != 0); assert((s).len > 0); } while(0)
|
||||||
#define slice_end(slice) ((slice).ptr + S_slice(slice) / S_(B1)) /* byte-ptr arithmetic; .len is in elements per slice convention */
|
#define slice_end(slice) ((slice).ptr + (slice).len)
|
||||||
#define S_slice(s) ((s).len * S_((s).ptr[0]))
|
#define S_slice(s) ((s).len * S_((s).ptr[0]))
|
||||||
|
|
||||||
#define slice_ut(ptr,len) slice_ut_(u4_(ptr), u4_(len))
|
#define slice_ut(ptr,len) slice_ut_(u4_(ptr), u4_(len))
|
||||||
@@ -72,30 +72,23 @@ typedef Slice_(B1);
|
|||||||
#define slice_to_ut(s) slice_ut_(u4_((s).ptr), S_slice(s))
|
#define slice_to_ut(s) slice_ut_(u4_((s).ptr), S_slice(s))
|
||||||
|
|
||||||
#define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter)
|
#define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter)
|
||||||
#define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = Array_decl(type,__VA_ARGS__), .len = Array_len( Array_decl(type,__VA_ARGS__)) }
|
#define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = array_decl(type,__VA_ARGS__), .len = array_len( array_decl(type,__VA_ARGS__)) }
|
||||||
#define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = Array_len(array) }
|
#define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = S_(array) }
|
||||||
|
|
||||||
FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(u4_(s.ptr), s.len); }
|
FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(s.ptr, s.len); }
|
||||||
#define slice_zero(s) slice_zero_(slice_to_ut(s))
|
#define slice_zero(s) slice_zero_(slice_to_ut(s))
|
||||||
|
|
||||||
FI_ void slice_copy_(Slice dest, Slice src) {
|
FI_ void slice_copy_(Slice dest, Slice src) {
|
||||||
assert(S_slice(dest) >= S_slice(src));
|
assert(dest.len >= src.len);
|
||||||
slice_assert(dest);
|
slice_assert(dest);
|
||||||
slice_assert(src);
|
slice_assert(src);
|
||||||
mem_copy(u4_(dest.ptr), u4_(src.ptr), S_slice(src));
|
mem_copy(dest.ptr, src.ptr, src.len);
|
||||||
}
|
}
|
||||||
#define slice_copy(dest, src) do { \
|
#define slice_copy(dest, src) do { \
|
||||||
static_assert(T_same(dest, src)); \
|
static_assert(T_same(dest, src)); \
|
||||||
slice_copy_(slice_to_ut(dest), slice_to_ut(src)); \
|
slice_copy_(slice_to_ut(dest), slice_to_ut(src)); \
|
||||||
} while(0)
|
} while(0)
|
||||||
|
|
||||||
FI_ Slice slice_bump(U4_R used, U4 start, U4 len, U4 amount) {
|
|
||||||
assert(len - used[0] - amount);
|
|
||||||
U4 ptr = start + used[0]; used[0] += amount;
|
|
||||||
return slice_ut(ptr, amount);
|
|
||||||
}
|
|
||||||
|
|
||||||
typedef Slice_(U1);
|
|
||||||
typedef Slice_(U4);
|
typedef Slice_(U4);
|
||||||
|
|
||||||
#pragma endregion Slice
|
#pragma endregion Slice
|
||||||
@@ -105,19 +98,18 @@ typedef Slice_(U4);
|
|||||||
typedef Opt_(farena) { U4 alignment, type_width; };
|
typedef Opt_(farena) { U4 alignment, type_width; };
|
||||||
typedef Struct_(FArena) { U4 start, capacity, used; };
|
typedef Struct_(FArena) { U4 start, capacity, used; };
|
||||||
FI_ void farena_init(FArena_R arena, Slice mem) { assert(arena != nullptr);
|
FI_ void farena_init(FArena_R arena, Slice mem) { assert(arena != nullptr);
|
||||||
arena->start = u4_(mem.ptr);
|
arena->start = mem.ptr;
|
||||||
arena->capacity = mem.len;
|
arena->capacity = mem.len;
|
||||||
arena->used = 0;
|
arena->used = 0;
|
||||||
}
|
}
|
||||||
FI_ FArena farena_make(Slice mem) { FArena a; farena_init(& a, mem); return a; }
|
FI_ FArena farena_make(Slice mem) { FArena a; farena_init(& a, mem); return a; }
|
||||||
FI_ Slice farena_bump(FArena_R a, U4 amount) { return slice_bump(& a->used, a->start, a->capacity, amount); }
|
I_ Slice farena_push(FArena_R arena, U4 amount, Opt_farena o) {
|
||||||
I_ Slice farena_push(FArena_R arena, U4 amount, Opt_farena o) {
|
|
||||||
if (amount == 0) { return (Slice){}; }
|
if (amount == 0) { return (Slice){}; }
|
||||||
U4 desired = amount * (o.type_width == 0 ? 1 : o.type_width);
|
U4 desired = amount * (o.type_width == 0 ? 1 : o.type_width);
|
||||||
U4 to_commit = align_pow2(desired, o.alignment ? o.alignment : MEM_ALIGNMENT_DEFAULT);
|
U4 to_commit = align_pow2(desired, o.alignment ? o.alignment : MEM_ALIGNMENT_DEFAULT);
|
||||||
U4 ptr = arena->start + arena->used;
|
U4 ptr = arena->start + arena->used;
|
||||||
mem_bump(arena->capacity, & arena->used, to_commit);
|
mem_bump(arena->start, arena->capacity, & arena->used, to_commit);
|
||||||
return (Slice){ (B1*)ptr, to_commit };
|
return (Slice){ ptr, to_commit };
|
||||||
}
|
}
|
||||||
FI_ void farena_reset (FArena_R arena) { arena->used = 0; }
|
FI_ void farena_reset (FArena_R arena) { arena->used = 0; }
|
||||||
FI_ void farena_rewind(FArena_R arena, U4 save_point) {
|
FI_ void farena_rewind(FArena_R arena, U4 save_point) {
|
||||||
@@ -125,21 +117,8 @@ FI_ void farena_rewind(FArena_R arena, U4 save_point) {
|
|||||||
arena->used -= save_point - arena->start;
|
arena->used -= save_point - arena->start;
|
||||||
}
|
}
|
||||||
FI_ U4 farena_save(FArena arena) { return arena.used; }
|
FI_ U4 farena_save(FArena arena) { return arena.used; }
|
||||||
FI_ U4 farena_unused_start(FArena arena) { return arena.start + arena.used; }
|
|
||||||
#define farena_push_(arena, amount, ...) farena_push((arena), (amount), opt_(farena, __VA_ARGS__))
|
#define farena_push_(arena, amount, ...) farena_push((arena), (amount), opt_(farena, __VA_ARGS__))
|
||||||
#define farena_push_type(arena, type, ...) C_(type*, farena_push((arena), 1, opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr)
|
#define farena_push_type(arena, type, ...) C_(type*, farena_push((arena), 1, opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr)
|
||||||
#define farena_push_array(arena, type, amount, ...) (tmpl(Slice,type)){ C_(type*, farena_push((arena), (amount), opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr), (amount) }
|
#define farena_push_array(arena, type, amount, ...) (tmpl(Slice,type)){ C_(type*, farena_push((arena), (amount), opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr), (amount) }
|
||||||
|
|
||||||
#pragma endregion FArena
|
#pragma endregion FArena
|
||||||
|
|
||||||
#pragma region BIOS Scratchpad
|
|
||||||
/* BIOS scratchpad location. 1 KB at 0x1F800000.
|
|
||||||
* TapeHostFrame occupies the final 44 bytes while tape code executes.
|
|
||||||
* Atom scratch is bounded by the TapeHostFrame_Loc declaration in lottes_tape.h. */
|
|
||||||
enum {
|
|
||||||
Scratchpad_Loc = 0x1F800000,
|
|
||||||
Scratchpad_Len = 0x400, /* 1 KB */
|
|
||||||
Scratchpad_End = Scratchpad_Loc + Scratchpad_Len, /* 0x1F800400 */
|
|
||||||
};
|
|
||||||
#define C_scratch(type) C_(type, Scratchpad_Loc)
|
|
||||||
#pragma endregion BIOS Scratchpad
|
|
||||||
|
|||||||
+15
-50
@@ -1,53 +1,18 @@
|
|||||||
#ifdef INTELLISENSE_DIRECTIVES
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
# include "gen/macs.h"
|
# include "gen/macs.h"
|
||||||
# include "gen/offsets.h"
|
# include "gen/offsets.h"
|
||||||
# include "bios.h"
|
|
||||||
# include "mips.h"
|
|
||||||
# include "lottes_tape.h"
|
# include "lottes_tape.h"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
ATOM_FILE_DEBUGGER_LINE_MARKER(mips_atom_c);
|
ATOM_FILE_DEBUGGER_LINE_MARKER(mips_atom_c);
|
||||||
|
|
||||||
#pragma region MACs (Mips Atom Components)
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_load_word_imm(AtomBuilder_R ab, Reg dst, U4 imm)
|
|
||||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
load_upper_i(dst, u4_hi(imm)),
|
|
||||||
or_i_self( dst, u4_lo(imm)),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_shift_aright_v3_self(AtomBuilder_R ab, Reg dt_x, Reg dt_y, Reg dt_z, U2 shift_amount)
|
|
||||||
MipsAtomComp_Proc_( ab, {
|
|
||||||
shift_aright(dt_x, dt_x, shift_amount),
|
|
||||||
shift_aright(dt_y, dt_y, shift_amount),
|
|
||||||
shift_aright(dt_z, dt_z, shift_amount),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_shift_aright_v3s4_self(AtomBuilder_R ab, Reg_(V3_S4) dt, U2 shift) MipsAtomComp_ProcMap_(ab, mac_shift_aright_v3_self(dt.x, dt.y, dt.z, shift))
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_shift_aright_var_v3(AtomBuilder_R ab
|
|
||||||
, Reg rd_v0, Reg rd_v1, Reg rd_v2
|
|
||||||
, Reg rs_v0, Reg rs_v1, Reg rs_v2
|
|
||||||
, Reg r_shift)
|
|
||||||
MipsAtomComp_Proc_(ab, {
|
|
||||||
shift_aright_var(rd_v0, rs_v0, r_shift),
|
|
||||||
shift_aright_var(rd_v1, rs_v1, r_shift),
|
|
||||||
shift_aright_var(rd_v2, rs_v2, r_shift),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_shift_aright_var_v3_self(AtomBuilder_R ab, Reg rds_v0, Reg rds_v1, Reg rds_v2, Reg r_shift)
|
|
||||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
shift_aright_var(rds_v0, rds_v0, r_shift),
|
|
||||||
shift_aright_var(rds_v1, rds_v1, r_shift),
|
|
||||||
shift_aright_var(rds_v2, rds_v2, r_shift),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_shift_aright_var_v3s4_self(AtomBuilder_R ab, Reg_(V3_S4) ds, Reg shift) MipsAtomComp_ProcMap_(ab, mac_shift_aright_var_v3_self(ds.x, ds.y, ds.z, shift))
|
|
||||||
|
|
||||||
#pragma endregion MACs (Mips Atom Components)
|
|
||||||
|
|
||||||
#pragma region Baked Atoms
|
#pragma region Baked Atoms
|
||||||
|
|
||||||
|
enum {
|
||||||
|
bios_flushcache = 0x44,
|
||||||
|
bios_table_addr = 0xA0,
|
||||||
|
};
|
||||||
|
|
||||||
/* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0).
|
/* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0).
|
||||||
* Sequence (per MIPS ABI; arguments in arg registers, RA pushed to stack):
|
* Sequence (per MIPS ABI; arguments in arg registers, RA pushed to stack):
|
||||||
* 1. sp -= 8; sw $ra, 4($sp) ; save RA
|
* 1. sp -= 8; sw $ra, 4($sp) ; save RA
|
||||||
@@ -59,14 +24,14 @@ FI_ Slice_MipsCode ac_shift_aright_var_v3s4_self(AtomBuilder_R ab, Reg_(V3_S4) d
|
|||||||
* 6. sp += 8
|
* 6. sp += 8
|
||||||
*/
|
*/
|
||||||
internal MipsAtom_(mips_flush_icache) {
|
internal MipsAtom_(mips_flush_icache) {
|
||||||
add_ui(R_SP, R_SP, -MipsStackAlignment), // sp -= 8
|
add_ui(rstack_ptr, rstack_ptr, -MipsStackAlignment), // sp -= 8
|
||||||
store_word(R_RA, R_SP, S_(U4)), // sw $ra, 4($sp)
|
store_word(rret_addr, rstack_ptr, S_(U4)), // sw $ra, 4($sp)
|
||||||
add_ui(R_V0, R_0, bios_flushcache), // addiu $a0, $0, 0x44
|
add_ui(rret_0, rdiscard, bios_flushcache), // addiu $a0, $0, 0x44
|
||||||
add_ui(R_T0, R_0, bios_table_addr), // addiu $t0, $0, 0xA0
|
add_ui(rtmp_0, rdiscard, bios_table_addr), // addiu $t0, $0, 0xA0
|
||||||
jump_link(R_T0, R_RA), nop, // jalr $t0, $ra, BD slot
|
jump_link(rtmp_0, rret_addr), nop, // jalr $t0, $ra, BD slot
|
||||||
load_word(R_RA, R_SP, S_(U4)), // lw $ra, 4($sp)
|
load_word(rret_addr, rstack_ptr, S_(U4)), // lw $ra, 4($sp)
|
||||||
jump_reg(R_RA), // jr $ra
|
jump_reg(rret_addr), // jr $ra
|
||||||
add_ui(R_SP, R_SP, MipsStackAlignment), // sp += 8 (BD)
|
add_ui(rstack_ptr, rstack_ptr, MipsStackAlignment), // sp += 8 (BD)
|
||||||
mac_yield(),
|
mac_yield(),
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|||||||
+49
-58
@@ -136,31 +136,31 @@ enum {
|
|||||||
|
|
||||||
/* Semantic Aliases for MIPS Registers (O32 ABI) */
|
/* Semantic Aliases for MIPS Registers (O32 ABI) */
|
||||||
|
|
||||||
// , rdiscard = R_0 /* Hardwired to 0 */
|
, rdiscard = R_0 /* Hardwired to 0 */
|
||||||
// , rasm_tmp = R_AT /* Assembler temporary (destroyed by some assembler pseudoinstructions!) */
|
, rasm_tmp = R_AT /* Assembler temporary (destroyed by some assembler pseudoinstructions!) */
|
||||||
// , rret_0 = R_V0 /* Function return value */
|
, rret_0 = R_V0 /* Function return value */
|
||||||
// , rret_1 = R_V1 /* Second return value (e.g., 64-bit) */
|
, rret_1 = R_V1 /* Second return value (e.g., 64-bit) */
|
||||||
// , rarg_0 = R_A0 /* First function argument */
|
, rarg_0 = R_A0 /* First function argument */
|
||||||
// , rarg_1 = R_A1 /* Second function argument */
|
, rarg_1 = R_A1 /* Second function argument */
|
||||||
// , rarg_2 = R_A2 /* Third function argument */
|
, rarg_2 = R_A2 /* Third function argument */
|
||||||
// , rarg_3 = R_A3 /* Fourth function argument */
|
, rarg_3 = R_A3 /* Fourth function argument */
|
||||||
// , rtmp_0 = R_T0 /* Temporary (Caller saved) */
|
, rtmp_0 = R_T0 /* Temporary (Caller saved) */
|
||||||
// , rtmp_1 = R_T1 /* Temporary (Caller saved) */
|
, rtmp_1 = R_T1 /* Temporary (Caller saved) */
|
||||||
// , rtmp_2 = R_T2 /* Temporary (Caller saved) */
|
, rtmp_2 = R_T2 /* Temporary (Caller saved) */
|
||||||
// , rtmp_3 = R_T3 /* Temporary (Caller saved) */
|
, rtmp_3 = R_T3 /* Temporary (Caller saved) */
|
||||||
// , rtmp_4 = R_T4 /* Temporary (Caller saved) — common GTE base pointer */
|
, rtmp_4 = R_T4 /* Temporary (Caller saved) — common GTE base pointer */
|
||||||
// , rtmp_9 = R_T9 /* Temporary (Caller saved) — common GTE base pointer */
|
, rtmp_9 = R_T9 /* Temporary (Caller saved) — common GTE base pointer */
|
||||||
// , rstatic_0 = R_S0 /* Static (Callee saved, preserved across calls) */
|
, rstatic_0 = R_S0 /* Static (Callee saved, preserved across calls) */
|
||||||
// , rstatic_1 = R_S1
|
, rstatic_1 = R_S1
|
||||||
// , rstatic_2 = R_S2
|
, rstatic_2 = R_S2
|
||||||
// , rstatic_3 = R_S3
|
, rstatic_3 = R_S3
|
||||||
// , rstatic_4 = R_S4
|
, rstatic_4 = R_S4
|
||||||
// , rstatic_5 = R_S5
|
, rstatic_5 = R_S5
|
||||||
// , rstatic_6 = R_S6
|
, rstatic_6 = R_S6
|
||||||
// , rstatic_7 = R_S7
|
, rstatic_7 = R_S7
|
||||||
// , rsaved_0 = R_S0 /* Alias for rstatic_0 (alternate vocabulary) */
|
, rsaved_0 = R_S0 /* Alias for rstatic_0 (alternate vocabulary) */
|
||||||
// , rstack_ptr = R_SP /* Stack Pointer */
|
, rstack_ptr = R_SP /* Stack Pointer */
|
||||||
// , rret_addr = R_RA /* Return Address (populated by JAL) */
|
, rret_addr = R_RA /* Return Address (populated by JAL) */
|
||||||
|
|
||||||
/* --- MIPS CPU Opcodes (Bits 31-26) --- */
|
/* --- MIPS CPU Opcodes (Bits 31-26) --- */
|
||||||
|
|
||||||
@@ -259,22 +259,22 @@ enum { _BitOffsets = 0
|
|||||||
, SHAMT_SHIFT = 6 /* Shift Amount */
|
, SHAMT_SHIFT = 6 /* Shift Amount */
|
||||||
, FC_SHIFT = 0
|
, FC_SHIFT = 0
|
||||||
|
|
||||||
/* IMM_MASK is the 16-bit two's-complement truncation for the immediate field.
|
/* Bit Masks to prevent overflow into adjacent fields */
|
||||||
* It is NOT a range guard — it is load-bearing for negative branch offsets
|
|
||||||
* (the metaprogram emits raw signed offsets; the mask truncates them to the
|
|
||||||
* 16-bit representation the hardware expects). The static analysis
|
|
||||||
* `immediate_field_width` check validates ranges at build time. */
|
|
||||||
|
|
||||||
|
, OPCODE_MASK = 0x3F
|
||||||
|
, REG_MASK = 0x1F
|
||||||
|
, SHAMT_MASK = 0x1F /* Shift Amount */
|
||||||
|
, FC_MASK = 0x3F
|
||||||
, IMM_MASK = 0xFFFF
|
, IMM_MASK = 0xFFFF
|
||||||
};
|
};
|
||||||
|
|
||||||
#define enc_op(op) ((op) << OPCODE_SHIFT)
|
#define enc_op(op) (((op) & OPCODE_MASK) << OPCODE_SHIFT)
|
||||||
#define enc_rs(rs) ((rs) << RS_SHIFT)
|
#define enc_rs(rs) (((rs) & REG_MASK) << RS_SHIFT)
|
||||||
#define enc_rt(rt) ((rt) << RT_SHIFT)
|
#define enc_rt(rt) (((rt) & REG_MASK) << RT_SHIFT)
|
||||||
#define enc_rd(rd) ((rd) << RD_SHIFT)
|
#define enc_rd(rd) (((rd) & REG_MASK) << RD_SHIFT)
|
||||||
#define enc_shamt(shamt) ((shamt) << SHAMT_SHIFT)
|
#define enc_shamt(shamt) (((shamt) & SHAMT_MASK) << SHAMT_SHIFT)
|
||||||
#define enc_fc(fc) ((fc) << FC_SHIFT)
|
#define enc_fc(fc) (((fc) & FC_MASK) << FC_SHIFT)
|
||||||
#define enc_imm(imm) ((imm) & IMM_MASK)
|
#define enc_imm(imm) (((imm) & IMM_MASK))
|
||||||
|
|
||||||
/* MIPS R-Type Instruction Format (Register-to-Register) */
|
/* MIPS R-Type Instruction Format (Register-to-Register) */
|
||||||
#define enc_r(op, rs, rt, rd, shamt, fc) (enc_op(op) | enc_rs(rs) | enc_rt(rt) | enc_rd(rd) | enc_shamt(shamt) | enc_fc(fc))
|
#define enc_r(op, rs, rt, rd, shamt, fc) (enc_op(op) | enc_rs(rs) | enc_rt(rt) | enc_rd(rd) | enc_shamt(shamt) | enc_fc(fc))
|
||||||
@@ -318,10 +318,7 @@ enum { _BitOffsets = 0
|
|||||||
#define load_half(rt, base, off) enc_i(op_lh, (base), (rt), (off))
|
#define load_half(rt, base, off) enc_i(op_lh, (base), (rt), (off))
|
||||||
#define load_byte_u(rt, base, off) enc_i(op_lbu, (base), (rt), (off))
|
#define load_byte_u(rt, base, off) enc_i(op_lbu, (base), (rt), (off))
|
||||||
#define load_half_u(rt, base, off) enc_i(op_lhu, (base), (rt), (off))
|
#define load_half_u(rt, base, off) enc_i(op_lhu, (base), (rt), (off))
|
||||||
#define LdSlot_
|
|
||||||
|
|
||||||
#define store_word(rt, base, off) enc_i(op_sw, (base), (rt), (off))
|
#define store_word(rt, base, off) enc_i(op_sw, (base), (rt), (off))
|
||||||
|
|
||||||
#define add_ui(rt, rs, imm) enc_i(op_addiu, (rs), (rt), (imm))
|
#define add_ui(rt, rs, imm) enc_i(op_addiu, (rs), (rt), (imm))
|
||||||
#define and_i(rt, rs, imm) enc_i(op_andi, (rs), (rt), (imm))
|
#define and_i(rt, rs, imm) enc_i(op_andi, (rs), (rt), (imm))
|
||||||
// #define and_si and_i
|
// #define and_si and_i
|
||||||
@@ -351,12 +348,6 @@ enum { _BitOffsets = 0
|
|||||||
#define shift_lright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_srl)
|
#define shift_lright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_srl)
|
||||||
#define shift_aright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_sra)
|
#define shift_aright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_sra)
|
||||||
|
|
||||||
/* Shift Variable — register-shift forms.
|
|
||||||
* shift_lleft_var(rd, rt, rs) → sllv rd, rt, rs (shamt in low 5 bits of rs)
|
|
||||||
* shift_aright_var(rd, rt, rs) → srav rd, rt, rs */
|
|
||||||
#define shift_lleft_var(rd, rt, rs) enc_r(op_special, (rs), (rt), (rd), 0, fc_sllv)
|
|
||||||
#define shift_aright_var(rd, rt, rs) enc_r(op_special, (rs), (rt), (rd), 0, fc_srav)
|
|
||||||
|
|
||||||
#define shift_lleft_self(rd_rt, shamt) enc_r(op_special, R_0, (rd_rt), (rd_rt), (shamt), fc_sll)
|
#define shift_lleft_self(rd_rt, shamt) enc_r(op_special, R_0, (rd_rt), (rd_rt), (shamt), fc_sll)
|
||||||
|
|
||||||
#define mask_upper(rd, rt, shamt) shift_lleft(rd, rt, shamt), shift_lright(rd, rt, shamt)
|
#define mask_upper(rd, rt, shamt) shift_lleft(rd, rt, shamt), shift_lright(rd, rt, shamt)
|
||||||
@@ -375,21 +366,20 @@ enum { _BitOffsets = 0
|
|||||||
* WARNING: `jump(off)` CANNOT BE USED for within-atom jumps in the current pipeline.
|
* WARNING: `jump(off)` CANNOT BE USED for within-atom jumps in the current pipeline.
|
||||||
* The MIPS j opcode encodes `(target_addr >> 2)` in its 26-bit immediate field; an ABSOLUTE byte address, not a relative word offset.
|
* The MIPS j opcode encodes `(target_addr >> 2)` in its 26-bit immediate field; an ABSOLUTE byte address, not a relative word offset.
|
||||||
* The metaprogram computes `off` as a relative word offset (`target_word_idx - branch_word_idx - 1`), which the assembler/linker does NOT resolve.
|
* The metaprogram computes `off` as a relative word offset (`target_word_idx - branch_word_idx - 1`), which the assembler/linker does NOT resolve.
|
||||||
|
*
|
||||||
* `jump(off)` is only safe when the BUILD PIPELINE owns the absolute position of the emitted code — i.e. when: s
|
* `jump(off)` is only safe when the BUILD PIPELINE owns the absolute position of the emitted code — i.e. when: s
|
||||||
* - the build emits a symbol-relative `.word` expression that the linker resolvess via `R_MIPS_26`, OR
|
* - the build emits a symbol-relative `.word` expression that the linker resolvess via `R_MIPS_26`, OR
|
||||||
* - the code is hand-assembled with explicit absolute targets, OR a custom post-build patcher resolves the 26-bit field.
|
* - the code is hand-assembled with explicit absolute targets, OR a custom post-build patcher resolves the 26-bit field.
|
||||||
* TODO(Ed): Review this.. technically we can resolve aboslute jumps on baked atoms? (Even proedurally generated ones...)
|
|
||||||
*/
|
*/
|
||||||
#define jump(off) enc_i(op_j, R_0, R_0, (off))
|
#define jump(off) enc_i(op_j, R_0, R_0, (off))
|
||||||
|
|
||||||
// Annotate an instruction as filling a branch-delay slot.
|
|
||||||
#define BdSlot_
|
|
||||||
|
|
||||||
/* jump_rel off — unconditional relative jump (the within-atom-safe `jump`).
|
/* jump_rel off — unconditional relative jump (the within-atom-safe `jump`).
|
||||||
* MIPS I R3000A has no "branch always" opcode. The idiom for an unconditional relative jump is `beq $0, $0, off`. */
|
* MIPS I R3000A has no "branch always" opcode. The idiom for an unconditional relative jump is `beq $0, $0, off`.
|
||||||
|
*/
|
||||||
#define jump_rel(off) branch_equal(R_0, R_0, (off))
|
#define jump_rel(off) branch_equal(R_0, R_0, (off))
|
||||||
|
|
||||||
/* call_addr off — jump-and-link to immediate address.
|
/* call_addr off — jump-and-link to immediate address.
|
||||||
|
*
|
||||||
* Same WARNING as `jump(off)` above: the jal opcode also encodes an absolute 26-bit target.
|
* Same WARNING as `jump(off)` above: the jal opcode also encodes an absolute 26-bit target.
|
||||||
* For within-atom calls, the current pipeline has no equivalent always-taken call-and-link idiom.
|
* For within-atom calls, the current pipeline has no equivalent always-taken call-and-link idiom.
|
||||||
* Workaround: `branch_link` (always-taken branch + explicit `la $ra, next_word_addr; jr $ra`), or just use `call_reg($tmp)` after loading the target into a register.
|
* Workaround: `branch_link` (always-taken branch + explicit `la $ra, next_word_addr; jr $ra`), or just use `call_reg($tmp)` after loading the target into a register.
|
||||||
@@ -407,7 +397,13 @@ enum { _BitOffsets = 0
|
|||||||
* sub_s / sub_u → sub / subu
|
* sub_s / sub_u → sub / subu
|
||||||
* mult_s / mult_u → mult / multu (writes HI/LO; result in LO)
|
* mult_s / mult_u → mult / multu (writes HI/LO; result in LO)
|
||||||
* div_s / div_u → div / divu (LO = quot, HI = rem)
|
* div_s / div_u → div / divu (LO = quot, HI = rem)
|
||||||
*/
|
*
|
||||||
|
* NOTE: dsl.h defines `add_s`/`sub_s`/`mut_s`/`gt_s`/etc. as _Generic-based signed integer-arithmetic helpers for U1/U2/U4.
|
||||||
|
* Those live in a different conceptual layer (generic arithmetic on DSL types) and would collide with the instruction encoders here.
|
||||||
|
* The `#undef` below lets the gas-style names below win; if a file needs both, the dsl.h versions can be reached via their long forms
|
||||||
|
* (e.g. `def_signed_op`-style or the underlying `add_s1/s2/s4`). */
|
||||||
|
#undef add_s
|
||||||
|
#undef sub_s
|
||||||
#define add_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_add)
|
#define add_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_add)
|
||||||
#define add_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_addu)
|
#define add_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_addu)
|
||||||
#define sub_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_sub)
|
#define sub_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_sub)
|
||||||
@@ -417,7 +413,6 @@ enum { _BitOffsets = 0
|
|||||||
#define div_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_div)
|
#define div_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_div)
|
||||||
#define div_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_divu)
|
#define div_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_divu)
|
||||||
|
|
||||||
// TODO(Ed): Change convention of 'self' to ds for (destination is source)?
|
|
||||||
#define add_u_self(rd_rs, rt) add_u(rd_rs, rd_rs, rt)
|
#define add_u_self(rd_rs, rt) add_u(rd_rs, rd_rs, rt)
|
||||||
|
|
||||||
/* --- Arithmetic I-type (immediate) --- */
|
/* --- Arithmetic I-type (immediate) --- */
|
||||||
@@ -460,13 +455,9 @@ enum { _BitOffsets = 0
|
|||||||
#define shift_amount(rd, rt, n) shift_lleft(rd, rt, n)
|
#define shift_amount(rd, rt, n) shift_lleft(rd, rt, n)
|
||||||
|
|
||||||
/* nop — sll $0, $0, 0 */
|
/* nop — sll $0, $0, 0 */
|
||||||
#define nop shift_lleft(R_0, R_0, 0)
|
#define nop shift_lleft(rdiscard, rdiscard, 0)
|
||||||
#define nop2 nop, nop
|
#define nop2 nop, nop
|
||||||
|
|
||||||
// li_s — load signed 16-bit immediate into GPR (addiu rt, $0, imm — sign-extends).
|
|
||||||
#define li_s(rt, imm) add_ui((rt), R_0, (imm))
|
|
||||||
// #define load_imm_s(rt, imm) add_ui((rt), R_0, (imm))
|
|
||||||
|
|
||||||
#define load_imm_1w(rt, imm) add_ui((rt), R_0, (imm))
|
#define load_imm_1w(rt, imm) add_ui((rt), R_0, (imm))
|
||||||
#define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm))
|
#define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm))
|
||||||
|
|
||||||
|
|||||||
+79
-91
@@ -9,34 +9,6 @@
|
|||||||
|
|
||||||
ATOM_FILE_DEBUGGER_LINE_MARKER(pad_atom_c);
|
ATOM_FILE_DEBUGGER_LINE_MARKER(pad_atom_c);
|
||||||
|
|
||||||
#pragma region MACs (Mips Atom Components)
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_pad_set_centered_axes(AtomBuilder_R ab, Reg state, Reg scratch) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
load_upper_i(scratch, (PadAxis_Centered >> 16) & 0xFFFF),
|
|
||||||
or_i_self( scratch, PadAxis_Centered & 0xFFFF), // mac_load_word_imm(scratch, PadAxis_Centered),
|
|
||||||
store_word( scratch, state, O_(PadState,axes)),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_pad_set_id_byte(AtomBuilder_R ab, Reg state, Reg r_id, U1 id_value) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
add_ui( r_id, R_0, id_value),
|
|
||||||
store_byte(r_id, state, O_(PadState,id)),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_pad_set_status(AtomBuilder_R ab, U4 r_tmp, U1 r_state, U4 pad_status) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
add_ui( r_tmp, R_0, pad_status),
|
|
||||||
store_word(r_tmp, r_state, O_(PadState,status)),
|
|
||||||
})
|
|
||||||
|
|
||||||
/* Invert r_buttons (active-low → active-high) and store to PadState.buttons.
|
|
||||||
* r_buttons must already be loaded (the caller is responsible for filling the load-delay slot of
|
|
||||||
* the preceding load_half_u with an instruction that doesn't read r_buttons). */
|
|
||||||
FI_ Slice_MipsCode ac_pad_store_inverted_buttons(AtomBuilder_R ab, U1 r_buttons, U1 r_pad_state) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
nor_u( r_buttons, r_buttons, R_0),
|
|
||||||
store_half(r_buttons, r_pad_state, O_(PadState,buttons)),
|
|
||||||
})
|
|
||||||
|
|
||||||
#pragma endregion MACs (Mips Atom Components)
|
|
||||||
|
|
||||||
#pragma region Baked Atoms
|
#pragma region Baked Atoms
|
||||||
|
|
||||||
/* ----- pad_bios_snapshot -----
|
/* ----- pad_bios_snapshot -----
|
||||||
@@ -54,16 +26,16 @@ FI_ Slice_MipsCode ac_pad_store_inverted_buttons(AtomBuilder_R ab, U1 r_buttons,
|
|||||||
* byte_swap16(x) = (x >> 8) | (x << 8); nor(x, R_0) = ~x. store_half truncates to 16 bits so the upper-16 mask is implicit in the store.
|
* byte_swap16(x) = (x >> 8) | (x << 8); nor(x, R_0) = ~x. store_half truncates to 16 bits so the upper-16 mask is implicit in the store.
|
||||||
*
|
*
|
||||||
* Register use (atom-local; no wave-context touched):
|
* Register use (atom-local; no wave-context touched):
|
||||||
* R_T0 = raw base : Kept throughout; axes loads read raw[4..7] from R_T0.
|
* R_T0 = raw base (kept throughout; axes loads read raw[4..7] from R_T0)
|
||||||
* R_T1 = state base : Kept throughout; all stores go through R_T1.
|
* R_T1 = state base (kept throughout; all stores go through R_T1)
|
||||||
* R_T2 = raw[0] status : Alive across the disc/pending/id dispatch, then dead.
|
* R_T2 = raw[0] status (alive across the disc/pending/id dispatch, then dead)
|
||||||
* R_T3 = raw[1] id : Alive across the id dispatch, then dead.
|
* R_T3 = raw[1] id (alive across the id dispatch, then dead)
|
||||||
* R_T4 = scratch : Shifts, compares, immediate loads, store values.
|
* R_T4 = scratch (shifts, compares, immediate loads, store values)
|
||||||
* R_T5 = scratch : Parallel lui + ori for the 0x80808080 axes constant + byte-swap target.
|
* R_T5 = scratch (parallel lui+ori for the 0x80808080 axes constant + byte-swap target)
|
||||||
*/
|
*/
|
||||||
enum {
|
enum {
|
||||||
R_PadRaw = R_T0 atom_reg atom_type(U1),
|
R_PadRaw = R_T0 atom_reg atom_type(U1),
|
||||||
R_PadState = R_T1 atom_reg atom_type(PadState*),
|
R_PadState = R_T1 atom_reg,
|
||||||
R_RawStatus = R_T2 atom_reg,
|
R_RawStatus = R_T2 atom_reg,
|
||||||
R_RawId = R_T3 atom_reg,
|
R_RawId = R_T3 atom_reg,
|
||||||
};
|
};
|
||||||
@@ -72,8 +44,8 @@ typedef Struct_(Binds_PadBiosSnapshot) {
|
|||||||
PadState* state;
|
PadState* state;
|
||||||
};
|
};
|
||||||
internal MipsAtom_(pad_bios_snapshot) atom_info(atom_bind(Binds_PadBiosSnapshot)
|
internal MipsAtom_(pad_bios_snapshot) atom_info(atom_bind(Binds_PadBiosSnapshot)
|
||||||
, atom_reads( R_PadRaw, R_PadState, R_RawStatus, R_RawId)
|
, atom_reads( R_PadRaw, R_PadState, R_RawStatus, R_RawId, R_T4, R_T5, R_TapePtr)
|
||||||
, atom_writes(R_PadRaw, R_PadState, R_RawStatus, R_RawId)
|
, atom_writes(R_PadRaw, R_PadState, R_RawStatus, R_RawId, R_T4, R_T5, R_TapePtr)
|
||||||
) {
|
) {
|
||||||
/* === Bind consumption: T0 = raw, T1 = state, advance R_TapePtr by 8. */
|
/* === Bind consumption: T0 = raw, T1 = state, advance R_TapePtr by 8. */
|
||||||
load_word(R_PadRaw, R_TapePtr, O_(Binds_PadBiosSnapshot,raw)),
|
load_word(R_PadRaw, R_TapePtr, O_(Binds_PadBiosSnapshot,raw)),
|
||||||
@@ -81,98 +53,111 @@ internal MipsAtom_(pad_bios_snapshot) atom_info(atom_bind(Binds_PadBiosSnapshot)
|
|||||||
add_ui_self( R_TapePtr, S_(Binds_PadBiosSnapshot)),
|
add_ui_self( R_TapePtr, S_(Binds_PadBiosSnapshot)),
|
||||||
|
|
||||||
/* === Read raw[0] (status) + raw[1] (id) */
|
/* === Read raw[0] (status) + raw[1] (id) */
|
||||||
load_byte_u(R_RawStatus, R_PadRaw, O_(PadBiosRaw,status)),
|
load_byte_u(R_RawStatus, R_PadRaw, 0),
|
||||||
load_byte_u(R_RawId, R_PadRaw, O_(PadBiosRaw,id)),
|
load_byte_u(R_RawId, R_PadRaw, 1),
|
||||||
|
|
||||||
atom_label(snap_root) /* === Case 1: Disconnected (status == 0xFF). */
|
atom_label(snap_root) /* === Case 1: Disconnected (status == 0xFF). */
|
||||||
add_ui(R_T4, R_0, PadRawStatus_Timeout), branch_ne(R_RawStatus, R_T4, atom_offset(snap_root, skip_disconnected)),
|
add_ui(R_T4, R_0, 0xFF), branch_ne(R_RawStatus, R_T4, atom_offset(snap_root, skip_disconnected)),
|
||||||
/* BD-slot: pre-compute PadStatus_Disconnected. Branch reads R_T4=0xFF in EX before this WB completes.
|
/* BD-slot: pre-compute PadStatus_Disconnected. Branch reads R_T4=0xFF in EX before this WB completes.
|
||||||
* If branch NOT taken (fall through to pending/id_dispatch), R_T4 is overwritten by the next case body's add_ui — harmless. */
|
* If branch NOT taken (fall through to pending/id_dispatch), R_T4 is overwritten by the next case body's add_ui — harmless. */
|
||||||
|
|
||||||
atom_label(disconnected) /* === Disconnected body. */
|
atom_label(disconnected) /* === Disconnected body. */
|
||||||
mac_pad_set_status(R_T4, R_PadState, PadStatus_Disconnected),
|
/* R_T4 = PadStatus_Disconnected from snap_root BD-slot. */
|
||||||
store_half( R_0, R_PadState, O_(PadState,buttons)),
|
store_word(R_T4, R_PadState, O_(PadState,status)),
|
||||||
mac_pad_set_centered_axes(R_PadState, R_T4),
|
store_half(R_0, R_PadState, O_(PadState,buttons)),
|
||||||
mac_pad_set_id_byte(R_PadState, R_RawId, PadRawStatus_Timeout),
|
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||||
|
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
|
||||||
|
store_word( R_T4, R_PadState, O_(PadState,left_x)),
|
||||||
|
store_byte( R_RawId, R_PadState, O_(PadState,id)),
|
||||||
jump_rel(atom_offset(disconnected, snap_end)),
|
jump_rel(atom_offset(disconnected, snap_end)),
|
||||||
/* BD-slot: load next atom's entry point (replaces the nop).
|
/* BD-slot: load next atom's entry point (replaces the nop).
|
||||||
* Always jumps to snap_end, where mac_yield_tail() transfers control to R_AtomJmp without re-loading it. */
|
* The unconditional branch always jumps to snap_end, where mac_yield_tail()
|
||||||
|
* transfers control to R_AtomJmp without re-loading it. */
|
||||||
mac_yield_load(),
|
mac_yield_load(),
|
||||||
atom_label(skip_disconnected)
|
atom_label(skip_disconnected)
|
||||||
|
|
||||||
/* === Case 2: Pending (status == 0 && id == 0)
|
/* === Case 2: Pending (status == 0 && id == 0)
|
||||||
* Combined check: if (status | id) != 0 then skip to id_dispatch. Falls through to the Pending case only when both are zero. */
|
* Combined check: if (status | id) != 0 then skip to id_dispatch.
|
||||||
|
* Falls through to the Pending case only when both are zero. */
|
||||||
or_u_self(R_RawStatus, R_RawId), branch_ne(R_RawStatus, R_0, atom_offset(case_2, id_dispatch)),
|
or_u_self(R_RawStatus, R_RawId), branch_ne(R_RawStatus, R_0, atom_offset(case_2, id_dispatch)),
|
||||||
/* BD-slot: pre-compute PadStatus_Pending. Branch reads R_RawStatus in EX before this WB completes.
|
/* BD-slot: pre-compute PadStatus_Pending. Branch reads R_RawStatus in EX before this WB completes.
|
||||||
* If branch NOT taken (fall through to id_dispatch), R_T4 is overwritten by the digital/analog body add_ui - harmless. */
|
* If branch NOT taken (fall through to id_dispatch), R_T4 is overwritten by the digital/analog body add_ui — harmless. */
|
||||||
|
|
||||||
atom_label(pending) /* === Pending body (status=0, id=0 — pre-IRQ-empty buffer). */
|
atom_label(pending) /* === Pending body */
|
||||||
mac_pad_set_status(R_T4, R_PadState, PadStatus_Pending),
|
/* R_T4 = PadStatus_Pending from case_2 BD-slot. */
|
||||||
store_half( R_0, R_PadState, O_(PadState,buttons)),
|
store_word(R_T4, R_PadState, O_(PadState,status)),
|
||||||
mac_pad_set_centered_axes(R_PadState, R_T4),
|
store_half(R_0, R_PadState, O_(PadState,buttons)),
|
||||||
store_byte(R_RawId, R_PadState, O_(PadState,id)),
|
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||||
|
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
|
||||||
|
store_word( R_T4, R_PadState, O_(PadState,left_x)),
|
||||||
|
store_byte( R_RawId, R_PadState, O_(PadState,id)),
|
||||||
jump_rel(atom_offset(pending, snap_end)),
|
jump_rel(atom_offset(pending, snap_end)),
|
||||||
mac_yield_load(),
|
mac_yield_load(),
|
||||||
|
|
||||||
atom_label(id_dispatch) /* === Case 3-6: ID dispatch */
|
atom_label(id_dispatch) /* === Case 3-6: ID dispatch */
|
||||||
add_ui(R_T4, R_0, PadRawId_Digital), branch_ne(R_RawId, R_T4, atom_offset(id_dispatch, try_analog_stick)),
|
add_ui(R_T4, R_0, 0x41), branch_ne(R_RawId, R_T4, atom_offset(id_dispatch, try_analog_stick)),
|
||||||
/* BD-slot: pre-compute PadStatus_Digital. Branch reads R_RawId in EX before this WB completes.
|
/* BD-slot: pre-compute PadStatus_Digital. Branch reads R_RawId in EX before this WB completes.
|
||||||
* If branch NOT taken (fall through to try_analog_stick), R_T4 is overwritten by the analog body add_ui. */
|
* If branch NOT taken (fall through to try_analog_stick), R_T4 is overwritten by the analog body add_ui. */
|
||||||
|
|
||||||
/* === Digital body (status, buttons normalize, axes=0x80, id, branch.
|
/* === Digital body (status, buttons normalize, axes=0x80, id, branch. */
|
||||||
* R_T5 holds the 0x80808080 axes constant (loaded into the load-delay slot of the buttons-load).
|
/* R_T4 = PadStatus_Digital from id_dispatch BD-slot. */
|
||||||
* R_T5 is then "dead" — only consumed at the analog_pad range check downstream. */
|
store_word( R_T4, R_PadState, O_(PadState,status)),
|
||||||
mac_pad_set_status(R_T4, R_PadState, PadStatus_Digital),
|
load_half_u(R_T4, R_PadRaw, 2 * S_(U1)),
|
||||||
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw, buttons)), /* R_T4 = raw_buttons; */
|
/* Fill R_T4's load-delay slot with the 0x80808080 axes constant into R_T5
|
||||||
mac_load_word_imm( R_T5, PadAxis_Centered), /* fills the buttons-load's delay slot (doesn't read R_T4) */
|
* (R_T5 is dead on this path; it's only consumed at the analog_pad range check). */
|
||||||
// load_upper_i(R_T5, PadAxis_Centered_Hi), or_i_self(R_T5, PadAxis_Centered_Lo),
|
load_upper_i(R_T5, 0x8080), or_i_self(R_T5, 0x8080),
|
||||||
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
|
nor_u( R_T4, R_T4, R_0), /* raw_buttons is already in host bit order; no swap needed */
|
||||||
store_word(R_T5, R_PadState, O_(PadState, axes)), /* single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y) */
|
store_half( R_T4, R_PadState, O_(PadState,buttons)),
|
||||||
mac_pad_set_id_byte(R_PadState, R_T4, PadRawId_Digital),
|
|
||||||
|
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||||
|
store_word( R_T5, R_PadState, O_(PadState,left_x)),
|
||||||
|
add_ui( R_T4, R_0, 0x41),
|
||||||
|
store_byte( R_T4, R_PadState, O_(PadState,id)),
|
||||||
|
|
||||||
jump_rel(atom_offset(id_dispatch, snap_end)),
|
jump_rel(atom_offset(id_dispatch, snap_end)),
|
||||||
mac_yield_load(),
|
mac_yield_load(),
|
||||||
|
|
||||||
atom_label(try_analog_stick) /* === Case 4: AnalogStick (id == 0x53)*/
|
atom_label(try_analog_stick) /* === Case 4: AnalogStick (id == 0x53)*/
|
||||||
add_ui(R_T4, R_0, PadRawId_AnalogStick), branch_ne(R_RawId, R_T4, atom_offset(try_analog_stick, try_analog_pad)),
|
add_ui(R_T4, R_0, 0x53), branch_ne(R_RawId, R_T4, atom_offset(try_analog_stick, try_analog_pad)),
|
||||||
/* BD-slot: pre-compute PadStatus_AnalogStick. Branch reads R_RawId in EX before this WB completes.
|
/* BD-slot: pre-compute PadStatus_AnalogStick. Branch reads R_RawId in EX before this WB completes.
|
||||||
* If branch NOT taken (fall through to try_analog_pad), R_T4 is overwritten by the analog_pad body add_ui. */
|
* If branch NOT taken (fall through to try_analog_pad), R_T4 is overwritten by the analog_pad body add_ui. */
|
||||||
|
|
||||||
atom_label(analog_stick) /* === AnalogStick body
|
atom_label(analog_stick) /* === AnalogStick body
|
||||||
* R_T5 holds left_xy (loaded into the load-delay slot of the buttons-load via the left-axis load_half_u).
|
* Axes are loaded as two halfwords: raw[6..7] → left_xy (sh at offset 8), raw[4..5] → right_xy (sh at offset 10).
|
||||||
* R_T4 holds right_xy (loaded into the load-delay slot of the left-load).
|
* R_T5 holds left_xy / id-value in turn (it's dead on this path — only consumed at the analog_pad range check). */
|
||||||
* R_T5 is then "dead" — reused for the id-byte value load in mac_pad_write_id_byte.
|
/* R_T4 = PadStatus_AnalogStick from try_analog_stick BD-slot. */
|
||||||
* The buttons invert+store happens BEFORE R_T4 is overwritten by the right_xy load. */
|
store_word( R_T4, R_PadState, O_(PadState,status)),
|
||||||
mac_pad_set_status(R_T4, R_PadState, PadStatus_AnalogStick),
|
load_half_u( R_T4, R_PadRaw, 2 * S_(U1)), /* R_T4 = raw_buttons */
|
||||||
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,buttons)), /* R_T4 = raw_buttons; delay slot at the next instruction */
|
load_half_u( R_T5, R_PadRaw, 6 * S_(U1)), /* R_T5 = left_xy; fills R_T4's load-delay slot (doesn't read R_T4) */
|
||||||
load_half_u( R_T5, R_PadRaw, O_(PadBiosRaw,left)), /* fills the buttons-load's delay slot (doesn't read R_T4) */
|
nor_u( R_T4, R_T4, R_0), /* R_T4 = ~raw_buttons */
|
||||||
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
|
store_half( R_T4, R_PadState, O_(PadState,buttons)),
|
||||||
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,right)), /* fills R_T5's load-delay slot (doesn't read R_T5); overwrites R_T4 (was buttons) with right_xy */
|
load_half_u( R_T4, R_PadRaw, 4 * S_(U1)), /* R_T4 = right_xy; fills R_T5's load-delay slot */
|
||||||
store_half( R_T5, R_PadState, O_(PadState, left)),
|
store_half( R_T5, R_PadState, O_(PadState,left_x)), /* R_T5 settled, store left_xy */
|
||||||
store_half( R_T4, R_PadState, O_(PadState, right)),
|
store_half( R_T4, R_PadState, O_(PadState,right_x)),
|
||||||
mac_pad_set_id_byte(R_PadState, R_T5, PadRawId_AnalogStick),
|
add_ui( R_T5, R_0, 0x53), /* R_T5 = id value (clobbers left_xy, already stored) */
|
||||||
|
store_byte( R_T5, R_PadState, O_(PadState,id)),
|
||||||
jump_rel(atom_offset(analog_stick, snap_end)),
|
jump_rel(atom_offset(analog_stick, snap_end)),
|
||||||
mac_yield_load(),
|
mac_yield_load(),
|
||||||
|
|
||||||
atom_label(try_analog_pad) /* === Case 5-6: AnalogPad (id & 0xF0 == 0x70) */
|
atom_label(try_analog_pad) /* === Case 5-6: AnalogPad (id & 0xF0 == 0x70) */
|
||||||
and_i( R_T4, R_RawId, PadRawId_AnalogPadMask),
|
and_i( R_T4, R_RawId, 0xF0),
|
||||||
add_ui( R_T5, R_0, PadRawId_AnalogPadValue),
|
add_ui( R_T5, R_0, 0x70),
|
||||||
branch_ne(R_T4, R_T5, atom_offset(try_analog_pad, try_unsupported)),
|
branch_ne(R_T4, R_T5, atom_offset(try_analog_pad, try_unsupported)),
|
||||||
/* BD-slot: pre-compute PadStatus_AnalogPad. Branch reads R_T4 in EX before this WB completes.
|
/* BD-slot: pre-compute PadStatus_AnalogPad. Branch reads R_T4 in EX before this WB completes.
|
||||||
* If branch NOT taken (fall through to try_unsupported), R_T4 is overwritten by the unsupported body add_ui. */
|
* If branch NOT taken (fall through to try_unsupported), R_T4 is overwritten by the unsupported body add_ui. */
|
||||||
|
|
||||||
atom_label(analog_pad) /* === AnalogPad body
|
atom_label(analog_pad) /* === AnalogPad body
|
||||||
* Same shape as AnalogStick with AnalogPad status. R_T5 holds left_xy (it's dead on this path).
|
* Same shape as AnalogStick with AnalogPad status. R_T5 holds left_xy (it's dead on this path). */
|
||||||
* The id byte is raw id from the BIOS buffer (R_RawId already holds raw[1]).
|
/* R_T4 = PadStatus_AnalogPad from try_analog_pad BD-slot. */
|
||||||
* Buttons invert + store happens before R_T4 is overwritten by the right_xy load. */
|
store_word( R_T4, R_PadState, O_(PadState,status)),
|
||||||
mac_pad_set_status(R_T4, R_PadState, PadStatus_AnalogPad),
|
load_half_u(R_T4, R_PadRaw, 2 * S_(U1)), /* R_T4 = raw_buttons */
|
||||||
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,buttons)), /* R_T4 = raw_buttons; delay slot at the next instruction */
|
load_half_u(R_T5, R_PadRaw, 6 * S_(U1)), /* R_T5 = left_xy; fills R_T4's load-delay slot */
|
||||||
load_half_u( R_T5, R_PadRaw, O_(PadBiosRaw,left)), /* fills the buttons-load's delay slot (doesn't read R_T4) */
|
nor_u( R_T4, R_T4, R_0), /* R_T4 = ~raw_buttons */
|
||||||
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
|
store_half( R_T4, R_PadState, O_(PadState,buttons)),
|
||||||
load_half_u(R_T4, R_PadRaw, O_(PadBiosRaw,right)), /* fills R_T5's load-delay slot (doesn't read R_T5); overwrites R_T4 with right_xy */
|
load_half_u(R_T4, R_PadRaw, 4 * S_(U1)), /* R_T4 = right_xy; fills R_T5's load-delay slot */
|
||||||
store_half( R_T5, R_PadState, O_(PadState, left)),
|
store_half( R_T5, R_PadState, O_(PadState,left_x)), /* R_T5 settled, store left_xy */
|
||||||
store_half( R_T4, R_PadState, O_(PadState, right)),
|
store_half( R_T4, R_PadState, O_(PadState,right_x)),
|
||||||
store_byte( R_RawId, R_PadState, O_(PadState, id)),
|
store_byte( R_RawId, R_PadState, O_(PadState,id)),
|
||||||
|
|
||||||
jump_rel(atom_offset(analog_pad, snap_end)),
|
jump_rel(atom_offset(analog_pad, snap_end)),
|
||||||
mac_yield_load(),
|
mac_yield_load(),
|
||||||
@@ -181,8 +166,11 @@ atom_label(try_unsupported) /* === Case 7: Unsupported — fall through from the
|
|||||||
add_ui( R_T4, R_0, PadStatus_Unsupported),
|
add_ui( R_T4, R_0, PadStatus_Unsupported),
|
||||||
store_word(R_T4, R_PadState, O_(PadState,status)),
|
store_word(R_T4, R_PadState, O_(PadState,status)),
|
||||||
store_half(R_0, R_PadState, O_(PadState,buttons)),
|
store_half(R_0, R_PadState, O_(PadState,buttons)),
|
||||||
mac_pad_set_centered_axes(R_PadState, R_T4),
|
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||||
mac_pad_set_id_byte(R_PadState, R_RawId, PadUnknownId_Sentinel),
|
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
|
||||||
|
store_word( R_T4, R_PadState, O_(PadState,left_x)),
|
||||||
|
add_ui( R_T4, R_0, 0xFF), /* 0xFF sentinel: "unknown id" */
|
||||||
|
store_byte( R_T4, R_PadState, O_(PadState,id)),
|
||||||
/* Fall through to snap_end. */
|
/* Fall through to snap_end. */
|
||||||
|
|
||||||
atom_label(no_jump_fallthrough)
|
atom_label(no_jump_fallthrough)
|
||||||
|
|||||||
@@ -1,78 +0,0 @@
|
|||||||
#ifdef INTELLISENSE_DIRECTIVES
|
|
||||||
# include "dsl.h"
|
|
||||||
# include "gcc_asm.h"
|
|
||||||
# include "mips.h"
|
|
||||||
# include "bios.h"
|
|
||||||
# include "pad.h"
|
|
||||||
#endif
|
|
||||||
|
|
||||||
/* Uses ONE 8-byte frame allocated via the compiler's standard prologue.
|
|
||||||
* 4 wasted-arg words for B(12h) InitPAD2 are at [SP+0..15] but are not explicitly allocated.
|
|
||||||
* Compiler handles the MIPS O32 "wasted stack" convention for us by treating the B-call as a 4-arg call.
|
|
||||||
*
|
|
||||||
* The buffer pointers are passed as arguments so the compiler keeps them in callee-saved registers;
|
|
||||||
* The B(12h) asm volatile block does NOT clobber those registers (it clobbers only the volatile GPRs + B-table arg registers explicitly).
|
|
||||||
* The C-level writes after the call re-load the pointers from their callee-saved homes.
|
|
||||||
*
|
|
||||||
* The clobber list for both B-calls names the full BIOS destroy set documented in kernelbios.md:167-174 (R1..R15, R24..R25, R31, HI/LO).
|
|
||||||
* The kernel-ABI "volatile GPRs" subset is clb_mem_drain; the rest of the destroy set is enumerated explicitly here. */
|
|
||||||
NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
|
|
||||||
{
|
|
||||||
/* Pin raw0 + raw1 to $a0 + $a1 via rgcc; the B(12h) call uses these directly.
|
|
||||||
* The `(void)` casts mark them as unread after the call so the compiler doesn't need to move them back. */
|
|
||||||
register PadBiosRaw* p0 rgcc(R_A0) = raw0;
|
|
||||||
register PadBiosRaw* p1 rgcc(R_A1) = raw1;
|
|
||||||
(void)p0; (void)p1;
|
|
||||||
|
|
||||||
// TODO(Ed): Properly annotate the raw values in the inline asm instructions.
|
|
||||||
// Use enums.
|
|
||||||
|
|
||||||
/* B(12h) InitPAD2(raw0, 0x22, raw1, 0x22)
|
|
||||||
* $a0 = raw0 (rgcc-bound; survives the sequence below)
|
|
||||||
* $a1 = raw1 (preserved into $a2 before $a1 is overwritten)
|
|
||||||
* $a2 = raw1 (moved from $a1; survives $a1's overwrite)
|
|
||||||
* $a3 = 0x22 (immediate)
|
|
||||||
* $t1 = 0x12 (function number)
|
|
||||||
* $t2 = 0xB0 (BIOS B-table address) */
|
|
||||||
asm volatile(
|
|
||||||
asm_words(
|
|
||||||
or_u( R_A2, R_A1, R_0), /* $a2 = $a1 = raw1 */
|
|
||||||
add_ui( R_A1, R_0, bios_pad_buffer_size), /* $a1 = 0x22 */
|
|
||||||
add_ui( R_A3, R_0, bios_pad_buffer_size), /* $a3 = 0x22 */
|
|
||||||
add_ui( R_T1, R_0, bios_init_pad_2), /* $t1 = 0x12 */
|
|
||||||
add_ui( R_T2, R_0, bios_btable_addr), /* $t2 = 0xB0 */
|
|
||||||
call_reg(R_T2), /* jalr $t2, $ra */
|
|
||||||
nop /* BD slot */
|
|
||||||
)
|
|
||||||
asm_rpins, r_use(p0), r_use(p1)
|
|
||||||
asm_clobber:
|
|
||||||
rlit(R_AT),
|
|
||||||
rlit(R_V0), rlit(R_V1),
|
|
||||||
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
|
||||||
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
|
||||||
rlit(R_RA),
|
|
||||||
clb_mem_drain
|
|
||||||
);
|
|
||||||
|
|
||||||
/* The C-level writes re-load the pointers via the parameter names and write 0xFF to each
|
|
||||||
* buffer's status byte to mark the initial-state hazard documented in kernelbios.md:1621-1624. */
|
|
||||||
u1_v(raw0)[0] = 0xFF;
|
|
||||||
u1_v(raw1)[0] = 0xFF;
|
|
||||||
|
|
||||||
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */
|
|
||||||
asm volatile(
|
|
||||||
asm_words(
|
|
||||||
add_ui( R_T1, R_0, bios_start_pad_2), /* $t1 = 0x13 */
|
|
||||||
add_ui( R_T2, R_0, bios_btable_addr), /* $t2 = 0xB0 (re-load) */
|
|
||||||
call_reg(R_T2), /* jalr $t2, $ra */
|
|
||||||
nop /* BD slot */
|
|
||||||
)
|
|
||||||
asm_clobber:
|
|
||||||
rlit(R_AT),
|
|
||||||
rlit(R_V0), rlit(R_V1),
|
|
||||||
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
|
||||||
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
|
||||||
rlit(R_RA),
|
|
||||||
clb_mem_drain
|
|
||||||
);
|
|
||||||
}
|
|
||||||
+36
-78
@@ -1,30 +1,28 @@
|
|||||||
#ifdef INTELLISENSE_DIRECTIVES
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
# pragma once
|
# pragma once
|
||||||
# include "dsl.h"
|
# include "dsl.h"
|
||||||
# include "math.h"
|
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
/* PSX button bit positions — 1:1 with PSX-SPX docs at docs/psx-spx/docs/controllersandmemorycards.md:405-421.
|
/* PSX button bit positions — 1:1 with PSX-SPX docs at docs/psx-spx/docs/controllersandmemorycards.md:405-421.
|
||||||
* Wire is active-low (0 = pressed).
|
* Wire is active-low (0 = pressed).
|
||||||
* The decoder atom computes buttons = (~raw_buttons) & 0xFFFF;
|
* The decoder atom computes buttons = (~raw_buttons) & 0xFFFF; the active-low-to-active-high inversion is applied bit-by-bit. */
|
||||||
* active-low-to-active-high inversion is applied bit-by-bit. */
|
enum {
|
||||||
typedef Enum_(U2, PadBtns) {
|
Bit_(Pad_Select, 0),
|
||||||
Bit_(Pad_Select, 0),
|
Bit_(Pad_L3, 1),
|
||||||
Bit_(Pad_L3, 1),
|
Bit_(Pad_R3, 2),
|
||||||
Bit_(Pad_R3, 2),
|
Bit_(Pad_Start, 3),
|
||||||
Bit_(Pad_Start, 3),
|
Bit_(Pad_Up, 4),
|
||||||
Bit_(Pad_Up, 4),
|
Bit_(Pad_Right, 5),
|
||||||
Bit_(Pad_Right, 5),
|
Bit_(Pad_Down, 6),
|
||||||
Bit_(Pad_Down, 6),
|
Bit_(Pad_Left, 7),
|
||||||
Bit_(Pad_Left, 7),
|
Bit_(Pad_L2, 8),
|
||||||
Bit_(Pad_L2, 8),
|
Bit_(Pad_R2, 9),
|
||||||
Bit_(Pad_R2, 9),
|
Bit_(Pad_L1, 10),
|
||||||
Bit_(Pad_L1, 10),
|
Bit_(Pad_R1, 11),
|
||||||
Bit_(Pad_R1, 11),
|
|
||||||
Bit_(Pad_Triangle, 12),
|
Bit_(Pad_Triangle, 12),
|
||||||
Bit_(Pad_Circle, 13),
|
Bit_(Pad_Circle, 13),
|
||||||
Bit_(Pad_Cross, 14),
|
Bit_(Pad_Cross, 14),
|
||||||
Bit_(Pad_Square, 15),
|
Bit_(Pad_Square, 15),
|
||||||
};
|
};
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
@@ -34,22 +32,18 @@ enum {
|
|||||||
Pad1 = 1 << PadId_Offset,
|
Pad1 = 1 << PadId_Offset,
|
||||||
};
|
};
|
||||||
|
|
||||||
/* =============================================================================
|
#define pad0_(btn_id) (btn_id << Pad0)
|
||||||
|
#define pad1_(btn_id) (btn_id << Pad1)
|
||||||
|
|
||||||
|
/* ============================================================
|
||||||
* BIOS pad-buffer subsystem: docs/psx-spx/docs/kernelbios.md (B(12h) + B(13h))
|
* BIOS pad-buffer subsystem: docs/psx-spx/docs/kernelbios.md (B(12h) + B(13h))
|
||||||
* ============================================================================= */
|
* ============================================================ */
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
PAD_BIOS_RAW_SIZE = 0x22,
|
PAD_BIOS_RAW_SIZE = 0x22,
|
||||||
};
|
};
|
||||||
// BIOS pad buffer layout (docs/psx-spx/docs/kernelbios.md (InitPAD2 returns 0x22 = 34 bytes per port)).
|
|
||||||
// Bytes 0..7 are the named snapshot region; bytes 8..33 are reserved (the BIOS writes the buffer raw; we only read bytes 0..7 via O_(PadBiosRaw, ...)).
|
|
||||||
typedef Struct_(PadBiosRaw) {
|
typedef Struct_(PadBiosRaw) {
|
||||||
U1 status; /* offset 0 (PadRawStatus_Ok / PadRawStatus_Timeout) */
|
U1 bytes[PAD_BIOS_RAW_SIZE];
|
||||||
U1 id; /* offset 1 (PadRawId_Digital / PadRawId_AnalogStick / 0x7x AnalogPad) */
|
|
||||||
U2 buttons; /* offset 2-3 (active-low 16-bit button map) */
|
|
||||||
V2_U1 right; /* offset 4-5 (right stick x, y) */
|
|
||||||
V2_U1 left; /* offset 6-7 (left stick x, y) */
|
|
||||||
U1 reserved[PAD_BIOS_RAW_SIZE - 8]; /* offset 8..33 */
|
|
||||||
};
|
};
|
||||||
|
|
||||||
typedef Enum_(U4, PadStatus) {
|
typedef Enum_(U4, PadStatus) {
|
||||||
@@ -62,54 +56,18 @@ typedef Enum_(U4, PadStatus) {
|
|||||||
PadStatus_Invalid,
|
PadStatus_Invalid,
|
||||||
};
|
};
|
||||||
|
|
||||||
/* Distinct from the game-facing PadStatus enum: PadRawStatus_Ok and PadRawStatus_Timeout are raw BIOS values;
|
/* PadState — per-port normalized runtime state.
|
||||||
* PadStatus_* are game-facing post-decode states. PadUnknownId_Sentinel is written by the decoder
|
* Field order is chosen so that the 4 axes (left_x, left_y, right_x, right_y)
|
||||||
* when the controller id does not match any known controller type.
|
* form a contiguous 4-byte block at offset 8, allowing a single `store_word` to clear-or-write all 4 axes in one MIPS instruction.
|
||||||
* PadAxisCentered_Word: Four-byte 0x80 pattern used to clear / center
|
* The struct size stays 12 bytes (unchanged from the prior order,
|
||||||
* four byte axes at PadState.left_x through PadState.right_y. */
|
* which left the C compiler to insert 1 byte of trailing pad to reach the 4-byte struct alignment). */
|
||||||
typedef Enum_(U1, PadRawStatus) {
|
|
||||||
PadRawStatus_Ok = 0x00,
|
|
||||||
PadRawStatus_Timeout = 0xFF,
|
|
||||||
};
|
|
||||||
typedef Enum_(U1, PadRawId) {
|
|
||||||
PadRawId_Digital = 0x41,
|
|
||||||
PadRawId_AnalogStick = 0x53,
|
|
||||||
PadRawId_AnalogPadMask = 0xF0,
|
|
||||||
PadRawId_AnalogPadValue = 0x70,
|
|
||||||
};
|
|
||||||
typedef Enum_(U1, PadUnknownId) {
|
|
||||||
PadUnknownId_Sentinel = 0xFF,
|
|
||||||
};
|
|
||||||
typedef Enum_(U4, PadAxisCentered) {
|
|
||||||
PadAxis_Centered_Hi = 0x8080,
|
|
||||||
PadAxis_Centered_Lo = 0x8080,
|
|
||||||
PadAxis_Centered = 0x80808080U,
|
|
||||||
};
|
|
||||||
typedef Enum_(U1, PadDeadZone) {
|
|
||||||
PadDeadZone_LowBound = 0x70, /* left_x < LowBound → active; delta = 0x80 - left_x > 0 (rightward pull) */
|
|
||||||
PadDeadZone_Center = 0x80, /* analog rest position; left_x == Center → delta = 0 (no rotation) */
|
|
||||||
PadDeadZone_HighBound = 0x90, /* left_x > HighBound → active; delta = 0x80 - left_x < 0 (leftward pull) */
|
|
||||||
};
|
|
||||||
|
|
||||||
|
|
||||||
typedef Struct_(PadAxes) {
|
|
||||||
V2_U1 left; /* offset 8-9 */
|
|
||||||
V2_U1 right; /* offset 10-11 */
|
|
||||||
};
|
|
||||||
// Field order is chosen so that the 4 axes (left_x, left_y, right_x, right_y)
|
|
||||||
// form a contiguous 4-byte block at offset 8, allowing a single `store_word` to clear-or-write all 4 axes in one MIPS instruction.
|
|
||||||
typedef Struct_(PadState) {
|
typedef Struct_(PadState) {
|
||||||
PadStatus status; /* offset 0, (U4) */
|
PadStatus status; /* offset 0, size 4 (U4) */
|
||||||
PadBtns buttons; /* offset 4, */
|
U2 buttons; /* offset 4, size 2 */
|
||||||
U1 id; /* offset 6, */
|
U1 id; /* offset 6, size 1 */
|
||||||
byte_pad(1); /* offset 7, explicit pad to align the axes block */
|
U1 pad; /* offset 7, size 1 — explicit pad to align the axes block */
|
||||||
union {
|
U1 left_x; /* offset 8, size 1 — store_word target (4-byte aligned) */
|
||||||
A2_V2_U1 axes; /* offset 8-11 store_target (4-byte aligned)*/
|
U1 left_y; /* offset 9, size 1 */
|
||||||
struct {
|
U1 right_x; /* offset 10, size 1 */
|
||||||
V2_U1 left; /* offset 8-9 */
|
U1 right_y; /* offset 11, size 1 */
|
||||||
V2_U1 right; /* offset 10-11 */
|
|
||||||
};
|
|
||||||
};
|
|
||||||
};
|
};
|
||||||
|
|
||||||
internal void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1);
|
|
||||||
|
|||||||
+5
-23
@@ -64,9 +64,9 @@ typedef Struct_(Tile) {
|
|||||||
Linear Algebra
|
Linear Algebra
|
||||||
*/
|
*/
|
||||||
|
|
||||||
MT3_S2S4* mt3s2s4_rotation (V3_S2* vec, MT3_S2S4* mat) asm("RotMatrix");
|
M3_S2* m3s2_rotation (V3_S2* vec, M3_S2* mat) asm("RotMatrix");
|
||||||
MT3_S2S4* mt3s2s4_translation(MT3_S2S4* mat, V3_S4* vec) asm("TransMatrix");
|
M3_S2* m3s2_translation(M3_S2* mat, V3_S4* vec) asm("TransMatrix");
|
||||||
MT3_S2S4* mt3s2s4_scale (MT3_S2S4* mat, V3_S4* vec) asm("ScaleMatrix");
|
M3_S2* m3s2_scale (M3_S2* mat, V3_S4* vec) asm("ScaleMatrix");
|
||||||
|
|
||||||
// Rotation, Translation, Perspective
|
// Rotation, Translation, Perspective
|
||||||
|
|
||||||
@@ -99,23 +99,5 @@ FI_ S4 rtp_avg_nclip_a4_v3s2(
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
void gte_matrix_set_rotation (MT3_S2S4* mat) asm("SetRotMatrix");
|
void gte_matrix_set_rotation (M3_S2* mat) asm("SetRotMatrix");
|
||||||
void gte_matrix_set_translation(MT3_S2S4* mat) asm("SetTransMatrix");
|
void gte_matrix_set_translation(M3_S2* mat) asm("SetTransMatrix");
|
||||||
|
|
||||||
// Einheit, Metrication to unit vector. "Normalization", not Orthogonal "Normal, Normalis". Directionalization.
|
|
||||||
// RGA(Lengyel): Normalize the bulk of a zero-weight direction. This is not finite-point unitization (which forces w=1).
|
|
||||||
S4 normalize_v3s4(V3_S4* v0, V3_S4* v1) asm("VectorNormal");
|
|
||||||
|
|
||||||
// RGA(Lengyel): Apply the matrix expansion of a rigid transformation.
|
|
||||||
// Motor antiproduct is equivalent for unitized points; LA form is what GTE consumes.
|
|
||||||
V3_S4* mul_m3s2_v3s4(MT3_S2S4* m, V3_S4* v, V3_S4* result) asm("ApplyMatrixLV");
|
|
||||||
|
|
||||||
// RGA(Lengyel): Store the full translation column. The motor translator would store half this displacement in m.xyz.
|
|
||||||
MT3_S2S4* trans_m3s2(MT3_S2S4* m, V3_S4* off) asm("TransMatrix");
|
|
||||||
|
|
||||||
MT3_S2S4* gte_comp_coord_m3s2(MT3_S2S4* m0, MT3_S2S4* m1, MT3_S2S4* result) asm("CompMatrixLV");
|
|
||||||
|
|
||||||
// RGA(Lengyel): Complement(Wedge(a,b)), i.e. the Euclidean 3D complement of the exterior product, stored as a V3_S4.
|
|
||||||
// The underlying GTE OP is a specialized signed-16-bit D x IR command; the wedge interpretation is a 3D dual of the same 3 scalars.
|
|
||||||
void cross_v3s4(V3_S4* v0, V3_S4* v1, V3_S4* result) asm("OuterProduct12");
|
|
||||||
|
|
||||||
|
|||||||
@@ -15,8 +15,6 @@
|
|||||||
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
||||||
|
|
||||||
WORD_COUNT(nop, 1)
|
WORD_COUNT(nop, 1)
|
||||||
WORD_COUNT(atom_label, 0)
|
|
||||||
WORD_COUNT(atom_offset, 0)
|
|
||||||
WORD_COUNT(load_upper_i, 1)
|
WORD_COUNT(load_upper_i, 1)
|
||||||
WORD_COUNT(jump_reg, 1)
|
WORD_COUNT(jump_reg, 1)
|
||||||
WORD_COUNT(jump_link, 1)
|
WORD_COUNT(jump_link, 1)
|
||||||
@@ -55,17 +53,7 @@ WORD_COUNT(gte_mv_to_ctrl_r, 1)
|
|||||||
WORD_COUNT(gte_sw, 1)
|
WORD_COUNT(gte_sw, 1)
|
||||||
WORD_COUNT(gte_cmdw_rtpt, 1)
|
WORD_COUNT(gte_cmdw_rtpt, 1)
|
||||||
WORD_COUNT(gte_cmdw_nclip, 1)
|
WORD_COUNT(gte_cmdw_nclip, 1)
|
||||||
WORD_COUNT(gte_cmdw_op, 1)
|
|
||||||
WORD_COUNT(gte_avg_sort_z3, 1)
|
WORD_COUNT(gte_avg_sort_z3, 1)
|
||||||
WORD_COUNT(gte_cmdw_sqr, 1)
|
|
||||||
WORD_COUNT(gte_cmdw_gpf, 1)
|
|
||||||
WORD_COUNT(shift_lleft_var, 1)
|
|
||||||
WORD_COUNT(shift_aright_var, 1)
|
|
||||||
WORD_COUNT(li_s, 1)
|
|
||||||
WORD_COUNT(and_i, 1)
|
|
||||||
WORD_COUNT(add_si, 1)
|
|
||||||
WORD_COUNT(branch_lt_zero, 1)
|
|
||||||
WORD_COUNT(sub_s, 1)
|
|
||||||
WORD_COUNT(sub_u, 1)
|
WORD_COUNT(sub_u, 1)
|
||||||
WORD_COUNT(nop2, 2)
|
WORD_COUNT(nop2, 2)
|
||||||
|
|
||||||
|
|||||||
@@ -1,13 +0,0 @@
|
|||||||
#ifdef INTELLISENSE_DIRECTIVES
|
|
||||||
#pragma once
|
|
||||||
#endif
|
|
||||||
// Auto-generated by ps1_meta.lua (passes/auto_reg.lua) — DO NOT EDIT
|
|
||||||
// Directory: C:\projects\Pikuma\ps1\code\hello_camera
|
|
||||||
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.c
|
|
||||||
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.h
|
|
||||||
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.atom.c
|
|
||||||
// Per-phase register allocations resolved by the lua pass.
|
|
||||||
// R_<Sym>_Code = <chosen GPR's _Code constant> for every marker in this directory.
|
|
||||||
|
|
||||||
#define R_GpTmp_Code R_V0_Code
|
|
||||||
|
|
||||||
@@ -8,7 +8,7 @@
|
|||||||
#pragma region hello_camera
|
#pragma region hello_camera
|
||||||
|
|
||||||
|
|
||||||
// --- atom: pad_input_cube_rotation (61 words) ---
|
// --- atom: pad_apply_input (60 words) ---
|
||||||
|
|
||||||
#define _atom_offset_dpad_left_exit_dpad_left 6
|
#define _atom_offset_dpad_left_exit_dpad_left 6
|
||||||
#define _atom_offset_dpad_right_exit_dpad_right 6
|
#define _atom_offset_dpad_right_exit_dpad_right 6
|
||||||
@@ -26,25 +26,7 @@ enum {
|
|||||||
atom_offset_end_low_exit_stick = _atom_offset_end_low_exit_stick,
|
atom_offset_end_low_exit_stick = _atom_offset_end_low_exit_stick,
|
||||||
};
|
};
|
||||||
|
|
||||||
// --- atom: pad_input_cam (40 words) ---
|
// --- atom: cube_g4_face (76 words) ---
|
||||||
|
|
||||||
#define _atom_offset_left_x_exit_left_x 3
|
|
||||||
#define _atom_offset_right_x_exit_right_x 3
|
|
||||||
#define _atom_offset_up_y_exit_up_y 3
|
|
||||||
#define _atom_offset_down_y_exit_down_y 3
|
|
||||||
#define _atom_offset_cross_z_exit_cross_z 3
|
|
||||||
#define _atom_offset_circle_z_exit_circle_z 3
|
|
||||||
|
|
||||||
enum {
|
|
||||||
atom_offset_left_x_exit_left_x = _atom_offset_left_x_exit_left_x,
|
|
||||||
atom_offset_right_x_exit_right_x = _atom_offset_right_x_exit_right_x,
|
|
||||||
atom_offset_up_y_exit_up_y = _atom_offset_up_y_exit_up_y,
|
|
||||||
atom_offset_down_y_exit_down_y = _atom_offset_down_y_exit_down_y,
|
|
||||||
atom_offset_cross_z_exit_cross_z = _atom_offset_cross_z_exit_cross_z,
|
|
||||||
atom_offset_circle_z_exit_circle_z = _atom_offset_circle_z_exit_circle_z,
|
|
||||||
};
|
|
||||||
|
|
||||||
// --- atom: cube_g4_face (75 words) ---
|
|
||||||
|
|
||||||
#define _atom_offset_cull_cube_g4_face_exit 41
|
#define _atom_offset_cull_cube_g4_face_exit 41
|
||||||
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
|
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
|
||||||
|
|||||||
@@ -10,14 +10,13 @@
|
|||||||
# include "duffle/pad.h"
|
# include "duffle/pad.h"
|
||||||
# include "duffle/word_count.metadata.h"
|
# include "duffle/word_count.metadata.h"
|
||||||
# include "duffle/psyq.h"
|
# include "duffle/psyq.h"
|
||||||
# include "duffle/math.atom.h"
|
# include "duffle/math.atom.c"
|
||||||
# include "duffle/mips.atom.c"
|
# include "duffle/mips.atom.c"
|
||||||
# include "duffle/gte.atom.c"
|
# include "duffle/gte.atom.c"
|
||||||
# include "duffle/gp.atom.c"
|
# include "duffle/gp.atom.c"
|
||||||
# include "duffle/psyq.atom.c"
|
# include "duffle/psyq.atom.c"
|
||||||
# include "gen/offsets.h"
|
# include "gen/offsets.h"
|
||||||
# include "gen/macs.h"
|
# include "gen/macs.h"
|
||||||
# include "gen/auto_reg.h"
|
|
||||||
# include "hello_camera.h"
|
# include "hello_camera.h"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
@@ -25,8 +24,8 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c);
|
|||||||
|
|
||||||
#pragma region MACs (Mips Atom components)
|
#pragma region MACs (Mips Atom components)
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_put_disp_env(AtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
FI_ Slice_MipsCode ac_put_disp_env(U4 reg_transfer, U4 reg_base, U2 port)
|
||||||
MipsAtomComp_Proc_(ab, {
|
MipsAtomComp_Proc_(ac_put_disp_env, {
|
||||||
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
|
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
|
||||||
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
|
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
|
||||||
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
||||||
@@ -36,8 +35,8 @@ MipsAtomComp_Proc_(ab, {
|
|||||||
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
||||||
})
|
})
|
||||||
|
|
||||||
I_ Slice_MipsCode ac_put_draw_env(AtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
FI_ Slice_MipsCode ac_put_draw_env(U4 reg_transfer, U4 reg_base, U2 port)
|
||||||
MipsAtomComp_Proc_(ab, {
|
MipsAtomComp_Proc_(ac_put_draw_env, {
|
||||||
/*
|
/*
|
||||||
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
|
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
|
||||||
* References:
|
* References:
|
||||||
@@ -51,18 +50,18 @@ MipsAtomComp_Proc_(ab, {
|
|||||||
* (binary; the PutDrawEnv implementation builds the 16-word DR_ENV from the user's DRAWENV struct and emits it via GP0 GPU commands.)
|
* (binary; the PutDrawEnv implementation builds the 16-word DR_ENV from the user's DRAWENV struct and emits it via GP0 GPU commands.)
|
||||||
*
|
*
|
||||||
* Word indices (libpsyx PutDrawEnv / SetDrawEnv order):
|
* Word indices (libpsyx PutDrawEnv / SetDrawEnv order):
|
||||||
* tag = (length << 24) | addr — 16-word packet (1 tag + 15 code)
|
* tag = (length << 24) | addr — 16-word packet (1 tag + 15 code)
|
||||||
* code[0] = DrawMode (dfe=1, dtd=0, tpage=0) — must come first per libpsyx
|
* code[0] = DrawMode (dfe=1, dtd=0, tpage=0) — must come first per libpsyx
|
||||||
* code[1] = TextureWindow (tw=(0,0)) — bare-cmd word; GPU uses current state
|
* code[1] = TextureWindow (tw=(0,0)) — bare-cmd word; GPU uses current state
|
||||||
* code[2] = DrawArea top-left (clip.x=0, clip.y=240)
|
* code[2] = DrawArea top-left (clip.x=0, clip.y=240)
|
||||||
* code[3] = DrawArea bottom-right (clip.x+w=320, clip.y+h=480)
|
* code[3] = DrawArea bottom-right (clip.x+w=320, clip.y+h=480)
|
||||||
* code[4] = DrawOffset (ofs=(0,0)) — bare-cmd word
|
* code[4] = DrawOffset (ofs=(0,0)) — bare-cmd word
|
||||||
* code[5] = Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit
|
* code[5] = Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit
|
||||||
* code[6] = Initial-bg-color (isbg=1, r=7, g=7, b=7)
|
* code[6] = Initial-bg-color (isbg=1, r=7, g=7, b=7)
|
||||||
* code[7] = DrawMode (isbg=1, tpage=0) — re-asserts DrawMode with isbg
|
* code[7] = DrawMode (isbg=1, tpage=0) — re-asserts DrawMode with isbg
|
||||||
* code[8..10] = padding (NOP) — 3 words to fill the packet
|
* code[8..10] = padding (NOP) — 3 words to fill the packet
|
||||||
* code[11..12] = TextureWindow bottom-right — defaults to (0,0,0,0)
|
* code[11..12] = TextureWindow bottom-right — defaults to (0,0,0,0)
|
||||||
* code[13..14] = padding (NOP) — completes the 16-word packet
|
* code[13..14] = padding (NOP) — completes the 16-word packet
|
||||||
*/
|
*/
|
||||||
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port), /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */
|
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port), /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */
|
||||||
mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port), /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */
|
mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port), /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */
|
||||||
@@ -91,193 +90,6 @@ MipsAtomComp_Proc_(ab, {
|
|||||||
|
|
||||||
#pragma endregion MACs
|
#pragma endregion MACs
|
||||||
|
|
||||||
#pragma region Atom Procs
|
|
||||||
// Modular Atoms
|
|
||||||
|
|
||||||
#define AtomBundle_(name) Struct_(tmpl(AtomBundle,name))
|
|
||||||
#define AtomBundle_Len(name) S_(tmpl(AtomBundle,name))/S_(MipsAtom*)
|
|
||||||
#define AtomBundleEntry_(bundle,entry) tmpl(bundle,entry)
|
|
||||||
|
|
||||||
#pragma region resolve_look_at
|
|
||||||
/* ─── resolve_look_at bundle chain atoms ──────────────────────────── */
|
|
||||||
|
|
||||||
typedef AtomBundle_(resolve_look_at) { MipsAtom
|
|
||||||
*input_and_sub,
|
|
||||||
*normalize_fwd_uz,
|
|
||||||
*cross_to_right,
|
|
||||||
*normalize_right_ux,
|
|
||||||
*cross_to_up,
|
|
||||||
*normalize_up_uy,
|
|
||||||
*pop_mv_trans;
|
|
||||||
};
|
|
||||||
|
|
||||||
typedef Struct_(ResolveLookAtScratch) {
|
|
||||||
V3_S4 fwd;
|
|
||||||
V3_S4 uz;
|
|
||||||
V3_S4 right;
|
|
||||||
V3_S4 ux;
|
|
||||||
V3_S4 up;
|
|
||||||
V3_S4 uy;
|
|
||||||
P3_S4 eye;
|
|
||||||
P3_S4 target;
|
|
||||||
V3_S4 up_in;
|
|
||||||
};
|
|
||||||
|
|
||||||
/* Binds_ResolveLookAtSub — what the C side pushes onto the tape before input_and_sub.
|
|
||||||
* The scratchpad base is no longer pushed because R_ScratchBase (= R_SP) is a tape carrier
|
|
||||||
* preserved across atoms; the atom body reads 0x1F800000 directly from R_SP. */
|
|
||||||
typedef Struct_(Binds_ResolveLookAt) {
|
|
||||||
MT3_S2S4* look_at;
|
|
||||||
P3_S4* eye;
|
|
||||||
P3_S4* target;
|
|
||||||
V3_S4* up_in;
|
|
||||||
};
|
|
||||||
typedef Struct_(Binds_ResolveLookAtSub) {
|
|
||||||
P3_S4* target;
|
|
||||||
P3_S4* eye;
|
|
||||||
V3_S4* up_in;
|
|
||||||
};
|
|
||||||
typedef Struct_(RegUse_resolve_look_at_input_and_sub) {
|
|
||||||
Reg target; Reg eye; Reg up_in;
|
|
||||||
Reg t0; Reg t1; Reg t2; Reg t3; Reg t4;
|
|
||||||
};
|
|
||||||
/* Atom 0 in the bundle: input_and_sub. Stages C-side inputs into the scratchpad and computes fwd = target - eye. */
|
|
||||||
internal MipsAtom* AtomBundleEntry_(resolve_look_at,input_and_sub)(AtomArena_R aa, RegUse_resolve_look_at_input_and_sub r)
|
|
||||||
atom_info(atom_bind(Binds_ResolveLookAtSub)) MipsAtom_Proc_(aa, {
|
|
||||||
load_word(r.target, R_TapePtr, O_(Binds_ResolveLookAtSub,target)),
|
|
||||||
load_word(r.eye, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)),
|
|
||||||
load_word(r.up_in, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)),
|
|
||||||
LdSlot_ add_ui_self(R_TapePtr, S_(Binds_ResolveLookAtSub)),
|
|
||||||
|
|
||||||
/* Stage up_in.x/y/z into the scratchpad. R_ScratchBase = R_SP = 0x1F800000. */
|
|
||||||
mac_load_word_v3( r.t0, r.t1, r.t2, r.up_in, 0), LdSlot_
|
|
||||||
mac_store_word_v3(r.t0, r.t1, r.t2, R_ScratchBase, O_(ResolveLookAtScratch,up_in)),
|
|
||||||
|
|
||||||
// Stage eye.x/y/z into the scratchpad (atom 6 reads these for the translation column).
|
|
||||||
mac_load_word_v3( r.t0, r.t1, r.t2, r.eye, 0), LdSlot_
|
|
||||||
mac_store_word_v3(r.t0, r.t1, r.t2, R_ScratchBase, O_(ResolveLookAtScratch,eye)),
|
|
||||||
|
|
||||||
/* Compute fwd = target - eye. */
|
|
||||||
// mac_load_p3s4(t3, R_AT, t4, r.eye, 0),
|
|
||||||
mac_load_word_v3(r.t3, R_AT, r.t4, r.target, 0), LdSlot_
|
|
||||||
mac_sub_s_v3_self(
|
|
||||||
r.t3, R_AT, r.t4,
|
|
||||||
r.t0, r.t1, r.t2),
|
|
||||||
mac_store_word_v3(r.t3, R_AT, r.t4, R_ScratchBase, O_(ResolveLookAtScratch,fwd)),
|
|
||||||
|
|
||||||
mac_yield()
|
|
||||||
})
|
|
||||||
|
|
||||||
|
|
||||||
typedef Struct_(Binds_ResolveLookAtPopMvTrans) {
|
|
||||||
U4 look_at; /* MT3_S2S4* — destination matrix address */
|
|
||||||
};
|
|
||||||
typedef Struct_(RegUse_resolve_look_at__pop_mv_trans) {
|
|
||||||
Reg look_at;
|
|
||||||
Reg_(V3_S4) row; /* populate phase: load ux/uy/uz */
|
|
||||||
union { Reg ux, v_x; } t6; /* populate addr (canonical) → matrix_vector v_x */
|
|
||||||
union { Reg uy, v_y; } t7; /* populate uy → matrix_vector v_y */
|
|
||||||
union { Reg uz, v_z; } t8; /* populate uz → matrix_vector v_z */
|
|
||||||
Reg eye; /* matrix_vector phase: load -eye */
|
|
||||||
};
|
|
||||||
/* Atom 6 (fused): write look_at->m[][] from ux/uy/uz as packed S2 (populate),
|
|
||||||
* ctc2 RT chain into C2[0..4] (matrix_vector), MVMVA RT*(-eye)>>12, store off
|
|
||||||
* directly to look_at->t[] (trans_matrix). Replaces the previous 3 separate atoms
|
|
||||||
* (populate + matrix_vector + trans_matrix).
|
|
||||||
*
|
|
||||||
* MT3_S2S4 { A3x3_S2 m; A3_S4 t; }
|
|
||||||
* m[][] is S2 packed (9 × 2 = 18 bytes at offset 0)
|
|
||||||
* t[0..2] is S4 (3 × 4 = 12 bytes at offset 18)
|
|
||||||
*
|
|
||||||
* C11 ApplyMatrixLV semantics (gte.atom.c ac_apply_matrix_lv; libgte reference):
|
|
||||||
* 1. ctc2 RT matrix (5 ctc2s to C2[0..4])
|
|
||||||
* 2. lw -eye from memory
|
|
||||||
* 3. S15 decomposition (eliminated here — the fused body takes the >>12 path
|
|
||||||
* directly via mtc2 IR + MVMVA pass2, matching the libgte canonical output)
|
|
||||||
* 4. mtc2 to IR1/2/3, nop2, MVMVA pass2 (sf=1, mx=0, v=3, cv=3)
|
|
||||||
* 5. mfc2 MACs → off
|
|
||||||
* 6. store off to look_at->t[] (skip scratch.eye intermediate)
|
|
||||||
*
|
|
||||||
* GPR codes (assigned by resolve_look_at_init):
|
|
||||||
* r_scratch : R_ResolveScratch (R_T4 carrier)
|
|
||||||
* r_look_at : ralloc() — also serves as the off-dst in the trans_matrix phase
|
|
||||||
* r_row : V3_S4, reused for ux/uy/uz loads in populate phase
|
|
||||||
* r_eye : ralloc() — &scratch.eye, used for -eye load in matrix_vector phase
|
|
||||||
* r_v_x/v_y/v_z : ralloc() — populate scratch addrs (ux/uy/uz), reused as
|
|
||||||
* ctc2 transfer + MVMVA -eye temp in matrix_vector phase
|
|
||||||
* (v_x/v_y/v_z alias ux/uy/uz via the union; lifetime ends for ux/uy/uz after
|
|
||||||
* populate's mac_load_v3s4, so reusing for v.x/v.y/v.z is safe)
|
|
||||||
* Pool cost: 1 carrier + 1 look_at + 3 row + 1 eye + 3 aliased = 9 GPRs
|
|
||||||
*
|
|
||||||
* Net word savings vs the previous 3-atom flow: ~15 words + 2 mac_yields + 1 tape pop.
|
|
||||||
* - 2 mac_yields (trans_matrix's + matrix_vector's) → fused into one yield
|
|
||||||
* - 1 redundant tb_data (look_at was pushed 2x; now once)
|
|
||||||
* - mac_trans_mt3s3s4 (6 words) → replaced by direct mac_store_v3s4
|
|
||||||
* - mac_store_v3s4 to scratch.eye (3 words intermediate) → eliminated
|
|
||||||
* - add_si for r_off_ptr (2 words) → eliminated
|
|
||||||
* - mac_store_v3s4 zero-store of t[] (3 words) → eliminated (matrix_vector writes
|
|
||||||
* off directly; no consumer needed the zero first)
|
|
||||||
* - 1 set_gte_mt3s2s4 ctc2 chain (13 baked words) → eliminated (matrix_vector
|
|
||||||
* has its own ctc2 RT chain; cube rendering atoms reload C2 state themselves)
|
|
||||||
*/
|
|
||||||
internal MipsAtom* resolve_look_at__pop_mv_trans(AtomArena_R aa,
|
|
||||||
RegUse_resolve_look_at__pop_mv_trans r
|
|
||||||
) MipsAtom_Proc_(aa, {
|
|
||||||
/* --- Tape pop: look_at pointer --- */
|
|
||||||
load_word(r.look_at, R_TapePtr, O_(Binds_ResolveLookAtPopMvTrans,look_at)),
|
|
||||||
LdSlot_ add_ui_self(R_TapePtr, S_(Binds_ResolveLookAtPopMvTrans)),
|
|
||||||
|
|
||||||
/* --- Scratch addresses for ux/uy/uz/eye (populate phase; t6/t7/t8 alias ux/uy/uz).
|
|
||||||
* R_ScratchBase (= R_SP) holds 0x1F800000; no per-atom bake is required because
|
|
||||||
* R_SP is a tape carrier preserved across atoms. --- */
|
|
||||||
add_si(r.t6.ux, R_ScratchBase, O_(ResolveLookAtScratch, ux)), LdSlot_
|
|
||||||
add_si(r.t7.uy, R_ScratchBase, O_(ResolveLookAtScratch, uy)),
|
|
||||||
add_si(r.t8.uz, R_ScratchBase, O_(ResolveLookAtScratch, uz)),
|
|
||||||
add_si(r.eye, R_ScratchBase, O_(ResolveLookAtScratch, eye)),
|
|
||||||
|
|
||||||
/* --- POPULATE phase: write look_at->m[][] from ux/uy/uz as packed S2 --- */
|
|
||||||
mac_load_v3s4(r.row, r.t6.ux, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4, m[0])),
|
|
||||||
mac_load_v3s4(r.row, r.t7.uy, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4, m[1])),
|
|
||||||
mac_load_v3s4(r.row, r.t8.uz, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4, m[2])),
|
|
||||||
|
|
||||||
/* --- MATRIX-VECTOR phase: ctc2 RT chain + MVMVA RT*(-eye)>>12 --- */
|
|
||||||
/* RT packing (per libgte ApplyMatrixLV convention; see gte.h:217-220 +
|
|
||||||
* atom_6b_disasm_comparison.md:28-32):
|
|
||||||
* C2[0] = (RT12<<16)|RT11 ← ctc2 RT11 from m[0][0..1] packed word
|
|
||||||
* C2[1] = (RT21<<16)|RT13 ← ctc2 RT12 from m[0][2..3] packed word
|
|
||||||
* C2[2] = (RT23<<16)|RT22 ← ctc2 RT13 from m[1][1..2] packed word
|
|
||||||
* C2[3] = (RT32<<16)|RT31 ← ctc2 RT21 from m[2][0..1] packed word
|
|
||||||
* C2[4] = (RT33<<16)|junk ← ctc2 RT22 from m[2][2] (half)
|
|
||||||
* Each ctc2 writes a WHOLE 32-bit C2 slot; the "macro name" identifies
|
|
||||||
* which C2 register, not which 16-bit half. */
|
|
||||||
load_word( r.t6.v_x, r.look_at, O_(MT3_S2S4, m[0][0])), /* RT11|RT12 */ LdSlot_
|
|
||||||
load_word( r.t7.v_y, r.look_at, O_(MT3_S2S4, m[0][2])), /* RT13|RT21 */ LdSlot_ gte_mv_to_ctrl_r(r.t6.v_x, gte_cr_RT11),
|
|
||||||
load_word( r.t8.v_z, r.look_at, O_(MT3_S2S4, m[1][1])), /* RT22|RT23 */ LdSlot_ gte_mv_to_ctrl_r(r.t7.v_y, gte_cr_RT12),
|
|
||||||
load_word( r.t6.v_x, r.look_at, O_(MT3_S2S4, m[2][0])), /* RT31|RT32 */ LdSlot_ gte_mv_to_ctrl_r(r.t8.v_z, gte_cr_RT13),
|
|
||||||
load_half_u(r.t7.v_y, r.look_at, O_(MT3_S2S4, m[2][2])), /* RT33 */ LdSlot_ gte_mv_to_ctrl_r(r.t6.v_x, gte_cr_RT21),
|
|
||||||
/* pos = -eye. The three loads also retire the last CTC2. */ gte_mv_to_ctrl_r(r.t7.v_y, gte_cr_RT22),
|
|
||||||
GteDelay_ mac_load_word_v3(r.t6.v_x, r.t7.v_y, r.t8.v_z, r.eye, 0), LdSlot_
|
|
||||||
mac_sub_s_v3(r.t6.v_x, r.t7.v_y, r.t8.v_z, R_0, R_0, R_0, r.t6.v_x, r.t7.v_y, r.t8.v_z),
|
|
||||||
/* mtc2 pos (as S16) to IR1/2/3. The GTE takes low 16 bits. pos fits in S16. For negative pos, the 32-bit sign-extended value's low 16 bits = correct S16. */
|
|
||||||
gte_mv_to_data_r(r.t6.v_x, C2_IR1),
|
|
||||||
gte_mv_to_data_r(r.t7.v_y, C2_IR2),
|
|
||||||
gte_mv_to_data_r(r.t8.v_z, C2_IR3),
|
|
||||||
GteDelay_ nop2,
|
|
||||||
|
|
||||||
/* MVMVA pass 2 — C11 ApplyMatrixLV command.
|
|
||||||
* sf=1, mx=0 (RT), v=3 (IR), cv=3. Reads RT × IR >> 12. */
|
|
||||||
gte_cmdw_mvmva_c11_pass2, GteDelay_ nop,
|
|
||||||
mac_gte_mv_from_data_r_mac123(r.t6.v_x, r.t7.v_y, r.t8.v_z), GteDelay_ nop,
|
|
||||||
|
|
||||||
/* --- TRANS-MATRIX phase: store off directly to look_at->t[] (skip scratch.eye intermediate) --- */
|
|
||||||
mac_store_word_v3(r.t6.v_x, r.t7.v_y, r.t8.v_z, r.look_at, O_(MT3_S2S4, t)),
|
|
||||||
|
|
||||||
mac_yield()
|
|
||||||
})
|
|
||||||
#pragma endregion resolve_look_at
|
|
||||||
|
|
||||||
#pragma endregion Atom Procs
|
|
||||||
|
|
||||||
#pragma region Baked Atoms
|
#pragma region Baked Atoms
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
@@ -293,112 +105,112 @@ internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
|
|||||||
) {
|
) {
|
||||||
/* display[0] = (0, 0, 320, 240); rest of struct zeroed. */
|
/* display[0] = (0, 0, 320, 240); rest of struct zeroed. */
|
||||||
add_ui(R_ScreenX, R_0, ScreenRes_X), add_ui(R_ScreenY, R_0, ScreenRes_Y),
|
add_ui(R_ScreenX, R_0, ScreenRes_X), add_ui(R_ScreenY, R_0, ScreenRes_Y),
|
||||||
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area.width) + O_(DoubleBuffer,display[0])),
|
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area.width) + OA_(DoubleBuffer,display,0)),
|
||||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,display_area) + O_(DoubleBuffer,display[0])),
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,display_area) + OA_(DoubleBuffer,display,0)),
|
||||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + O_(DoubleBuffer,display[0])),
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + OA_(DoubleBuffer,display,0)),
|
||||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + O_(DoubleBuffer,display[0])),
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,0)),
|
||||||
|
|
||||||
/* display[1] = (0, 240, 320, 240); rest of struct zeroed. */
|
/* display[1] = (0, 240, 320, 240); rest of struct zeroed. */
|
||||||
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area) + O_(DoubleBuffer,display[1])),
|
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area) + OA_(DoubleBuffer,display,1)),
|
||||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + O_(DoubleBuffer,display[1])),
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + OA_(DoubleBuffer,display,1)),
|
||||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + O_(DoubleBuffer,display[1])),
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,1)),
|
||||||
|
|
||||||
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + O_(DoubleBuffer,draw[0])), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
|
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + OA_(DoubleBuffer,draw,0)), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
|
||||||
mac_store_v2s2( R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + O_(DoubleBuffer,draw[0])), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
|
mac_store_v2s2( R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + OA_(DoubleBuffer,draw,0)), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
|
||||||
|
|
||||||
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + O_(DoubleBuffer,draw[1])),
|
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
|
||||||
/* draw[0].texture_window = (0, 0, 0, 0); two word-zeroes cover the full 8-byte tw field. */
|
/* draw[0].texture_window = (0, 0, 0, 0); two word-zeroes cover the full 8-byte tw field. */
|
||||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + O_(DoubleBuffer,draw[0])),
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + OA_(DoubleBuffer,draw,0)),
|
||||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + O_(DoubleBuffer,draw[0])),
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,0)),
|
||||||
|
|
||||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,drawing_offset[0].x) + O_(DoubleBuffer,draw[1])),
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,drawing_offset[0].x) + OA_(DoubleBuffer,draw,1)),
|
||||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + O_(DoubleBuffer,draw[1])),
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + OA_(DoubleBuffer,draw,1)),
|
||||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + O_(DoubleBuffer,draw[1])),
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
|
||||||
/* draw[0].texture_page = 10 (gp0_tpage_default). C11 SetDefDrawEnv at C11_only.elf:0x8001273C writes the same 0x0A. . */
|
/* draw[0].texture_page = 10 (gp0_tpage_default). C11 SetDefDrawEnv at C11_only.elf:0x8001273C writes the same 0x0A. . */
|
||||||
add_ui(R_T0, R_0, gp0_tpage_default),
|
add_ui(R_T0, R_0, gp0_tpage_default),
|
||||||
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + O_(DoubleBuffer,draw[0])),
|
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,0)),
|
||||||
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + O_(DoubleBuffer,draw[1])),
|
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
|
||||||
/* draw[0] control bytes: flag_dither=1, flag_draw_on_display=1 (the dfe bit per psx-spx; libpsyx sets it via `SetDefDrawEnv`'s conditional at C11_only.elf:0x80012728), enable_auto_clear=1. Each byte is named;
|
/* draw[0] control bytes: flag_dither=1, flag_draw_on_display=1 (the dfe bit per psx-spx; libpsyx sets it via `SetDefDrawEnv`'s conditional at C11_only.elf:0x80012728), enable_auto_clear=1. Each byte is named;
|
||||||
* the previous `store_word(R_0, ..., +20)` overwrote all four with zero. */
|
* the previous `store_word(R_0, ..., +20)` overwrote all four with zero. */
|
||||||
add_ui(R_T0, R_0, 1),
|
add_ui(R_T0, R_0, 1),
|
||||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + O_(DoubleBuffer,draw[0])),
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + OA_(DoubleBuffer,draw,0)),
|
||||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + O_(DoubleBuffer,draw[0])),
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + OA_(DoubleBuffer,draw,0)),
|
||||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + O_(DoubleBuffer,draw[0])),
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,0)),
|
||||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + O_(DoubleBuffer,draw[1])),
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + OA_(DoubleBuffer,draw,1)),
|
||||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + O_(DoubleBuffer,draw[1])),
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + OA_(DoubleBuffer,draw,1)),
|
||||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + O_(DoubleBuffer,draw[1])),
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
|
||||||
/* draw[0].initial_bg_color = (r=7, g=7, b=7). */
|
/* draw[0].initial_bg_color = (r=7, g=7, b=7). */
|
||||||
add_ui(R_T0, R_0, 7),
|
add_ui(R_T0, R_0, 7),
|
||||||
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + O_(DoubleBuffer,draw[0])),
|
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,0)),
|
||||||
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + O_(DoubleBuffer,draw[1])),
|
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
|
||||||
mac_yield(),
|
mac_yield(),
|
||||||
};
|
};
|
||||||
|
|
||||||
/* gp_screen_init's GPR setup. Tests the mixed user-pinning + auto-reg pattern:
|
|
||||||
* - R_IO_BaseAddr = R_T4 (user-pinned via atom_reg; pre-existing)
|
|
||||||
* - R_GP1_Offset = R_T2 (user-pinned via atom_reg; NEW -- for GPIO_PORT1_OFFSET)
|
|
||||||
* - R_ScreenX = R_T5 (user-pinned via atom_reg; used as a transfer and GTE setup reg)
|
|
||||||
* - R_GpTmp = auto-allocated by the lua pass and used for several GPU transfers;
|
|
||||||
* the C preprocessor resolves it to the chosen free pool GPR.
|
|
||||||
*
|
|
||||||
* For gp_screen_init, the auto-reg pool exclusions are:
|
|
||||||
* user_pinned (from the corpus register_alias_registry) : R_T0..R_T7 (all 8 user-pinned across hello_camera.atom.c)
|
|
||||||
* body-parsed physical registers : aliases resolve through the registry;
|
|
||||||
* the body uses R_ScreenX, not raw R_T5
|
|
||||||
* source_pool after both subtractions : {R_V0, R_V1} only
|
|
||||||
* R_GpTmp gets R_V0 (the first-fit choice). Its repeated GPU-transfer use proves that the
|
|
||||||
* auto-reg allocation is active while the R_ScreenX references prove the pinned alias is used.
|
|
||||||
* R_TapePtr (R_T9), R_AtomJmp (R_T8), R_AT are excluded from the POOL by construction in
|
|
||||||
* passes/auto_reg.lua -- see the "obvious exclusions" comment block at the top of that file.
|
|
||||||
*/
|
|
||||||
enum {
|
enum {
|
||||||
R_IO_BaseAddr = R_T4 atom_reg, /* Caller-pinned: IO_BASE_ADDR = 0x1F800000 */
|
R_IO_BaseAddr = R_T4 atom_reg, /* Caller-pinned: IO_BASE_ADDR = 0x1F800000 */
|
||||||
R_GP1_Offset = R_T2 atom_reg, /* Caller-pinned: GPIO_PORT1_OFFSET = 0x10 */
|
|
||||||
atom_auto_reg(gp_screen_init, R_GpTmp), /* Auto-allocated scratch; resolved to a free pool GPR by the lua pass. C-preprocessor expands to R_GpTmp = R_GpTmp_Code with an atom_auto_reg trailing comment. */
|
|
||||||
#define R_IO_BaseAddr_Code R_T4_Code
|
#define R_IO_BaseAddr_Code R_T4_Code
|
||||||
#define R_GP1_Offset_Code R_T2_Code
|
|
||||||
};
|
};
|
||||||
internal MipsAtom_(gp_screen_init) atom_info(atom_phase(screen_init), atom_reads(R_IO_BaseAddr)) {
|
internal MipsAtom_(gp_screen_init) atom_info(atom_phase(screen_init), atom_reads(R_IO_BaseAddr)) {
|
||||||
store_word(R_0, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(00h) Reset */
|
store_word(R_0, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(00h) Reset */
|
||||||
mac_gcmd_push(gp1_word_ResetCmdBuffer(), R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(01h) ClearFIFO; uses pinned R_ScreenX as the transfer reg. */
|
mac_gcmd_push(gp1_word_ResetCmdBuffer(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(01h) ClearFIFO */
|
||||||
mac_gcmd_push(gp1_word_AcknowledgeIRQ(), R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(02h) AckIRQ; uses pinned R_ScreenX as the transfer reg. */
|
mac_gcmd_push(gp1_word_AcknowledgeIRQ(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(02h) AckIRQ */
|
||||||
mac_gcmd_push(gp1_word_DisplayOn(), R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(03h) Display ON; uses pinned R_ScreenX as the transfer reg. */
|
mac_gcmd_push(gp1_word_DisplayOn(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(03h) Display ON */
|
||||||
mac_gcmd_push(gp1_word_dma_to_gpu(), R_GpTmp, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(04h) DMADirection=2 (CPU->GPU). libpsyx's per-frame PutDrawEnv/DrawOTag use DMA2; without this the DMA queue never drains. Uses auto-allocated R_GpTmp. */
|
mac_gcmd_push(gp1_word_dma_to_gpu(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(04h) DMADirection=2 (CPU→GPU). libpsyx's per-frame PutDrawEnv/DrawOTag use DMA2; without this the DMA queue never drains. */
|
||||||
mac_gcmd_push(gp1_word_StartDisplayArea(), R_GpTmp, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(05h) StartDisplayArea (X=0, Y=0); uses auto-allocated R_GpTmp. */
|
mac_gcmd_push(gp1_word_StartDisplayArea(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(05h) StartDisplayArea (X=0, Y=0) */
|
||||||
|
|
||||||
/* GP1: DisplayMode + Display Ranges. */
|
/* GP1: DisplayMode + Display Ranges */
|
||||||
mac_gcmd_push(gp1_word_display_mode_320x240_15bit_ntsc, R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
mac_gcmd_push(gp1_word_display_mode_320x240_15bit_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||||
mac_gcmd_push(gp1_word_horizontal_range_ntsc, R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
mac_gcmd_push(gp1_word_horizontal_range_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||||
mac_gcmd_push(gp1_word_vertical_range_ntsc, R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
mac_gcmd_push(gp1_word_vertical_range_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||||
|
|
||||||
/* GTE: SetGeomOffset (OFX, OFY) — ScreenRes_CenterX, ScreenRes_CenterY. */
|
/* GTE: SetGeomOffset (OFX, OFY) — ScreenRes_CenterX, ScreenRes_CenterY. */
|
||||||
load_upper_i(R_ScreenX, ScreenRes_CenterX), gte_mv_to_ctrl_r(R_ScreenX, gte_cr_OFX_Code),
|
load_upper_i(R_T5, ScreenRes_CenterX), gte_mv_to_ctrl_r(R_T5, gte_cr_OFX_Code),
|
||||||
load_upper_i(R_ScreenX, ScreenRes_CenterY), gte_mv_to_ctrl_r(R_ScreenX, gte_cr_OFY_Code),
|
load_upper_i(R_T5, ScreenRes_CenterY), gte_mv_to_ctrl_r(R_T5, gte_cr_OFY_Code),
|
||||||
|
|
||||||
/* GTE: SetGeomScreen (H) — CR26 (per PSX-SPX / libpsyx), value is the raw projection-plane distance, NOT shifted. */
|
/* GTE: SetGeomScreen (H) — CR26 (per PSX-SPX / libpsyx), value is the raw projection-plane distance, NOT shifted. */
|
||||||
add_ui(R_ScreenX, R_0, ScreenZ), gte_mv_to_ctrl_r(R_ScreenX, gte_cr_H_Code),
|
add_ui(R_T5, R_0, ScreenZ), gte_mv_to_ctrl_r(R_T5, gte_cr_H_Code),
|
||||||
|
|
||||||
/* GP1: DisplayEnable — bit 0 = 0 (Display ON). */
|
/* GP1: DisplayEnable — bit 0 = 0 (Display ON). */
|
||||||
mac_gcmd_push(gp1_word_DisplayOn(), R_GpTmp, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* Uses auto-allocated R_GpTmp. */
|
mac_gcmd_push(gp1_word_DisplayOn(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||||
mac_yield(),
|
mac_yield(),
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/* ----- pad_apply_input -----
|
||||||
|
* Reads pad[0].buttons + pad[0].left_x;
|
||||||
|
* Applies the input-semantics deltas to cube_rot.y + floor_rot.y:
|
||||||
|
* - D-pad Left: cube_rot.y += 30, floor_rot.y += 5
|
||||||
|
* - D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5
|
||||||
|
* - Analog stick X (dead zone 0x70..0x90):
|
||||||
|
* cube delta = (0x80 - left_x) >> 2 (range approx -32..+32)
|
||||||
|
* floor delta = (0x80 - left_x) >> 5 (range approx -4..+4)
|
||||||
|
* - D-pad + analog deltas add when used together.
|
||||||
|
*
|
||||||
|
* Convention:
|
||||||
|
* pad_state = 0 means no buttons active.
|
||||||
|
* The fail-safe zero-button value flows through unchanged, so a disconnected/fresh pad produces no rotation.
|
||||||
|
* The branch_le_zero pattern below matches the existing pad_input_demo convention (atom body lines 248/257).
|
||||||
|
*
|
||||||
|
* Signed-delta trick:
|
||||||
|
* load_byte_u zero-extends left_x to 32 bits; sub_u from 0x80 wraps to a SIGNED two's-complement value in the negative range;
|
||||||
|
* shift_aright (sra) then correctly sign-extends the shift for both positive (left_x < 0x80) and negative (left_x > 0x80) cases.
|
||||||
|
* Digital pads publish left_x = 0x80 → delta = 0 → no rotation, so the analog step is naturally a no-op for digital controllers.
|
||||||
|
*/
|
||||||
typedef Struct_(Binds_PadApplyInput) {
|
typedef Struct_(Binds_PadApplyInput) {
|
||||||
PadState* state;
|
PadState* state;
|
||||||
V3_S2* cube_rot;
|
V3_S2* cube_rot;
|
||||||
V3_S2* floor_rot;
|
V3_S2* floor_rot;
|
||||||
};
|
};
|
||||||
enum {
|
enum {
|
||||||
R_PadStateT5 = R_T5 atom_reg,
|
R_PadStateT5 = R_T5 atom_reg,
|
||||||
R_CubeRot = R_T1 atom_reg,
|
R_CubeRot = R_T1 atom_reg,
|
||||||
R_FloorRot = R_T2 atom_reg,
|
R_FloorRot = R_T2 atom_reg,
|
||||||
};
|
};
|
||||||
internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyInput)
|
internal MipsAtom_(pad_apply_input) atom_info(atom_bind(Binds_PadApplyInput)
|
||||||
, atom_reads(R_T0, R_CubeRot, R_FloorRot, R_T3, R_T4, R_PadStateT5, R_TapePtr)
|
, atom_reads(R_T0, R_CubeRot, R_FloorRot, R_T3, R_T4, R_PadStateT5, R_TapePtr)
|
||||||
, atom_writes( R_CubeRot, R_FloorRot)
|
, atom_writes( R_CubeRot, R_FloorRot)
|
||||||
) {
|
) {
|
||||||
@@ -406,15 +218,15 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
|
|||||||
load_word(R_PadStateT5, R_TapePtr, O_(Binds_PadApplyInput,state)),
|
load_word(R_PadStateT5, R_TapePtr, O_(Binds_PadApplyInput,state)),
|
||||||
load_word(R_CubeRot, R_TapePtr, O_(Binds_PadApplyInput,cube_rot)),
|
load_word(R_CubeRot, R_TapePtr, O_(Binds_PadApplyInput,cube_rot)),
|
||||||
load_word(R_FloorRot, R_TapePtr, O_(Binds_PadApplyInput,floor_rot)),
|
load_word(R_FloorRot, R_TapePtr, O_(Binds_PadApplyInput,floor_rot)),
|
||||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_PadApplyInput)),
|
add_ui_self( R_TapePtr, S_(Binds_PadApplyInput)),
|
||||||
|
|
||||||
/* Load pad[0].buttons into R_T0. */
|
/* Load pad[0].buttons into R_T0. */
|
||||||
load_word(R_T0, R_PadStateT5, O_(PadState,buttons)), LdSlot_ nop,
|
load_word(R_T0, R_PadStateT5, O_(PadState,buttons)), nop,
|
||||||
// Note(Ed): Potential op with delay slot?
|
// Note(Ed): Potential op with delay slot?
|
||||||
|
|
||||||
/* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */
|
/* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */
|
||||||
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)), BdSlot_
|
and_i(R_T3, R_T0, pad0_(Pad_Left)), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)),
|
||||||
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), LdSlot_
|
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
|
||||||
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||||
add_si( R_T4, R_T4, 30),
|
add_si( R_T4, R_T4, 30),
|
||||||
add_si( R_T3, R_T3, 5),
|
add_si( R_T3, R_T3, 5),
|
||||||
@@ -423,8 +235,8 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
|
|||||||
atom_label(exit_dpad_left)
|
atom_label(exit_dpad_left)
|
||||||
|
|
||||||
/* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */
|
/* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */
|
||||||
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)), BdSlot_
|
and_i(R_T3, R_T0, pad0_(Pad_Right)), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)),
|
||||||
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), LdSlot_
|
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
|
||||||
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||||
add_si( R_T4, R_T4, -30),
|
add_si( R_T4, R_T4, -30),
|
||||||
add_si( R_T3, R_T3, -5),
|
add_si( R_T3, R_T3, -5),
|
||||||
@@ -434,23 +246,23 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
|
|||||||
|
|
||||||
/* Analog left-stick X: dead zone 0x70..0x90.
|
/* Analog left-stick X: dead zone 0x70..0x90.
|
||||||
* Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */
|
* Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */
|
||||||
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), LdSlot_ //?
|
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left_x)),
|
||||||
|
|
||||||
/* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly).
|
/* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly).
|
||||||
* set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */
|
* set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */
|
||||||
add_ui(R_T4, R_0, PadDeadZone_HighBound), set_lt_u(R_T4, R_T3, R_T4), branch_ne(R_T4, R_0, atom_offset(dead_zone_low_check, dead_low_active)),
|
add_ui(R_T4, R_0, 0x70), set_lt_u(R_T4, R_T3, R_T4), branch_ne(R_T4, R_0, atom_offset(dead_zone_low_check, dead_low_active)),
|
||||||
add_ui(R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_low_active */
|
add_ui(R_T4, R_0, 0x80), /* BD-slot: pre-load 0x80 for dead_low_active */
|
||||||
|
|
||||||
atom_label(dead_check_upper)
|
atom_label(dead_check_upper)
|
||||||
/* left_x >= 0x70 → check upper bound. */
|
/* left_x >= 0x70 → check upper bound. */
|
||||||
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), /* reload */ LdSlot_ //?
|
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left_x)), /* reload */
|
||||||
add_ui( R_T4, R_0, PadDeadZone_HighBound),
|
add_ui( R_T4, R_0, 0x90),
|
||||||
|
|
||||||
/* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */
|
/* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */
|
||||||
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)), BdSlot_
|
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)),
|
||||||
add_ui( R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_high_active */
|
add_ui( R_T4, R_0, 0x80), /* BD-slot: pre-load 0x80 for dead_high_active */
|
||||||
jump_rel(atom_offset(dead_zone_skip, exit_stick)),
|
jump_rel(atom_offset(dead_zone_skip, exit_stick)),
|
||||||
BdSlot_ mac_yield_load(), LdSlot_
|
mac_yield_load(),
|
||||||
|
|
||||||
atom_label(dead_low_active)
|
atom_label(dead_low_active)
|
||||||
/* R_T3 = left_x (from line 632 lbu; not clobbered between dead_zone_low_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
/* R_T3 = left_x (from line 632 lbu; not clobbered between dead_zone_low_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
||||||
@@ -461,18 +273,19 @@ atom_label(dead_low_active)
|
|||||||
|
|
||||||
/* R_T4 = cube_delta */
|
/* R_T4 = cube_delta */
|
||||||
shift_aright(R_T4, R_T3, 2),
|
shift_aright(R_T4, R_T3, 2),
|
||||||
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), LdSlot_ nop,
|
load_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||||
|
nop,
|
||||||
add_u( R_T0, R_T0, R_T4),
|
add_u( R_T0, R_T0, R_T4),
|
||||||
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||||
/* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap;
|
/* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap;
|
||||||
* doesn't read R_T0; R_T4 settles by the subsequent add_u). */
|
* doesn't read R_T0; R_T4 settles by the subsequent add_u). */
|
||||||
load_half( R_T0, R_FloorRot, O_(V3_S2,y)), LdSlot_
|
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||||
shift_aright(R_T4, R_T3, 5),
|
shift_aright(R_T4, R_T3, 5),
|
||||||
add_u( R_T0, R_T0, R_T4),
|
add_u( R_T0, R_T0, R_T4),
|
||||||
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
|
||||||
jump_rel(atom_offset(end_low, exit_stick)),
|
jump_rel(atom_offset(end_low, exit_stick)),
|
||||||
BdSlot_ mac_yield_load(), LdSlot_
|
mac_yield_load(),
|
||||||
|
|
||||||
atom_label(dead_high_active)
|
atom_label(dead_high_active)
|
||||||
/* R_T3 = left_x (from line 641 lbu in dead_check_upper; not clobbered between dead_zone_high_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
/* R_T3 = left_x (from line 641 lbu in dead_check_upper; not clobbered between dead_zone_high_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
||||||
@@ -482,18 +295,19 @@ atom_label(dead_high_active)
|
|||||||
/* delta = 0x80 - left_x (signed negative). */
|
/* delta = 0x80 - left_x (signed negative). */
|
||||||
|
|
||||||
shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */
|
shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */
|
||||||
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), LdSlot_ nop,
|
load_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||||
|
nop,
|
||||||
add_u( R_T0, R_T0, R_T4),
|
add_u( R_T0, R_T0, R_T4),
|
||||||
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||||
|
|
||||||
/* R_T4 = floor_delta (signed) — moved into the load-delay slot of the floor load below. */
|
/* R_T4 = floor_delta (signed) — moved into the load-delay slot of the floor load below. */
|
||||||
load_half( R_T0, R_FloorRot, O_(V3_S2,y)), LdSlot_
|
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||||
shift_aright(R_T4, R_T3, 5),
|
shift_aright(R_T4, R_T3, 5),
|
||||||
add_u( R_T0, R_T0, R_T4),
|
add_u( R_T0, R_T0, R_T4),
|
||||||
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
|
||||||
atom_label(no_jump_fallthrough)
|
atom_label(no_jump_fallthrough)
|
||||||
mac_yield_load(), LdSlot_
|
mac_yield_load(),
|
||||||
|
|
||||||
atom_label(exit_stick)
|
atom_label(exit_stick)
|
||||||
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the dead-zone/exit branch. */
|
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the dead-zone/exit branch. */
|
||||||
@@ -501,63 +315,7 @@ atom_label(exit_stick)
|
|||||||
};
|
};
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
R_Cam = R_T4 atom_reg,
|
R_PrimCursor = R_T7 atom_reg atom_type(U4*), /* VRAM output cursor (primitive buffer) */
|
||||||
R_CamPadState = R_T5 atom_reg,
|
|
||||||
};
|
|
||||||
typedef Struct_(Binds_PadInputCam) {
|
|
||||||
PadState* state;
|
|
||||||
Camera* cam;
|
|
||||||
};
|
|
||||||
internal MipsAtom_(pad_input_cam) atom_info(atom_bind(Binds_PadInputCam)
|
|
||||||
, atom_reads( R_Cam, R_CamPadState, R_TapePtr)
|
|
||||||
, atom_writes(R_Cam)
|
|
||||||
) {
|
|
||||||
/* Bind pop: state → R_CamPadState (R_T5), cam → R_Cam (R_T4), advance R_TapePtr by 8. */
|
|
||||||
load_word(R_CamPadState, R_TapePtr, O_(Binds_PadInputCam,state)),
|
|
||||||
load_word(R_Cam, R_TapePtr, O_(Binds_PadInputCam,cam)),
|
|
||||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_PadInputCam)),
|
|
||||||
|
|
||||||
/* Load pad[0].buttons into R_T0; nop fills the load-delay slot. */
|
|
||||||
load_word(R_T0, R_CamPadState, O_(PadState,buttons)), LdSlot_
|
|
||||||
load_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
|
||||||
|
|
||||||
// D-pad Left → cam.pos.x -= 50. and_i fulfills BD-slot for load on R_Cam.
|
|
||||||
LdSlot_ and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(left_x, exit_left_x)), BdSlot_ mac_yield_load(), LdSlot_
|
|
||||||
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
|
||||||
atom_label(exit_left_x)
|
|
||||||
|
|
||||||
/* D-pad Right → cam.pos.x += 50. Reuses R_T1 from Left. */
|
|
||||||
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(right_x, exit_right_x)), BdSlot_ nop,
|
|
||||||
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
|
||||||
atom_label(exit_right_x)
|
|
||||||
|
|
||||||
/* D-pad Up → cam.pos.y -= 50. Load pos.y BEFORE the andi. */
|
|
||||||
load_word(R_T1, R_Cam, O_(Camera,pos.y)), LdSlot_
|
|
||||||
and_i(R_T3, R_T0, Pad_Up), branch_le_zero(R_T3, atom_offset(up_y, exit_up_y)), BdSlot_ nop,
|
|
||||||
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
|
|
||||||
atom_label(exit_up_y)
|
|
||||||
|
|
||||||
/* D-pad Down → cam.pos.y += 50. Reuses R_T1 from Up. */
|
|
||||||
and_i(R_T3, R_T0, Pad_Down), branch_le_zero(R_T3, atom_offset(down_y, exit_down_y)), BdSlot_ nop,
|
|
||||||
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
|
|
||||||
atom_label(exit_down_y)
|
|
||||||
|
|
||||||
/* D-pad Cross → cam.pos.z -= 50. Load pos.z BEFORE the andi. */
|
|
||||||
load_word(R_T1, R_Cam, O_(Camera,pos.z)), LdSlot_
|
|
||||||
and_i(R_T3, R_T0, Pad_Cross), branch_le_zero(R_T3, atom_offset(cross_z, exit_cross_z)), BdSlot_ nop,
|
|
||||||
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
|
|
||||||
atom_label(exit_cross_z)
|
|
||||||
|
|
||||||
/* D-pad Circle → cam.pos.z += 50. Reuses R_T1 from Cross. */
|
|
||||||
and_i(R_T3, R_T0, Pad_Circle), branch_le_zero(R_T3, atom_offset(circle_z, exit_circle_z)), BdSlot_ nop,
|
|
||||||
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
|
|
||||||
atom_label(exit_circle_z)
|
|
||||||
|
|
||||||
mac_yield_tail(),
|
|
||||||
};
|
|
||||||
|
|
||||||
enum {
|
|
||||||
R_PrimCursor = R_T7 atom_reg atom_type(U4*), /* Output cursor (primitive buffer) */
|
|
||||||
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2*), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
|
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2*), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
|
||||||
R_VertBase = R_T5 atom_reg atom_type(V3_S2*), /* Base address of the vertex array */
|
R_VertBase = R_T5 atom_reg atom_type(V3_S2*), /* Base address of the vertex array */
|
||||||
R_OtBase = R_T6 atom_reg atom_type(U4*), /* Base address of the Ordering Table */
|
R_OtBase = R_T6 atom_reg atom_type(U4*), /* Base address of the Ordering Table */
|
||||||
@@ -566,6 +324,7 @@ enum {
|
|||||||
#define R_VertBase_Code R_T5_Code
|
#define R_VertBase_Code R_T5_Code
|
||||||
#define R_OtBase_Code R_T6_Code
|
#define R_OtBase_Code R_T6_Code
|
||||||
};
|
};
|
||||||
|
|
||||||
typedef Struct_(Binds_CubeTri) {
|
typedef Struct_(Binds_CubeTri) {
|
||||||
U4 PrimCursor;
|
U4 PrimCursor;
|
||||||
V4_S2* FaceCursor;
|
V4_S2* FaceCursor;
|
||||||
@@ -581,11 +340,11 @@ internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_
|
|||||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
|
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
|
||||||
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
|
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
|
||||||
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
|
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
|
||||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
||||||
mac_yield()
|
mac_yield()
|
||||||
};
|
};
|
||||||
|
|
||||||
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
|
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
|
||||||
internal
|
internal
|
||||||
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||||
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
||||||
@@ -594,20 +353,20 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
|||||||
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||||
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
||||||
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
||||||
// load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
||||||
|
|
||||||
LdSlot_ mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2), GteDelay_ load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)), LdSlot_
|
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
||||||
GteDelay_ nop, gte_cmdw_rotate_translate_perspective_triple,
|
nop2, gte_cmdw_rotate_translate_perspective_triple, // required cpu -> gte delay slot
|
||||||
gte_cmdw_nclip,
|
gte_cmdw_nclip,
|
||||||
|
|
||||||
gte_mv_from_data_r(R_T0, C2_MAC0), GteDelay_ nop,
|
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
|
||||||
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
|
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
|
||||||
/* BD-slot: Write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
|
/* BD-slot: write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
|
||||||
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
|
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
|
||||||
* harmless because the OT entry that points to this prim is created later. */
|
* harmless because the OT entry that points to this prim is created later, only on the body path. */
|
||||||
BdSlot_ store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
||||||
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
||||||
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)), LdSlot_
|
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
|
||||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||||
|
|
||||||
mac_gte_store_g4_p012(R_PrimCursor),
|
mac_gte_store_g4_p012(R_PrimCursor),
|
||||||
@@ -619,8 +378,8 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
|||||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||||
set_lt_u( R_AT, R_T1, R_AT),
|
set_lt_u( R_AT, R_T1, R_AT),
|
||||||
|
|
||||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), BdSlot_ nop,
|
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
|
||||||
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_G4)),
|
mac_insert_ot_tag_g4(R_OtBase, R_PrimCursor),
|
||||||
mac_format_g4_color(R_PrimCursor,
|
mac_format_g4_color(R_PrimCursor,
|
||||||
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
||||||
/* c1 yellow */ 0xFF, 0xFF, 0x00,
|
/* c1 yellow */ 0xFF, 0xFF, 0x00,
|
||||||
@@ -651,7 +410,7 @@ MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(f
|
|||||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
|
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
|
||||||
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
|
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
|
||||||
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
|
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
|
||||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
||||||
mac_yield()
|
mac_yield()
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -661,7 +420,7 @@ MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
|||||||
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||||
, atom_writes(R_PrimCursor, R_FaceCursor)
|
, atom_writes(R_PrimCursor, R_FaceCursor)
|
||||||
) {
|
) {
|
||||||
mac_load_tri_indices(R_FaceCursor, R_T0, R_T1, R_T2),
|
mac_load_tri_indices( R_FaceCursor, R_T0, R_T1, R_T2),
|
||||||
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
||||||
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
|
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
|
||||||
gte_cmdw_nclip,
|
gte_cmdw_nclip,
|
||||||
@@ -680,7 +439,7 @@ MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
|||||||
set_lt_u( R_AT, R_T1, R_AT),
|
set_lt_u( R_AT, R_T1, R_AT),
|
||||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
|
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
|
||||||
mac_format_f3_color(R_PrimCursor, 0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
|
mac_format_f3_color(R_PrimCursor, 0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
|
||||||
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_F3)), /* Insert into Ordering Table Linked List */
|
mac_insert_ot_tag_f3(R_OtBase, R_PrimCursor), /* Insert into Ordering Table Linked List */
|
||||||
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
|
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
|
||||||
// Note(Ed): No bounds checking, should be checked before atom runs.
|
// Note(Ed): No bounds checking, should be checked before atom runs.
|
||||||
// end: branch(bounds_chk)
|
// end: branch(bounds_chk)
|
||||||
@@ -698,7 +457,7 @@ internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitive
|
|||||||
, atom_writes(R_TapePtr)
|
, atom_writes(R_TapePtr)
|
||||||
){
|
){
|
||||||
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
|
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
|
||||||
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)), LdSlot_
|
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
|
||||||
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
|
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
|
||||||
/* Calculate byte offset and store directly back to RAM */
|
/* Calculate byte offset and store directly back to RAM */
|
||||||
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
|
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
|
||||||
|
|||||||
+116
-232
@@ -1,7 +1,7 @@
|
|||||||
#pragma region Vendors
|
#pragma region Vendors
|
||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
// #include <assert.h>
|
#include <assert.h>
|
||||||
// #include "libgpu.h"
|
// #include "libgpu.h"
|
||||||
// #include "libetc.h"
|
// #include "libetc.h"
|
||||||
// #include "libgte.h"
|
// #include "libgte.h"
|
||||||
@@ -26,13 +26,11 @@
|
|||||||
#include "duffle/dsl.atom.h"
|
#include "duffle/dsl.atom.h"
|
||||||
#include "duffle/lottes_tape.h"
|
#include "duffle/lottes_tape.h"
|
||||||
|
|
||||||
#include "duffle/bios.h"
|
|
||||||
#include "duffle/psyq.h"
|
#include "duffle/psyq.h"
|
||||||
#pragma endregion Duffle Headers
|
#pragma endregion Duffle Headers
|
||||||
|
|
||||||
#pragma region Duffle TUs
|
#pragma region Duffle TUs
|
||||||
#include "duffle/pad.c"
|
#include "duffle/math.atom.c"
|
||||||
#include "duffle/math.atom.h"
|
|
||||||
#include "duffle/mips.atom.c"
|
#include "duffle/mips.atom.c"
|
||||||
#include "duffle/gte.atom.c"
|
#include "duffle/gte.atom.c"
|
||||||
#include "duffle/gp.atom.c"
|
#include "duffle/gp.atom.c"
|
||||||
@@ -43,7 +41,6 @@
|
|||||||
#pragma region Hello Camera Headers
|
#pragma region Hello Camera Headers
|
||||||
# include "gen/macs.h"
|
# include "gen/macs.h"
|
||||||
# include "gen/offsets.h"
|
# include "gen/offsets.h"
|
||||||
# include "gen/auto_reg.h"
|
|
||||||
|
|
||||||
#include "hello_camera.h"
|
#include "hello_camera.h"
|
||||||
#pragma endregion Hello Camera Headers
|
#pragma endregion Hello Camera Headers
|
||||||
@@ -53,13 +50,8 @@
|
|||||||
#pragma endregion Hello Joypad TUs
|
#pragma endregion Hello Joypad TUs
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
MemTape_Len = 512,
|
Scratchpad_Len = 1024,
|
||||||
|
MemTape_Len = 512,
|
||||||
ResolveLookAtArena_Words = 1024,
|
|
||||||
ResolveLookAtArena_Size = ResolveLookAtArena_Words * S_(MipsCode),
|
|
||||||
|
|
||||||
CT_InitAtomMem_Words = Kilo_(4),
|
|
||||||
CT_InitAtomMem_Size = CT_InitAtomMem_Words * S_(MipsCode),
|
|
||||||
};
|
};
|
||||||
typedef Struct_(SMemory) {
|
typedef Struct_(SMemory) {
|
||||||
PrimitiveArena primitives;
|
PrimitiveArena primitives;
|
||||||
@@ -69,10 +61,7 @@ typedef Struct_(SMemory) {
|
|||||||
|
|
||||||
U4 MemTape[MemTape_Len];
|
U4 MemTape[MemTape_Len];
|
||||||
|
|
||||||
MT3_S2S4 tform_world;
|
M3_S2 tform_world;
|
||||||
MT3_S2S4 tform_view;
|
|
||||||
|
|
||||||
Camera cam;
|
|
||||||
|
|
||||||
Ent_Cube cube;
|
Ent_Cube cube;
|
||||||
Ent_Floor floor;
|
Ent_Floor floor;
|
||||||
@@ -80,22 +69,11 @@ typedef Struct_(SMemory) {
|
|||||||
PadBiosRaw pad_raw[2];
|
PadBiosRaw pad_raw[2];
|
||||||
PadState pad[2];
|
PadState pad[2];
|
||||||
|
|
||||||
// TODO(Ed): We don't need this we can just cast at any point an address to a desired view of scratchpad, we have the address.
|
|
||||||
U4_V scratchpad; // d-cache
|
U4_V scratchpad; // d-cache
|
||||||
|
|
||||||
U1 ct_init_atom_mem[CT_InitAtomMem_Size];
|
|
||||||
MipsAtom* normalize_v3s4;
|
|
||||||
MipsAtom* gte_cross_v3s4;
|
|
||||||
|
|
||||||
U1 resolve_look_at_mem[ResolveLookAtArena_Size];
|
|
||||||
MipsAtom* resolve_look_at_bundle[AtomBundle_Len(resolve_look_at)];
|
|
||||||
};
|
};
|
||||||
global SMemory smem;
|
global SMemory smem;
|
||||||
extern SMemory smem;
|
extern SMemory smem;
|
||||||
|
|
||||||
#define pad0_btn_(btn) btn & smem.pad[0].buttons
|
|
||||||
#define pad1_btn_(btn) btn & smem.pad[1].buttons
|
|
||||||
|
|
||||||
I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
|
I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
|
||||||
gknown PrimitiveArena* pa = & smem.primitives;
|
gknown PrimitiveArena* pa = & smem.primitives;
|
||||||
gknown B1* buf = (B1*) r_(smem.primitives.buf)[smem.active_buf_id];
|
gknown B1* buf = (B1*) r_(smem.primitives.buf)[smem.active_buf_id];
|
||||||
@@ -106,152 +84,75 @@ I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
|
|||||||
}
|
}
|
||||||
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
|
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
|
||||||
|
|
||||||
I_ void resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4* up_in) {
|
/* Uses ONE 8-byte frame allocated via the compiler's standard prologue.
|
||||||
// RGA(Lengyel): Build matrix expansion of a rigid transformation. Corresponding motor is not constructed; we write the LA form for GTE.
|
* The 4 wasted-arg words for B(12h) InitPAD2 live at [SP+0..15] but are not explicitly allocated.
|
||||||
// Preconditions: eye != target, up_in not collinear with (target - eye).
|
* The compiler handles the MIPS O32 "wasted stack" convention for us by treating the B-call as a 4-arg call.
|
||||||
V3_S4 right, up, forward;
|
*
|
||||||
V3_S4 ux, uy, uz;
|
* The buffer pointers are passed as arguments so the compiler keeps them in callee-saved registers;
|
||||||
V3_S4 pos, off;
|
* The B(12h) asm volatile block does NOT clobber those registers (it clobbers only the volatile GPRs + the B-table arg registers explicitly).
|
||||||
|
* The C-level writes after the call re-load the pointers from their callee-saved homes.
|
||||||
|
*
|
||||||
|
* The clobber list for both B-calls names the full BIOS destroy set documented in kernelbios.md:167-174 (R1..R15, R24..R25, R31, HI/LO).
|
||||||
|
* The kernel-ABI "volatile GPRs" subset is clb_system; the rest of the destroy set is enumerated explicitly here. */
|
||||||
|
NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
|
||||||
|
{
|
||||||
|
/* Pin raw0 + raw1 to $a0 + $a1 via rgcc; the B(12h) call uses these directly.
|
||||||
|
* The `(void)` casts mark them as unread after the call so the compiler doesn't need to move them back. */
|
||||||
|
register PadBiosRaw* p0 rgcc(R_A0) = raw0;
|
||||||
|
register PadBiosRaw* p1 rgcc(R_A1) = raw1;
|
||||||
|
(void)p0; (void)p1;
|
||||||
|
|
||||||
forward = target[0]; sub_v3s4(& forward, eye[0]); // RGA(Lengyel): Affine point - point = zero-weight direction.
|
// TODO(Ed): Properly annotate the raw values in the inline asm instructions.
|
||||||
normalize_v3s4(& forward, & uz); // RGA(Lengyel): Normalize the direction bulk. Not finite-point unitization.
|
// Use enums.
|
||||||
|
|
||||||
cross_v3s4(& uz, up_in, & right); normalize_v3s4(& right, & ux); // RGA(Lengyel): Complement(Wedge(forward, up_in)) -> right axis.
|
/* B(12h) InitPAD2(raw0, 0x22, raw1, 0x22)
|
||||||
cross_v3s4(& uz, & ux, & up); normalize_v3s4(& up, & uy); // RGA(Lengyel): Complement(Wedge(forward, right)) -> up axis.
|
* $a0 = raw0 (rgcc-bound; survives the sequence below)
|
||||||
|
* $a1 = raw1 (preserved into $a2 before $a1 is overwritten)
|
||||||
|
* $a2 = raw1 (moved from $a1; survives $a1's overwrite)
|
||||||
|
* $a3 = 0x22 (immediate)
|
||||||
|
* $t1 = 0x12 (function number)
|
||||||
|
* $t2 = 0xB0 (BIOS B-table address) */
|
||||||
|
asm volatile(
|
||||||
|
asm_words(
|
||||||
|
or_u( rarg_2, rarg_1, rdiscard), /* $a2 = $a1 = raw1 */
|
||||||
|
add_ui( rarg_1, rdiscard, 0x22), /* $a1 = 0x22 */
|
||||||
|
add_ui( rarg_3, rdiscard, 0x22), /* $a3 = 0x22 */
|
||||||
|
add_ui( rtmp_1, rdiscard, 0x12), /* $t1 = 0x12 */
|
||||||
|
add_ui( rtmp_2, rdiscard, 0xB0), /* $t2 = 0xB0 */
|
||||||
|
call_reg(rtmp_2), /* jalr $t2, $ra */
|
||||||
|
nop /* BD slot */
|
||||||
|
)
|
||||||
|
asm_rpins, r_use(p0), r_use(p1)
|
||||||
|
asm_clobber:
|
||||||
|
rlit(R_AT),
|
||||||
|
rlit(R_V0), rlit(R_V1),
|
||||||
|
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||||
|
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
||||||
|
rlit(R_RA),
|
||||||
|
clb_mem_drain
|
||||||
|
);
|
||||||
|
|
||||||
// RGA(Lengyel): matrix expansion of the world-to-camera rotation (basis rows).
|
/* The C-level writes re-load the pointers via the parameter names and write 0xFF to each
|
||||||
look_at->m[0][0] = ux.x; look_at->m[0][1] = ux.y; look_at->m[0][2] = ux.z;
|
* buffer's status byte to mark the initial-state hazard documented in kernelbios.md:1621-1624. */
|
||||||
look_at->m[1][0] = uy.x; look_at->m[1][1] = uy.y; look_at->m[1][2] = uy.z;
|
u1_v(raw0)[0] = 0xFF;
|
||||||
look_at->m[2][0] = uz.x; look_at->m[2][1] = uz.y; look_at->m[2][2] = uz.z;
|
u1_v(raw1)[0] = 0xFF;
|
||||||
|
|
||||||
pos = eye[0]; mul_v3s4(& pos, v3s4(-1,-1,-1)); // RGA(Lengyel): -eye in world coordinates (spatial bulk only; implicit weight is dropped).
|
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */
|
||||||
|
asm volatile(
|
||||||
// RGA(Lengyel): R * (-eye) is the full matrix translation column.
|
asm_words(
|
||||||
// Motor translator would store half this displacement in m.xyz; GTE consumes full column.
|
add_ui( rtmp_1, rdiscard, 0x13), /* $t1 = 0x13 */
|
||||||
mul_m3s2_v3s4(look_at, & pos, & off);
|
add_ui( rtmp_2, rdiscard, 0xB0), /* $t2 = 0xB0 (re-load) */
|
||||||
trans_m3s2( look_at, & off);
|
call_reg(rtmp_2), /* jalr $t2, $ra */
|
||||||
}
|
nop /* BD slot */
|
||||||
FI_ void camera_look_at_c11(Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at_c11(& c->look_at, & c->pos, target, up_in); }
|
)
|
||||||
|
asm_clobber:
|
||||||
internal void compile_init_atoms(void) {
|
rlit(R_AT),
|
||||||
AtomArena ab = atomarena_make(slice_ut_arr(smem.ct_init_atom_mem));
|
rlit(R_V0), rlit(R_V1),
|
||||||
RegFile rf = regfile(regfile_abi_mask);
|
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||||
#define ralloc() regfile_alloc(& rf)
|
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
||||||
#define ralloc_v3() { ralloc(), ralloc(), ralloc() }
|
rlit(R_RA),
|
||||||
|
clb_mem_drain
|
||||||
smem.gte_cross_v3s4 = gte_cross_v3s4(& ab,
|
);
|
||||||
RegUse_(gte_cross_v3s4) {
|
|
||||||
.a = ralloc_v3(),
|
|
||||||
.b = ralloc_v3(),
|
|
||||||
.x = ralloc(),
|
|
||||||
.y = ralloc(),
|
|
||||||
.z = ralloc(),
|
|
||||||
});
|
|
||||||
regfile_reset(& rf);
|
|
||||||
|
|
||||||
smem.normalize_v3s4 = build_normalize_v3s4(& ab,
|
|
||||||
RegUse_(build_normalize_v3s4) {
|
|
||||||
.scratch = ralloc(),
|
|
||||||
.src_ptr = ralloc(),
|
|
||||||
.dst_ptr = ralloc(),
|
|
||||||
.recip_est = ralloc(),
|
|
||||||
.norm = ralloc(),
|
|
||||||
.shift = ralloc(),
|
|
||||||
.src_x = ralloc(),
|
|
||||||
// .shift_count = ralloc(), /* dedicated slot for stage-3 → stage-4 shift count */
|
|
||||||
.t3 = ralloc(),
|
|
||||||
.t4 = ralloc(),
|
|
||||||
.t5 = ralloc(),
|
|
||||||
});
|
|
||||||
regfile_reset(& rf);
|
|
||||||
|
|
||||||
assert(ab.used <= CT_InitAtomMem_Size);
|
|
||||||
#undef ralloc
|
|
||||||
#undef ralloc_v3
|
|
||||||
}
|
|
||||||
|
|
||||||
internal void compile_resolve_look_at(void) {
|
|
||||||
AtomArena ab = atomarena_make(slice_ut_arr(smem.resolve_look_at_mem));
|
|
||||||
AtomBundle_resolve_look_at_R bundle = C_(void*, smem.resolve_look_at_bundle);
|
|
||||||
|
|
||||||
/* R_ScratchBase (= R_SP) is a tape carrier preserved across atoms; no carrier
|
|
||||||
* pin is needed in the regfile. The standard 24-register pool is sufficient. */
|
|
||||||
RegFile rf = regfile(regfile_abi_mask);
|
|
||||||
#define ralloc() regfile_alloc(& rf)
|
|
||||||
#define ralloc_v3() { ralloc(), ralloc(), ralloc() }
|
|
||||||
|
|
||||||
bundle->input_and_sub = AtomBundleEntry_(resolve_look_at, input_and_sub)(& ab,
|
|
||||||
RegUse_(resolve_look_at_input_and_sub) {
|
|
||||||
.target = ralloc(),
|
|
||||||
.eye = ralloc(),
|
|
||||||
.up_in = ralloc(),
|
|
||||||
.t0 = ralloc(),
|
|
||||||
.t1 = ralloc(),
|
|
||||||
.t2 = ralloc(),
|
|
||||||
.t3 = ralloc(),
|
|
||||||
.t4 = ralloc(),
|
|
||||||
});
|
|
||||||
regfile_reset(& rf);
|
|
||||||
|
|
||||||
bundle->normalize_fwd_uz = smem.normalize_v3s4;
|
|
||||||
bundle->cross_to_right = smem.gte_cross_v3s4;
|
|
||||||
bundle->normalize_right_ux = smem.normalize_v3s4;
|
|
||||||
bundle->cross_to_up = smem.gte_cross_v3s4;
|
|
||||||
bundle->normalize_up_uy = smem.normalize_v3s4;
|
|
||||||
|
|
||||||
bundle->pop_mv_trans = resolve_look_at__pop_mv_trans(& ab,
|
|
||||||
RegUse_(resolve_look_at__pop_mv_trans){
|
|
||||||
.look_at = ralloc(),
|
|
||||||
.eye = ralloc(),
|
|
||||||
.row = ralloc_v3(),
|
|
||||||
.t6 = ralloc(),
|
|
||||||
.t7 = ralloc(),
|
|
||||||
.t8 = ralloc(),
|
|
||||||
});
|
|
||||||
|
|
||||||
/* Sanity check: arena didn't overflow. */
|
|
||||||
assert(ab.used <= ResolveLookAtArena_Size);
|
|
||||||
#undef ralloc
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Emit the resolve_look_at bundle into the tape. Called once per frame from update(). */
|
|
||||||
I_ void resolve_look_at(TapeBuilder_R tb
|
|
||||||
, MT3_S2S4* look_at
|
|
||||||
, P3_S4* eye
|
|
||||||
, P3_S4* target
|
|
||||||
, V3_S4* up_in
|
|
||||||
){
|
|
||||||
/* Typed view of the scratchpad for field-address arithmetic. */
|
|
||||||
ResolveLookAtScratch* sp = C_scratch(ResolveLookAtScratch*);
|
|
||||||
AtomBundle_resolve_look_at_R bundle = C_(void*, smem.resolve_look_at_bundle);
|
|
||||||
|
|
||||||
tb_emit(tb, bundle->input_and_sub); {
|
|
||||||
tb_data(tb, u4_(target));
|
|
||||||
tb_data(tb, u4_(eye));
|
|
||||||
tb_data(tb, u4_(up_in));
|
|
||||||
}
|
|
||||||
tb_emit(tb, bundle->normalize_fwd_uz); {
|
|
||||||
tb_data(tb, u4_(O_(ResolveLookAtScratch, fwd) | (O_(ResolveLookAtScratch, uz) << 16)));
|
|
||||||
}
|
|
||||||
tb_emit(tb, bundle->cross_to_right); {
|
|
||||||
tb_data(tb, u4_(& sp->uz));
|
|
||||||
tb_data(tb, u4_(& sp->up_in));
|
|
||||||
tb_data(tb, u4_(& sp->right));
|
|
||||||
}
|
|
||||||
tb_emit(tb, bundle->normalize_right_ux); {
|
|
||||||
tb_data(tb, u4_(O_(ResolveLookAtScratch, right) | (O_(ResolveLookAtScratch, ux) << 16)));
|
|
||||||
}
|
|
||||||
tb_emit(tb, bundle->cross_to_up); {
|
|
||||||
tb_data(tb, u4_(& sp->uz));
|
|
||||||
tb_data(tb, u4_(& sp->ux));
|
|
||||||
tb_data(tb, u4_(& sp->up));
|
|
||||||
}
|
|
||||||
tb_emit(tb, bundle->normalize_up_uy); {
|
|
||||||
tb_data(tb, u4_(O_(ResolveLookAtScratch, up) | (O_(ResolveLookAtScratch, uy) << 16)));
|
|
||||||
}
|
|
||||||
tb_emit(tb, bundle->pop_mv_trans); {
|
|
||||||
tb_data(tb, u4_(look_at));
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
GCC_OPTIMIZATION_DISABLE
|
GCC_OPTIMIZATION_DISABLE
|
||||||
@@ -259,25 +160,21 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
|||||||
{
|
{
|
||||||
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
|
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
|
||||||
|
|
||||||
// Pad Input
|
if (1) // Pad Input
|
||||||
{
|
{
|
||||||
tb.used = 0; tb_scope_run(& tb) {
|
tb.used = 0; tb_scope_run(& tb) {
|
||||||
// Grab latest state from bios.
|
/* BIOS-owned polling: per-frame snapshot of both ports. */
|
||||||
tb_emit_(pad_bios_snapshot);
|
tb_emit_(pad_bios_snapshot);
|
||||||
tb_data_(raw, & smem.pad_raw[0]);
|
tb_data_(raw, & smem.pad_raw[0]);
|
||||||
tb_data_(state, & smem.pad[0]);
|
tb_data_(state, & smem.pad[0]);
|
||||||
// tb_emit_(pad_bios_snapshot);
|
tb_emit_(pad_bios_snapshot);
|
||||||
// tb_data_(raw, & smem.pad_raw[1]);
|
tb_data_(raw, & smem.pad_raw[1]);
|
||||||
// tb_data_(state, & smem.pad[1]);
|
tb_data_(state, & smem.pad[1]);
|
||||||
|
/* Per-frame rotation apply: consume pad[0].buttons + pad[0].left_x */
|
||||||
tb_emit_(pad_input_cam);
|
tb_emit_(pad_apply_input);
|
||||||
tb_data_(state, & smem.pad[0]);
|
tb_data_(state, & smem.pad[0]);
|
||||||
tb_data_(cam, & smem.cam);
|
tb_data_(cube_rot, & smem.cube.rot);
|
||||||
|
tb_data_(floor_rot, & smem.floor.rot);
|
||||||
// tb_emit_(pad_input_cube_rotation);
|
|
||||||
// tb_data_(state, & smem.pad[0]);
|
|
||||||
// tb_data_(cube_rot, & smem.cube.rot);
|
|
||||||
// tb_data_(floor_rot, & smem.floor.rot);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -289,37 +186,30 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
|||||||
gknown V3_S4_R acc = & smem.cube.accel;
|
gknown V3_S4_R acc = & smem.cube.accel;
|
||||||
add_v3s4(vel, acc[0]);
|
add_v3s4(vel, acc[0]);
|
||||||
add_v3s4_fp(pos, vel[0]);
|
add_v3s4_fp(pos, vel[0]);
|
||||||
|
// vel->x += acc->x;
|
||||||
|
// vel->y += acc->y;
|
||||||
|
// vel->z += acc->z;
|
||||||
|
// pos->x += vel->x;
|
||||||
|
// pos->y += vel->y;
|
||||||
|
// pos->z += vel->z;
|
||||||
|
|
||||||
if (pos->y + 150 > smem.floor.pos.y) vel->y *= -1;
|
if (pos->y + 150 > smem.floor.pos.y) vel->y *= -1;
|
||||||
|
|
||||||
// Prep
|
// Prep
|
||||||
S4 nclip = 0;
|
S4 nclip = 0;
|
||||||
S4 orderingtbl_z = 0;
|
S4 orderingtbl_z = 0;
|
||||||
A2_S2 p; //???
|
A2_S2 p; //???
|
||||||
S4 flag; //????
|
S4 flag; //????
|
||||||
|
|
||||||
B4 use_c11_path = false;
|
|
||||||
if (use_c11_path) {
|
|
||||||
camera_look_at_c11(& smem.cam, & smem.cube.pos, & v3s4(0, -fp_one, 0));
|
|
||||||
}
|
|
||||||
if (use_c11_path == false)
|
|
||||||
{
|
|
||||||
tb.used = 0; tb_scope_run(& tb) {
|
|
||||||
resolve_look_at(& tb, & smem.cam.look_at, & smem.cam.pos, & smem.cube.pos, & v3s4(0, -fp_one, 0));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Draw cube
|
// Draw cube
|
||||||
if (1)
|
if (1)
|
||||||
{
|
{
|
||||||
mt3s2s4_rotation (& smem.cube.rot, & smem.tform_world);
|
m3s2_rotation (& smem.cube.rot, & smem.tform_world);
|
||||||
mt3s2s4_translation(& smem.tform_world, & smem.cube.pos);
|
m3s2_translation(& smem.tform_world, & smem.cube.pos);
|
||||||
mt3s2s4_scale (& smem.tform_world, & smem.cube.scale);
|
m3s2_scale (& smem.tform_world, & smem.cube.scale);
|
||||||
|
gte_matrix_set_rotation (& smem.tform_world);
|
||||||
// Combine world and look_at matrix.
|
gte_matrix_set_translation(& smem.tform_world);
|
||||||
gte_comp_coord_m3s2(& smem.cam.look_at, & smem.tform_world, & smem.tform_view);
|
|
||||||
gte_matrix_set_rotation (& smem.tform_view);
|
|
||||||
gte_matrix_set_translation(& smem.tform_view);
|
|
||||||
|
|
||||||
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||||
U4 prim_cursor = prim_base + pa->used;
|
U4 prim_cursor = prim_base + pa->used;
|
||||||
@@ -340,22 +230,16 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
|||||||
tb_data(& tb, u4_(& pa->used));
|
tb_data(& tb, u4_(& pa->used));
|
||||||
tb_data(& tb, prim_base);
|
tb_data(& tb, prim_base);
|
||||||
}
|
}
|
||||||
tape_run(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
|
tape_run(tb_slice(tb));
|
||||||
|
|
||||||
// smem.cube.rot.y += 30;
|
// smem.cube.rot.y += 30;
|
||||||
}
|
}
|
||||||
// Draw floor
|
// Draw floor
|
||||||
if (1)
|
if (1)
|
||||||
{
|
{
|
||||||
mt3s2s4_rotation (& smem.floor.rot, & smem.tform_world);
|
m3s2_rotation (& smem.floor.rot, & smem.tform_world);
|
||||||
mt3s2s4_translation(& smem.tform_world, & smem.floor.pos);
|
m3s2_translation(& smem.tform_world, & smem.floor.pos);
|
||||||
mt3s2s4_scale (& smem.tform_world, & smem.floor.scale);
|
m3s2_scale (& smem.tform_world, & smem.floor.scale);
|
||||||
|
|
||||||
// Combine world and look_at matrix.
|
|
||||||
gte_comp_coord_m3s2(& smem.cam.look_at, & smem.tform_world, & smem.tform_view);
|
|
||||||
|
|
||||||
gte_matrix_set_rotation (& smem.tform_view);
|
|
||||||
gte_matrix_set_translation(& smem.tform_view);
|
|
||||||
|
|
||||||
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||||
U4 prim_cursor = prim_base + pa->used;
|
U4 prim_cursor = prim_base + pa->used;
|
||||||
@@ -365,11 +249,11 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
|||||||
|
|
||||||
// Prepare the tape. (Push protocol to tape)
|
// Prepare the tape. (Push protocol to tape)
|
||||||
tb.used = 0; tb_scope(& tb) {
|
tb.used = 0; tb_scope(& tb) {
|
||||||
// tb_emit(& tb, set_gte_mt3s2s4);
|
tb_emit(& tb, set_gte_world);
|
||||||
// tb_data(& tb, u4_(& smem.tform_view));
|
tb_data(& tb, u4_(& smem.tform_world));
|
||||||
|
|
||||||
tb_emit(& tb, rbind_floor_f3_face);
|
tb_emit(& tb, rbind_floor_f3_face);
|
||||||
// TODO(Ed): Just use a single context struct ref?
|
// TODO(Ed): Just use a single context struct ref
|
||||||
tb_data(& tb, prim_cursor);
|
tb_data(& tb, prim_cursor);
|
||||||
tb_data(& tb, u4_(smem.floor.faces));
|
tb_data(& tb, u4_(smem.floor.faces));
|
||||||
tb_data(& tb, u4_(smem.floor.verts));
|
tb_data(& tb, u4_(smem.floor.verts));
|
||||||
@@ -382,7 +266,7 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
|||||||
tb_data(& tb, u4_(& pa->used));
|
tb_data(& tb, u4_(& pa->used));
|
||||||
tb_data(& tb, prim_base);
|
tb_data(& tb, prim_base);
|
||||||
}
|
}
|
||||||
tape_run(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
|
tape_run(tb_slice(tb));// Fire off the tape.
|
||||||
|
|
||||||
// C-side state (pa->used) has already been updated by the tape!
|
// C-side state (pa->used) has already been updated by the tape!
|
||||||
// smem.floor.rot.y += 5;
|
// smem.floor.rot.y += 5;
|
||||||
@@ -406,14 +290,25 @@ void gp_display_frame(DoubleBuffer* screen_buf, S4* active_buf_id, U4* ordering_
|
|||||||
}
|
}
|
||||||
|
|
||||||
GCC_OPTIMIZATION_DISABLE
|
GCC_OPTIMIZATION_DISABLE
|
||||||
|
void hot_reload_entry(void)
|
||||||
|
{
|
||||||
|
smem.primitives.used = 0;
|
||||||
|
while (1) {
|
||||||
|
gknown S4* active_buf_id = & smem.active_buf_id;
|
||||||
|
gknown U4* ordering_buf = r_(smem.ordering_tbl)[active_buf_id[0]];
|
||||||
|
gknown PrimitiveArena* pa = & smem.primitives;
|
||||||
|
update(pa, ordering_buf);
|
||||||
|
render();
|
||||||
|
gp_display_frame(& smem.screen_buf, active_buf_id, ordering_buf, pa);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
int main(void)
|
int main(void)
|
||||||
{
|
{
|
||||||
smem = (SMemory){0};
|
smem = (SMemory){0};
|
||||||
// TODO(Ed): remove this field we don't need it in smem.
|
smem.scratchpad = C_(U4_V, 0x1F800000);
|
||||||
smem.scratchpad = C_(U4_V, Scratchpad_Loc);
|
|
||||||
// smem.primitives.used = 0;
|
// smem.primitives.used = 0;
|
||||||
// smem.active_buf_id = 0;
|
// smem.active_buf_id = 0;
|
||||||
smem.cam.pos = v3s4(500, -1000, -1500);
|
|
||||||
/*Persistent Entity Setup*/{
|
/*Persistent Entity Setup*/{
|
||||||
ent_cube128_init(& smem.cube.verts, & smem.cube.faces); {
|
ent_cube128_init(& smem.cube.verts, & smem.cube.faces); {
|
||||||
Ent_Cube* cube = & smem.cube;
|
Ent_Cube* cube = & smem.cube;
|
||||||
@@ -433,10 +328,6 @@ int main(void)
|
|||||||
reset_graph(0);
|
reset_graph(0);
|
||||||
/* Direct BIOS: poll both ports during VBlank. */
|
/* Direct BIOS: poll both ports during VBlank. */
|
||||||
pad_bios_init_start(& smem.pad_raw[0], & smem.pad_raw[1]);
|
pad_bios_init_start(& smem.pad_raw[0], & smem.pad_raw[1]);
|
||||||
|
|
||||||
compile_init_atoms();
|
|
||||||
compile_resolve_look_at();
|
|
||||||
|
|
||||||
/* Pinned registers for the GPU init atom. */
|
/* Pinned registers for the GPU init atom. */
|
||||||
register U4* io_base_addr rgcc(R_IO_BaseAddr) = u4_r(IO_BASE_ADDR);
|
register U4* io_base_addr rgcc(R_IO_BaseAddr) = u4_r(IO_BASE_ADDR);
|
||||||
register DoubleBuffer* screen_buf rgcc(R_ScreenBuf) = & smem.screen_buf;
|
register DoubleBuffer* screen_buf rgcc(R_ScreenBuf) = & smem.screen_buf;
|
||||||
@@ -445,14 +336,7 @@ int main(void)
|
|||||||
tb_emit(& tb, gp_screen_init);
|
tb_emit(& tb, gp_screen_init);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
while (1) {
|
hot_reload_entry();
|
||||||
gknown S4* active_buf_id = & smem.active_buf_id;
|
|
||||||
gknown U4* ordering_buf = r_(smem.ordering_tbl)[active_buf_id[0]];
|
|
||||||
gknown PrimitiveArena* pa = & smem.primitives;
|
|
||||||
update(pa, ordering_buf);
|
|
||||||
render();
|
|
||||||
gp_display_frame(& smem.screen_buf, active_buf_id, ordering_buf, pa);
|
|
||||||
};
|
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
GCC_OPTIMIZATION_ENABLE
|
GCC_OPTIMIZATION_ENABLE
|
||||||
|
|||||||
@@ -21,6 +21,12 @@ enum {
|
|||||||
ScreenRes_CenterY = (ScreenRes_Y >> 1),
|
ScreenRes_CenterY = (ScreenRes_Y >> 1),
|
||||||
};
|
};
|
||||||
|
|
||||||
|
enum {
|
||||||
|
fp_one = (1 << 12),
|
||||||
|
};
|
||||||
|
|
||||||
|
#define v3s4_fp_one() v3s4(fp_one, fp_one, fp_one)
|
||||||
|
|
||||||
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
||||||
typedef Array_(OrderingTable_Buffer, 2);
|
typedef Array_(OrderingTable_Buffer, 2);
|
||||||
|
|
||||||
@@ -61,7 +67,7 @@ I_ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
|||||||
typedef Struct_(Ent_Cube) {
|
typedef Struct_(Ent_Cube) {
|
||||||
V3_S4 accel;
|
V3_S4 accel;
|
||||||
V3_S4 vel;
|
V3_S4 vel;
|
||||||
V3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
|
V3_S4 pos;
|
||||||
V3_S4 scale;
|
V3_S4 scale;
|
||||||
V3_S2 rot;
|
V3_S2 rot;
|
||||||
A8_V3_S2 verts;
|
A8_V3_S2 verts;
|
||||||
@@ -88,15 +94,9 @@ I_ void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
|
|||||||
};
|
};
|
||||||
typedef Struct_(Ent_Floor) {
|
typedef Struct_(Ent_Floor) {
|
||||||
V3_S4 accel;
|
V3_S4 accel;
|
||||||
V3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
|
V3_S4 pos;
|
||||||
V3_S4 scale;
|
V3_S4 scale;
|
||||||
V3_S2 rot;
|
V3_S2 rot;
|
||||||
A4_V3_S2 verts;
|
A4_V3_S2 verts;
|
||||||
A2_V3_S2 faces;
|
A2_V3_S2 faces;
|
||||||
};
|
};
|
||||||
|
|
||||||
typedef Struct_(Camera) {
|
|
||||||
P3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
|
|
||||||
V3_S2 rot;
|
|
||||||
MT3_S2S4 look_at;
|
|
||||||
};
|
|
||||||
|
|||||||
@@ -24,8 +24,8 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c);
|
|||||||
|
|
||||||
#pragma region MACs (Mips Atom components)
|
#pragma region MACs (Mips Atom components)
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_put_disp_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
FI_ Slice_MipsCode ac_put_disp_env(U4 reg_transfer, U4 reg_base, U2 port)
|
||||||
MipsAtomComp_Proc_(ab, {
|
MipsAtomComp_Proc_(ac_put_disp_env, {
|
||||||
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
|
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
|
||||||
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
|
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
|
||||||
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
||||||
@@ -35,8 +35,8 @@ MipsAtomComp_Proc_(ab, {
|
|||||||
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
||||||
})
|
})
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_put_draw_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
FI_ Slice_MipsCode ac_put_draw_env(U4 reg_transfer, U4 reg_base, U2 port)
|
||||||
MipsAtomComp_Proc_(ab, {
|
MipsAtomComp_Proc_(ac_put_draw_env, {
|
||||||
/*
|
/*
|
||||||
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
|
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
|
||||||
* References:
|
* References:
|
||||||
@@ -116,7 +116,7 @@ internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
|
|||||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,1)),
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,1)),
|
||||||
|
|
||||||
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + OA_(DoubleBuffer,draw,0)), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
|
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + OA_(DoubleBuffer,draw,0)), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
|
||||||
mac_store_v2s2(R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + OA_(DoubleBuffer,draw,0)), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
|
mac_store_v2s2( R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + OA_(DoubleBuffer,draw,0)), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
|
||||||
|
|
||||||
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + OA_(DoubleBuffer,draw,1)),
|
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
|
||||||
@@ -286,7 +286,7 @@ MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
|||||||
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||||
, atom_writes(R_PrimCursor, R_FaceCursor)
|
, atom_writes(R_PrimCursor, R_FaceCursor)
|
||||||
) {
|
) {
|
||||||
mac_load_tri_indices(R_FaceCursor, R_T0, R_T1, R_T2),
|
mac_load_tri_indices( R_FaceCursor, R_T0, R_T1, R_T2),
|
||||||
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
||||||
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
|
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
|
||||||
gte_cmdw_nclip,
|
gte_cmdw_nclip,
|
||||||
|
|||||||
@@ -24,8 +24,8 @@
|
|||||||
* Emits 9 instructions (status/buttons/axes/attempt stores plus the
|
* Emits 9 instructions (status/buttons/axes/attempt stores plus the
|
||||||
* two-instruction zero-extended buttons load).
|
* two-instruction zero-extended buttons load).
|
||||||
*/
|
*/
|
||||||
FI_ Slice_MipsCode ac_pad_sio_write_pad_state(MipsAtomBuilder_R ab, U4 status_val, U4 state_ptr_reg, U4 scratch_reg)
|
FI_ Slice_MipsCode ac_pad_sio_write_pad_state(U4 status_val, U4 state_ptr_reg, U4 scratch_reg)
|
||||||
MipsAtomComp_Proc_(ac_pad_sio_write_pad_state, ab, {
|
MipsAtomComp_Proc_(ac_pad_sio_write_pad_state, {
|
||||||
add_ui(scratch_reg, R_0, status_val),
|
add_ui(scratch_reg, R_0, status_val),
|
||||||
store_word(scratch_reg, state_ptr_reg, O_(PadState,status)),
|
store_word(scratch_reg, state_ptr_reg, O_(PadState,status)),
|
||||||
/* FIX 2026-08-02: buttons = 0x0000FFFF = "no buttons pressed" in
|
/* FIX 2026-08-02: buttons = 0x0000FFFF = "no buttons pressed" in
|
||||||
|
|||||||
-10625
File diff suppressed because one or more lines are too long
+269
-50
@@ -1,3 +1,13 @@
|
|||||||
|
# --- Parameter Surface (Task 8) -----------------------------------------
|
||||||
|
# -Reload : After a successful build, invoke reload.ps1 as a child pwsh and propagate its exit code.
|
||||||
|
# -HelperZipOnly : Skip the build entirely; regenerate the helper zip and exit. Honors -HelperZipOutput for out-of-tree paths.
|
||||||
|
# -HelperZipOutput: When -HelperZipOnly is set, writes the archive to this path instead of the scripts/pcsx_debug_helper.zip.
|
||||||
|
param(
|
||||||
|
[switch]$Reload,
|
||||||
|
[switch]$HelperZipOnly,
|
||||||
|
[string]$HelperZipOutput = ''
|
||||||
|
)
|
||||||
|
|
||||||
$path_root = split-path -Path $PSScriptRoot -Parent
|
$path_root = split-path -Path $PSScriptRoot -Parent
|
||||||
$path_build = join-path $path_root 'build'
|
$path_build = join-path $path_root 'build'
|
||||||
$path_code = join-path $path_root 'code'
|
$path_code = join-path $path_root 'code'
|
||||||
@@ -8,6 +18,98 @@ if ((test-path $path_build) -eq $false) {
|
|||||||
new-item -itemtype directory -path $path_build
|
new-item -itemtype directory -path $path_build
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# --- HelperZipOnly short-circuit ----------------------------------------
|
||||||
|
# Must run before any compile/link work.
|
||||||
|
# Inlines the same logic as Make-HelperZip below to avoid an extra pwsh process spawn (~200 ms).
|
||||||
|
#The helper zip is small and the BCL call is in-process; cold ~14 ms, warm ~10 ms (assembly load + tiny zip write).
|
||||||
|
if ($HelperZipOnly) {
|
||||||
|
$zipDest = if ([string]::IsNullOrEmpty($HelperZipOutput)) {
|
||||||
|
join-path $path_scripts 'pcsx_debug_helper.zip'
|
||||||
|
}
|
||||||
|
else {
|
||||||
|
$HelperZipOutput
|
||||||
|
}
|
||||||
|
$HelperDir = join-path $path_scripts 'pcsx_debug_helper'
|
||||||
|
$elf32Src = join-path $path_scripts 'elf32.lua'
|
||||||
|
$elf32Dest = join-path $HelperDir 'elf32.lua'
|
||||||
|
if (-not (test-path -LiteralPath $HelperDir)) {
|
||||||
|
write-error "helper dir not found: $HelperDir"
|
||||||
|
exit 1
|
||||||
|
}
|
||||||
|
if (-not (test-path -LiteralPath $elf32Src)) {
|
||||||
|
write-error "elf32.lua not found at $elf32Src"
|
||||||
|
exit 1
|
||||||
|
}
|
||||||
|
write-host "[build] HelperZipOnly mode -> $zipDest"
|
||||||
|
|
||||||
|
# --- Timestamp gate (Fix 1) -------------------------------------------
|
||||||
|
# PCSX-Redux holds pcsx_debug_helper.zip open via -archive at startup.
|
||||||
|
# The zip is consumed once at startup; the reload endpoint reads it
|
||||||
|
# from package.loaded on subsequent calls. Writing it on every build
|
||||||
|
# is dead work that fights the file lock. Skip the rewrite when the
|
||||||
|
# three sources (autoexec.lua, reload.lua, elf32.lua) are all older
|
||||||
|
# than the existing zip.
|
||||||
|
$sources = @(
|
||||||
|
(join-path $HelperDir 'autoexec.lua'),
|
||||||
|
(join-path $HelperDir 'reload.lua'),
|
||||||
|
$elf32Src
|
||||||
|
)
|
||||||
|
$zipMtime = $null
|
||||||
|
if (test-path -LiteralPath $zipDest) {
|
||||||
|
$zipMtime = (Get-Item -LiteralPath $zipDest).LastWriteTime
|
||||||
|
}
|
||||||
|
$needsRewrite = $false
|
||||||
|
if ($null -eq $zipMtime) {
|
||||||
|
$needsRewrite = $true
|
||||||
|
}
|
||||||
|
else {
|
||||||
|
foreach ($s in $sources) {
|
||||||
|
if (-not (test-path -LiteralPath $s)) { continue }
|
||||||
|
if ((Get-Item -LiteralPath $s).LastWriteTime -gt $zipMtime) {
|
||||||
|
$needsRewrite = $true
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (-not $needsRewrite) {
|
||||||
|
$sz = (Get-Item -LiteralPath $zipDest).Length
|
||||||
|
Write-Host "[build] helper zip up to date: $zipDest ($sz bytes); skipping"
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
Copy-Item -LiteralPath $elf32Src -Destination $elf32Dest -Force
|
||||||
|
try {
|
||||||
|
# Force the inode release so CreateFromDirectory can write fresh.
|
||||||
|
# ZipFile.CreateFromDirectory throws if the destination exists.
|
||||||
|
# If PCSX-Redux holds the file open, Remove-Item raises — fall
|
||||||
|
# back to writing pcsx_debug_helper.zip.new alongside. The next
|
||||||
|
# PCSX-Redux restart will read the canonical path; the .new file
|
||||||
|
# is a hint for the optional launch-script patch in fix 3.
|
||||||
|
if (test-path -LiteralPath $zipDest) {
|
||||||
|
try {
|
||||||
|
# -ErrorAction Stop is required so the catch below fires.
|
||||||
|
# Remove-Item raises a non-terminating error by default
|
||||||
|
# (ErrorActionPreference=Continue), which bypasses catch.
|
||||||
|
Remove-Item -LiteralPath $zipDest -Force -ErrorAction Stop
|
||||||
|
}
|
||||||
|
catch {
|
||||||
|
$zipDest = [System.IO.Path]::ChangeExtension($zipDest, '.zip.new')
|
||||||
|
Write-Warning "[build] canonical helper zip is locked; writing to $zipDest instead"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Add-Type -AssemblyName System.IO.Compression.FileSystem
|
||||||
|
[System.IO.Compression.ZipFile]::CreateFromDirectory(
|
||||||
|
$HelperDir, $zipDest,
|
||||||
|
[System.IO.Compression.CompressionLevel]::Optimal, $false) | Out-Null
|
||||||
|
$sz = (Get-Item -LiteralPath $zipDest).Length
|
||||||
|
Write-Host "[build] wrote $sz bytes to $zipDest"
|
||||||
|
}
|
||||||
|
finally {
|
||||||
|
if (test-path -LiteralPath $elf32Dest) { Remove-Item -LiteralPath $elf32Dest -Force }
|
||||||
|
}
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
# --- Toolchain Definition ---
|
# --- Toolchain Definition ---
|
||||||
# Assumes 'mipsel-none-elf' toolchain is in your system's PATH.
|
# Assumes 'mipsel-none-elf' toolchain is in your system's PATH.
|
||||||
$Prefix = "mipsel-none-elf"
|
$Prefix = "mipsel-none-elf"
|
||||||
@@ -180,9 +282,12 @@ function link-modules { param([string[]]$link_modules, [string] $elf, [string[]
|
|||||||
$link_args += ($f_link_pass_through_prefix + $f_link_mapfile + $map)
|
$link_args += ($f_link_pass_through_prefix + $f_link_mapfile + $map)
|
||||||
|
|
||||||
$link_args += ($f_link_pass_through_prefix + $f_link_start_group)
|
$link_args += ($f_link_pass_through_prefix + $f_link_start_group)
|
||||||
# 16 removed entries (c2, card, cd, comb, ds, gs, gun, hmd, math, mcrd, mcx, press, sio, snd, spu, tap)
|
# raw_sio_pad_poll_20260802 — Task 5.1c surgical library-list trim.
|
||||||
# had LOAD lines in the map but ZERO .o files pulled in — they were unused.
|
# The 16 removed entries (c2, card, cd, comb, ds, gs, gun, hmd, math,
|
||||||
# 5 kept libraries (api, c, etc, gpu, gte) are required by the C-side calls in hello_joypad.c (reset_graph, draw_sync, vsync, etc.).
|
# mcrd, mcx, press, sio, snd, spu, tap) had LOAD lines in the map but
|
||||||
|
# ZERO .o files pulled in — they were unused. The 5 kept libraries
|
||||||
|
# (api, c, etc, gpu, gte) are required by the C-side calls in
|
||||||
|
# hello_joypad.c (reset_graph, draw_sync, vsync, etc.).
|
||||||
$libraries = @(
|
$libraries = @(
|
||||||
"api",
|
"api",
|
||||||
"c",
|
"c",
|
||||||
@@ -217,14 +322,16 @@ function make-binary { param([string]$elf, [string]$exe)
|
|||||||
}
|
}
|
||||||
|
|
||||||
function ps1-meta { param(
|
function ps1-meta { param(
|
||||||
[string]$unity_root,
|
[string] $unity_root,
|
||||||
[string[]]$sources,
|
[string[]]$sources,
|
||||||
[Parameter(Mandatory=$true)][string]$metadata,
|
[Parameter(Mandatory=$true)][string]$metadata,
|
||||||
[string]$out_root = (join-path $path_build 'gen'),
|
[string] $out_root = (join-path $path_build 'gen'),
|
||||||
[string[]]$passes = @('--pre-link'),
|
[string[]]$passes = @('--pre-link'),
|
||||||
[string[]]$extra_args = @()
|
[string[]]$extra_args = @()
|
||||||
)
|
)
|
||||||
# `--unity-root` and `--source` are mutually exclusive. Exactly one of `$unity_root` / `$sources` must be supplied; the other must be absent.
|
# `--unity-root` and `--source` are
|
||||||
|
# mutually exclusive. Exactly one of `$unity_root` / `$sources` must
|
||||||
|
# be supplied; the other must be absent.
|
||||||
if ($null -ne $unity_root -and $unity_root -ne '')
|
if ($null -ne $unity_root -and $unity_root -ne '')
|
||||||
{
|
{
|
||||||
if ($null -ne $sources -and $sources.Count -gt 0) {
|
if ($null -ne $sources -and $sources.Count -gt 0) {
|
||||||
@@ -237,6 +344,40 @@ function ps1-meta { param(
|
|||||||
exit 2
|
exit 2
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# --- Defensive attribute clear on tracked gen files ------------------------
|
||||||
|
# Git tracks code/<dir>/gen/*.h files and Windows keeps the Archive bit set
|
||||||
|
# on them. Combined with transient editor locks or co-running processes,
|
||||||
|
# this can make io.open(path, "wb") fail with Access Denied / Sharing
|
||||||
|
# Violation even though Get-ChildItem shows IsReadOnly = False. Clearing
|
||||||
|
# the Read-only + Archive bits locally is safe; git re-asserts them on
|
||||||
|
# the next operation but the metaprogram write always wins.
|
||||||
|
#
|
||||||
|
# Derived from the caller's parameters: $metadata lives in $path_duffle
|
||||||
|
# (so its parent is the duffle dir), and $unity_root / $sources[0] lives
|
||||||
|
# in $path_module (so its parent is the module dir).
|
||||||
|
$pathToDuffle = split-path -Path $metadata -Parent
|
||||||
|
$pathToModule = $null
|
||||||
|
if ($null -ne $unity_root -and $unity_root -ne '') {
|
||||||
|
$pathToModule = split-path -Path $unity_root -Parent
|
||||||
|
}
|
||||||
|
elseif ($null -ne $sources -and $sources.Count -gt 0) {
|
||||||
|
$pathToModule = split-path -Path $sources[0] -Parent
|
||||||
|
}
|
||||||
|
$genFiles = @(
|
||||||
|
join-path $pathToDuffle 'gen\macs.h'
|
||||||
|
join-path $pathToDuffle 'gen\offsets.h'
|
||||||
|
)
|
||||||
|
if ($null -ne $pathToModule) {
|
||||||
|
$genFiles += join-path $pathToModule 'gen\macs.h'
|
||||||
|
$genFiles += join-path $pathToModule 'gen\offsets.h'
|
||||||
|
}
|
||||||
|
foreach ($f in $genFiles) {
|
||||||
|
if (test-path -LiteralPath $f) {
|
||||||
|
attrib -R $f 2>&1 | Out-Null
|
||||||
|
attrib -A $f 2>&1 | Out-Null
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
$script = join-path $path_scripts 'ps1_meta.lua'
|
$script = join-path $path_scripts 'ps1_meta.lua'
|
||||||
$input_summary = if ($null -ne $unity_root -and $unity_root -ne '') {
|
$input_summary = if ($null -ne $unity_root -and $unity_root -ne '') {
|
||||||
"unity=$unity_root"
|
"unity=$unity_root"
|
||||||
@@ -265,14 +406,14 @@ function inject-dwarf { param(
|
|||||||
[string]$path_gen
|
[string]$path_gen
|
||||||
)
|
)
|
||||||
$base_name = [System.IO.Path]::GetFileNameWithoutExtension($elf)
|
$base_name = [System.IO.Path]::GetFileNameWithoutExtension($elf)
|
||||||
$path_dwarf_line_bin = join-path $path_gen "$base_name.dwarf_line.bin"
|
$path_dwarf_line_bin = join-path $path_gen "$base_name.dwarf_line.bin"
|
||||||
$path_dwarf_aranges_bin = join-path $path_gen "$base_name.dwarf_aranges.bin"
|
$path_dwarf_aranges_bin = join-path $path_gen "$base_name.dwarf_aranges.bin"
|
||||||
$path_dwarf_rnglists_bin = join-path $path_gen "$base_name.dwarf_rnglists.bin"
|
$path_dwarf_rnglists_bin = join-path $path_gen "$base_name.dwarf_rnglists.bin"
|
||||||
$path_dwarf_info_bin = join-path $path_gen "$base_name.dwarf_info.bin"
|
$path_dwarf_info_bin = join-path $path_gen "$base_name.dwarf_info.bin"
|
||||||
$path_dwarf_abbrev_bin = join-path $path_gen "$base_name.dwarf_abbrev.bin"
|
$path_dwarf_abbrev_bin = join-path $path_gen "$base_name.dwarf_abbrev.bin"
|
||||||
$path_dwarf_str_bin = join-path $path_gen "$base_name.dwarf_str.bin"
|
$path_dwarf_str_bin = join-path $path_gen "$base_name.dwarf_str.bin"
|
||||||
$path_dwarf_loc_bin = join-path $path_gen "$base_name.dwarf_loc.bin"
|
$path_dwarf_loc_bin = join-path $path_gen "$base_name.dwarf_loc.bin"
|
||||||
$path_dwarf_loclists_bin = join-path $path_gen "$base_name.dwarf_loclists.bin"
|
$path_dwarf_loclists_bin = join-path $path_gen "$base_name.dwarf_loclists.bin"
|
||||||
$path_inject_elf = join-path $path_build "$base_name.dwarf-injected.elf"
|
$path_inject_elf = join-path $path_build "$base_name.dwarf-injected.elf"
|
||||||
|
|
||||||
if (-not (Test-Path $path_dwarf_line_bin)) { return }
|
if (-not (Test-Path $path_dwarf_line_bin)) { return }
|
||||||
@@ -517,7 +658,7 @@ function build-hello_camera {
|
|||||||
$path_build_gen = join-path $path_build 'gen'
|
$path_build_gen = join-path $path_build 'gen'
|
||||||
|
|
||||||
$src_c = join-path $path_module 'hello_camera.c'
|
$src_c = join-path $path_module 'hello_camera.c'
|
||||||
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--pre-link')
|
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen
|
||||||
|
|
||||||
$assemble_args = @()
|
$assemble_args = @()
|
||||||
$assemble_args += $f_debug
|
$assemble_args += $f_debug
|
||||||
@@ -532,7 +673,6 @@ function build-hello_camera {
|
|||||||
|
|
||||||
$compile_args = @()
|
$compile_args = @()
|
||||||
$compile_args += $f_debug
|
$compile_args += $f_debug
|
||||||
$compile_args += ($f_define + 'BUILD_DEBUG')
|
|
||||||
$compile_args += $f_optimize_none
|
$compile_args += $f_optimize_none
|
||||||
# $compile_args += $f_optimize_intrinsics
|
# $compile_args += $f_optimize_intrinsics
|
||||||
# $compile_args += $f_optimize_size
|
# $compile_args += $f_optimize_size
|
||||||
@@ -553,50 +693,129 @@ function build-hello_camera {
|
|||||||
link-modules $link_modules $elf $link_args
|
link-modules $link_modules $elf $link_args
|
||||||
make-binary $elf $exe
|
make-binary $elf $exe
|
||||||
|
|
||||||
|
# Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start).
|
||||||
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
|
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
|
||||||
|
|
||||||
inject-dwarf $elf $path_build_gen
|
inject-dwarf $elf $path_build_gen
|
||||||
}
|
}
|
||||||
build-hello_camera
|
build-hello_camera
|
||||||
|
|
||||||
# NO idea if this works yet...
|
# ── Helper-zip + reload helpers (Task 8) ──
|
||||||
function Send-ToEmulator { param( [string]$exePath )
|
# Defined right after the final build-hello_camera function so they're in scope for the post-build calls below.
|
||||||
$uri = "http://localhost:8080/api/v1/load-exec"
|
# The Make-HelperZip function is also reused by the -HelperZipOnly short-circuit at the top of this script.
|
||||||
|
# Both call the in-process BCL CreateFromDirectory rather than spawning a child pwsh to avoid the ~200 ms process-spawn overhead.
|
||||||
|
function Make-HelperZip {
|
||||||
|
param([string]$OutputPath = '')
|
||||||
|
|
||||||
# Absolute path is safest for the emulator web server
|
$dest = if ([string]::IsNullOrEmpty($OutputPath)) {
|
||||||
$absolutePath = [System.IO.Path]::GetFullPath($exePath)
|
join-path $path_scripts 'pcsx_debug_helper.zip'
|
||||||
|
}
|
||||||
|
else {
|
||||||
|
$OutputPath
|
||||||
|
}
|
||||||
|
|
||||||
# Create JSON payload pointing to your compiled .ps-exe
|
$HelperDir = join-path $path_scripts 'pcsx_debug_helper'
|
||||||
$body = @{ filename = $absolutePath } | ConvertTo-Json
|
$elf32Src = join-path $path_scripts 'elf32.lua'
|
||||||
|
$elf32Dest = join-path $HelperDir 'elf32.lua'
|
||||||
|
if (-not (test-path -LiteralPath $HelperDir)) {
|
||||||
|
write-warning "[build] helper dir not found: $HelperDir; skipping helper zip"
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if (-not (test-path -LiteralPath $elf32Src)) {
|
||||||
|
write-warning "[build] elf32.lua not found at $elf32Src; skipping helper zip"
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
Write-Host "Pushing hot-reload to PCSX-Redux..." -ForegroundColor Magenta
|
# --- Timestamp gate (Fix 1) -------------------------------------------
|
||||||
try {
|
# PCSX-Redux holds pcsx_debug_helper.zip open via -archive at startup.
|
||||||
$response = Invoke-RestMethod -Uri $uri -Method Post -Body $body -ContentType "application/json"
|
# The zip is consumed once at startup; the reload endpoint reads it
|
||||||
Write-Host "Hot-reload successful!" -ForegroundColor Green
|
# from package.loaded on subsequent calls. Writing it on every build
|
||||||
} catch {
|
# is dead work that fights the file lock. Skip the rewrite when the
|
||||||
Write-Warning "Could not connect to PCSX-Redux web server. Ensure the emulator is running and Web Server is enabled."
|
# three sources (autoexec.lua, reload.lua, elf32.lua) are all older
|
||||||
}
|
# than the existing zip.
|
||||||
|
$sources = @(
|
||||||
|
(join-path $HelperDir 'autoexec.lua'),
|
||||||
|
(join-path $HelperDir 'reload.lua'),
|
||||||
|
$elf32Src
|
||||||
|
)
|
||||||
|
$zipMtime = $null
|
||||||
|
if (test-path -LiteralPath $dest) {
|
||||||
|
$zipMtime = (Get-Item -LiteralPath $dest).LastWriteTime
|
||||||
|
}
|
||||||
|
$needsRewrite = $false
|
||||||
|
if ($null -eq $zipMtime) {
|
||||||
|
$needsRewrite = $true
|
||||||
|
}
|
||||||
|
else {
|
||||||
|
foreach ($s in $sources) {
|
||||||
|
if (-not (test-path -LiteralPath $s)) { continue }
|
||||||
|
if ((Get-Item -LiteralPath $s).LastWriteTime -gt $zipMtime) {
|
||||||
|
$needsRewrite = $true
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (-not $needsRewrite) {
|
||||||
|
$sz = (Get-Item -LiteralPath $dest).Length
|
||||||
|
Write-Host "[build] helper zip up to date: $dest ($sz bytes); skipping"
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
write-host "[build] regenerating helper zip -> $dest"
|
||||||
|
Copy-Item -LiteralPath $elf32Src -Destination $elf32Dest -Force
|
||||||
|
try {
|
||||||
|
# Force the inode release so CreateFromDirectory can write fresh.
|
||||||
|
# ZipFile.CreateFromDirectory throws if the destination exists.
|
||||||
|
# If PCSX-Redux holds the file open, Remove-Item raises — fall
|
||||||
|
# back to writing pcsx_debug_helper.zip.new alongside. The next
|
||||||
|
# PCSX-Redux restart will read the canonical path; the .new file
|
||||||
|
# is a hint for the optional launch-script patch in fix 3.
|
||||||
|
if (test-path -LiteralPath $dest) {
|
||||||
|
try {
|
||||||
|
# -ErrorAction Stop is required so the catch below fires.
|
||||||
|
# Remove-Item raises a non-terminating error by default
|
||||||
|
# (ErrorActionPreference=Continue), which bypasses catch.
|
||||||
|
Remove-Item -LiteralPath $dest -Force -ErrorAction Stop
|
||||||
|
}
|
||||||
|
catch {
|
||||||
|
$dest = [System.IO.Path]::ChangeExtension($dest, '.zip.new')
|
||||||
|
Write-Warning "[build] canonical helper zip is locked; writing to $dest instead"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Add-Type -AssemblyName System.IO.Compression.FileSystem
|
||||||
|
[System.IO.Compression.ZipFile]::CreateFromDirectory(
|
||||||
|
$HelperDir, $dest,
|
||||||
|
[System.IO.Compression.CompressionLevel]::Optimal, $false) | Out-Null
|
||||||
|
$sz = (Get-Item -LiteralPath $dest).Length
|
||||||
|
Write-Host "[build] wrote $sz bytes to $dest"
|
||||||
|
}
|
||||||
|
finally {
|
||||||
|
if (test-path -LiteralPath $elf32Dest) { Remove-Item -LiteralPath $elf32Dest -Force }
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
# # Automatically hot-reloads it into the running emulator
|
# Invokes reload.ps1 as a child pwsh instead of POSTing to the nonexistent /api/v1/load-exec endpoint.
|
||||||
# Send-ToEmulator (join-path $path_build 'hello_gte.ps-exe')
|
# Exit code is propagated so the build fails loud if the reload fails.
|
||||||
|
function Send-ToEmulator {
|
||||||
|
param([string]$ElfPath = (join-path $path_build 'hello_camera.elf'))
|
||||||
|
|
||||||
# --- Hot Reload via PCSX-Redux Web Server ---
|
$reloadScript = join-path $path_scripts 'reload.ps1'
|
||||||
# $exe_path = join-path $path_build 'hello_gte.ps-exe'
|
if (-not (test-path -LiteralPath $reloadScript)) {
|
||||||
# $absolute_path = [System.IO.Path]::GetFullPath($exe_path)
|
write-error "[build] reload.ps1 not found at $reloadScript"
|
||||||
|
exit 1
|
||||||
|
}
|
||||||
|
|
||||||
# PCSX-Redux expects the file location in the URL query string?
|
write-host "[build] hot-reloading $ElfPath via reload.ps1" -ForegroundColor Magenta
|
||||||
# We URL-encode the path to ensure backslashes and spaces don't break the HTTP request?
|
& pwsh -NoProfile -File $reloadScript -Mode elf -Target hello_camera -ElfPath $ElfPath
|
||||||
# $encoded_path = [uri]::EscapeDataString($absolute_path)
|
if ($LASTEXITCODE -ne 0) {
|
||||||
# $uri = "http://localhost:8080/api/v1/load-exec?path=$encoded_path"
|
write-error "[build] reload.ps1 failed (exit $LASTEXITCODE)"
|
||||||
|
exit $LASTEXITCODE
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
# Write-Host "Pushing hot-reload to PCSX-Redux..." -ForegroundColor Magenta
|
# Post-build: Regenerate the helper zip (canonical output) and, if -Reload was passed, kick a hot-reload against the just-built ELF.
|
||||||
# try {
|
# Any future targets compiled by this script should add their own Make-HelperZip call after their build step; today's only target is hello_camera.
|
||||||
# # Send the request with the query string included
|
Make-HelperZip
|
||||||
# Invoke-RestMethod -Uri $uri -Method Post
|
if ($Reload) {
|
||||||
# Write-Host "Hot-reload successful!" -ForegroundColor Green
|
Send-ToEmulator
|
||||||
# } catch {
|
}
|
||||||
# Write-Host "Failed to hot-reload." -ForegroundColor Red
|
|
||||||
# # This will print the *actual* HTTP error instead of our generic warning
|
|
||||||
# Write-Host $_.Exception.Message -ForegroundColor Yellow
|
|
||||||
# }
|
|
||||||
|
|||||||
+2603
-36
File diff suppressed because it is too large
Load Diff
@@ -1,912 +0,0 @@
|
|||||||
--- duffle_emit.lua — project_emission + decl finders.
|
|
||||||
local scan = require("duffle_scan")
|
|
||||||
local isa = require("duffle_isa")
|
|
||||||
local M = {}
|
|
||||||
for k, v in pairs(scan) do M[k] = v end
|
|
||||||
for k, v in pairs(isa) do M[k] = v end
|
|
||||||
|
|
||||||
-- Section 8: Cross-source component-body index + word-event expansion
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
|
||||||
--
|
|
||||||
-- Shared, memoized helpers: a single emitted-word event stream that every downstream pass reads from,
|
|
||||||
-- built once from the pre-tokenized bodies.
|
|
||||||
|
|
||||||
--- @class ComponentBodyEntry
|
|
||||||
--- @field body_tokens table -- pre-tokenized {{tok=string, rel=integer}, ...}
|
|
||||||
--- @field body_off integer -- byte offset of body[1] in `source`
|
|
||||||
--- @field line_of fun(pos:integer):integer -- byte-offset → 1-based line number in `source`
|
|
||||||
--- @field source string -- absolute path of the source containing the declaration
|
|
||||||
--- @field declaration integer -- 1-based line number of the MipsAtomComp_(ac_X) declaration
|
|
||||||
--- @field kind string -- "comp_bare" | "comp_proc"
|
|
||||||
|
|
||||||
-- The cross-source component-body index is owned by the corpus (`corpus.component_body_index`, populated by `passes/components.lua`).
|
|
||||||
-- Consumers (`passes/static_analysis.lua`, `passes/emission_model.lua`) read it directly; per-pass memoization helpers stay out of scope.
|
|
||||||
|
|
||||||
-- ASCII byte constants used by split_call_args (kept local to keep Section 8 self-contained).
|
|
||||||
local E_BYTE_OPEN_PAREN = 0x28
|
|
||||||
local E_BYTE_OPEN_BRACE = 0x7B
|
|
||||||
local E_BYTE_OPEN_BRACK = 0x5B
|
|
||||||
local E_BYTE_DQUOTE = 0x22
|
|
||||||
local E_BYTE_SQUOTE = 0x27
|
|
||||||
local E_BYTE_COMMA = 0x2C
|
|
||||||
|
|
||||||
-- Map an open-delimiter byte to its matching close string for read_balanced.
|
|
||||||
local E_OPEN_CLOSE = {
|
|
||||||
[E_BYTE_OPEN_PAREN] = ")",
|
|
||||||
[E_BYTE_OPEN_BRACE] = "}",
|
|
||||||
[E_BYTE_OPEN_BRACK] = "]",
|
|
||||||
}
|
|
||||||
|
|
||||||
--- Split the INSIDE of a `f(...)` call on top-level commas.
|
|
||||||
--- Honors nested parens / braces / brackets and skips strings / comments.
|
|
||||||
--- Returns a list of trimmed argument strings in source order.
|
|
||||||
--- (Mirrors split_top_level_commas but for paren-body args; intentionally distinct so a caller's brace-body split isn't confused with an arg list.)
|
|
||||||
--- @param inner string
|
|
||||||
--- @return string[]
|
|
||||||
local function split_call_args(inner)
|
|
||||||
local args = {}
|
|
||||||
if not inner or inner == "" then return args end
|
|
||||||
local pos = 1
|
|
||||||
local len = #inner
|
|
||||||
local start = 1
|
|
||||||
while pos <= len do
|
|
||||||
local c = inner:byte(pos)
|
|
||||||
local close = E_OPEN_CLOSE[c]
|
|
||||||
if close then
|
|
||||||
local _, after = M.read_balanced(inner, string.char(c), close, pos)
|
|
||||||
pos = after
|
|
||||||
elseif c == E_BYTE_DQUOTE or c == E_BYTE_SQUOTE then
|
|
||||||
pos = M.skip_str_or_cmt(inner, pos)
|
|
||||||
elseif c == E_BYTE_COMMA then
|
|
||||||
args[#args + 1] = M.trim(inner:sub(start, pos - 1))
|
|
||||||
start = pos + 1
|
|
||||||
pos = pos + 1
|
|
||||||
else
|
|
||||||
pos = pos + 1
|
|
||||||
end
|
|
||||||
end
|
|
||||||
if start <= len then args[#args + 1] = M.trim(inner:sub(start, len)) end
|
|
||||||
return args
|
|
||||||
end
|
|
||||||
|
|
||||||
--- Extract the leading identifier + top-level args list from a token string.
|
|
||||||
--- Returns (ident, args). For tokens without a `(...)` call, args is `{}`.
|
|
||||||
--- @param tok string
|
|
||||||
--- @return string, string[]
|
|
||||||
local function token_ident_and_args(tok)
|
|
||||||
local ident, after = M.read_ident(tok, 1)
|
|
||||||
if not ident then return "?", {} end
|
|
||||||
local paren_pos = M.skip_ws_and_cmt(tok, after)
|
|
||||||
if tok:sub(paren_pos, paren_pos) ~= "(" then return ident, {} end
|
|
||||||
local inner = M.read_parens(tok, paren_pos)
|
|
||||||
if not inner then return ident, {} end
|
|
||||||
return ident, split_call_args(inner)
|
|
||||||
end
|
|
||||||
|
|
||||||
-- The macro-name prefix that marks a `mac_X(...)` component invocation.
|
|
||||||
local E_MAC_PREFIX = "mac_"
|
|
||||||
local E_MAC_PREFIX_LEN = 4
|
|
||||||
|
|
||||||
--- Expand a body entry into the flat sequence of emitted machine-word events.
|
|
||||||
---
|
|
||||||
--- Semantics (one event per emitted machine word):
|
|
||||||
--- * Direct one-word encoders `load_word`, `add_ui`, `nop`, `gte_lw`, ...: One event with `ident` = leading ident, `args` = parsed top-level args.
|
|
||||||
--- * `nop2` (2-word pseudo-instruction): Two events, both with `ident = "nop"` so the recognized "this slot is a no-op" semantic is visible to downstream analyses.
|
|
||||||
--- * Any other N-word token in `word_counts`: N events sharing the same `ident` + `args` so useful CPU words retire slots in the cycle budget.
|
|
||||||
--- * Known `mac_X(...)` calls: Recursively expand the indexed component body, including nested components. Every event from the expansion carries:
|
|
||||||
--- - `source` / `line` = the COMPONENT'S source path + the line of the token within the component body (i.e. "definition site").
|
|
||||||
--- - `call_source` / `call_line` = the ROOT atom's source path + call-site line, PRESERVED across recursion so nested events still point at the original root.
|
|
||||||
--- * Unknown `mac_X` (not in `component_index`): fall back to `word_counts[ident]` if present; otherwise emit one opaque event so the cycle budget accounts for the word.
|
|
||||||
--- * Marker Tokens (`atom_label(...)` / `atom_offset(...)`): Zero events (they are pure metaprogram hints).
|
|
||||||
---
|
|
||||||
--- Cycle protection: a per-expansion `visiting` set tracks components currently on the expansion stack;
|
|
||||||
--- a re-entry produces a deterministic `{kind = "cycle", ...}` error and aborts that branch (does NOT hang, does NOT recurse).
|
|
||||||
---
|
|
||||||
--- Pure: reads `body_entry` / `component_index` / `word_counts`. Memoization is the caller's responsibility.
|
|
||||||
--- Callers wanting `word_events` / `word_event_errors` precomputed for many atoms should memoize them per atom.
|
|
||||||
--- @param body_entry table -- `{body_tokens, body_off, line_of, source, declaration}` (declaration = root atom's atom.line)
|
|
||||||
--- @param component_index table -- the bare-name → ComponentBodyEntry map from M.get_component_body_index
|
|
||||||
--- @param word_counts table -- macro name → emitted-word count (from `ctx.shared.word_counts`)
|
|
||||||
--- @return WordEvent[], WordEventError[]
|
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
|
||||||
-- Section 11: project_emission (per-atom emission projection)
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
|
||||||
--
|
|
||||||
-- Per-atom emission projection is owned by `passes/emission_model.lua`.
|
|
||||||
-- The projection is built from the root atom body only; invocation ancestry recursively expands nested components.
|
|
||||||
-- The items stream is the single ordered source of truth; `word_events` and `markers` are dense views over it.
|
|
||||||
--
|
|
||||||
-- The helper below operates on a body string (not a body_entry) so the pass can call it without depending on the older SourceScan / body_off conventions.
|
|
||||||
-- component_index argument is reserved for recursive component expansion.
|
|
||||||
-- word_counts table is authored-metadata + current-component count table.
|
|
||||||
|
|
||||||
--- @class EmissionProjection
|
|
||||||
--- @field items table[] -- Ordered stream of word|label|offset|invoke_begin|invoke_end
|
|
||||||
--- @field word_events table[] -- Dense view of items where kind == "word"
|
|
||||||
--- @field markers table[] -- Dense view of items where kind == "label"|"offset"
|
|
||||||
--- @field invocations InvocationRecord[] -- dense view of items where kind == "invoke_begin"|"invoke_end"
|
|
||||||
--- @field errors table[] -- Token-resolution failures surfaced without fail-loud
|
|
||||||
--- @field warnings table[] -- Opaque warnings (e.g. unknown uncounted macro)
|
|
||||||
|
|
||||||
--- @class InvocationRecord
|
|
||||||
--- Lives at `atom.paths.invocations[*]`. Constructed once at the single invocation-construction site
|
|
||||||
--- (`emit_invoke_begin` inside `_project_emission_inner`); `invoke_begin` / `invoke_end` markers in the items stream share the same `id`.
|
|
||||||
--- @field id integer -- 1-based, monotonic per-atom invocation id (0 is reserved for "no open invocation")
|
|
||||||
--- @field parent_id integer -- 0 for the outermost (root) call; otherwise the id of the immediately enclosing invocation
|
|
||||||
--- @field kind string -- "comp_bare" | "comp_proc" (component form that triggered the expansion)
|
|
||||||
--- @field component_name string -- Bare component name without the `mac_` prefix
|
|
||||||
--- @field call_text string -- Immediate `mac_X(...)` token text (or root call text for the outermost entry)
|
|
||||||
--- @field root_call_text string -- IMMUTABLE outermost `mac_X(...)` token text for every word emitted in this call's expansion
|
|
||||||
--- @field call_path string -- Source path of the call site (root atom source for direct calls, component source for nested expansions)
|
|
||||||
--- @field call_line integer -- Source line of the call site
|
|
||||||
--- @field def_path string -- Source path of the component definition
|
|
||||||
--- @field def_line integer -- Source line of the component declaration
|
|
||||||
--- @field start_pos integer -- 0-based emitted-word position of the FIRST word inside this invocation (the value of `word_idx` AT `emit_invoke_begin` time, BEFORE the first word is emitted). Words emitted inside this invocation occupy `start_pos..start_pos+#body_lines-1` (inclusive, 0-based). Downstream DWARF/provenance consumers MUST read this; do NOT reconstruct it from `start_word` (which is the 1-based items index including `invoke_begin`/`invoke_end` markers).
|
|
||||||
--- @field end_pos integer -- 0-based position of the LAST word inside this invocation (set by `emit_invoke_end` to `word_idx - 1` AFTER all body words are emitted).
|
|
||||||
--- @field start_word integer -- 1-based items index of the `invoke_begin` item
|
|
||||||
--- @field end_word integer -- 1-based items index of the `invoke_end` item (set by `emit_invoke_end`)
|
|
||||||
--- @field word_count integer -- Number of `word` items emitted between `start_word` and `end_word` (inclusive)
|
|
||||||
--- @field debug_skip boolean -- `debug_skip` stamp; true iff `corpus.components[name].debug_skip` is true at construction. Always boolean (never `nil`).
|
|
||||||
--- @field errors table[] -- Per-invocation construction errors (cycle / count_mismatch); does not include pass-level errors
|
|
||||||
|
|
||||||
-- Internal recursive walker. The items stream holds every emitted event in order; `word_events`, `markers`,
|
|
||||||
-- `invocations`, `errors`, `warnings` are dense views / side outputs appended alongside.
|
|
||||||
--
|
|
||||||
-- Output rules:
|
|
||||||
-- * `word` items record: `invocation_ids` (innermost last) and `outermost_invocation_id` (0 if no invocation is open).
|
|
||||||
-- * `invoke_begin` / `invoke_end` items are zero-width at the current word index; the same `word_index` is recorded on both.
|
|
||||||
-- * `root_call_text` is the outermost `mac_X(...)` token text for every word emitted inside a component expansion;
|
|
||||||
-- it is `nil` for direct words emitted from the root atom body.
|
|
||||||
-- * `call_text` is the IMMEDIATE top-level token spelling for the word (for nested words this is the inner `mac_X(...)` token;
|
|
||||||
-- for direct words it is the trimmed encoder token).
|
|
||||||
-- * `def_path` / `def_line` are the definition site of the current body (component source for nested words; root atom source for direct words, filled in by the pass caller).
|
|
||||||
-- * Unknown uncounted macros emit one opaque word + one warning. Unknown metadata-backed macros (entry in `word_counts`) emit the declared word count, no warning.
|
|
||||||
-- * Cycle detection uses an active DFS stack (`visiting`); a cycle appends a construction error to BOTH the projection errors and the cycle invocation's own errors,
|
|
||||||
-- then breaks out without recursing (the cycle entry still receives an invocation ID + paired `invoke_begin` / `invoke_end` items, so the boundary invariant is preserved).
|
|
||||||
-- * Component declared-count mismatch (declared vs. measured) is a construction error (kind = "count_mismatch"); recorded on the invocation record and pass-level errors list.
|
|
||||||
-- * Final boundary check: if any invocation is still open at end of walk, surface a "unbalanced" construction error.
|
|
||||||
local function _project_emission_inner(root_body_entry, ctx_table)
|
|
||||||
local items = {}
|
|
||||||
local word_events = {}
|
|
||||||
local markers = {}
|
|
||||||
local invocations = {}
|
|
||||||
local errors = {}
|
|
||||||
local warnings = {}
|
|
||||||
|
|
||||||
local word_idx = 0
|
|
||||||
local invocation_stack = {} -- stack of currently-open invocation records
|
|
||||||
local next_inv_id = 0
|
|
||||||
|
|
||||||
local reg_use_schema = ctx_table.reg_use_schema
|
|
||||||
local reg_use_param = ctx_table.reg_use_param
|
|
||||||
local atom_name = ctx_table.atom_name
|
|
||||||
|
|
||||||
local slot_readonly = {}
|
|
||||||
if reg_use_schema then
|
|
||||||
for _, slot in ipairs(reg_use_schema.slots or {}) do
|
|
||||||
slot_readonly[slot.name] = slot.readonly == true
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
local function apply_sub(sub_map, operand)
|
|
||||||
if not (sub_map and type(operand) == "string") then return operand end
|
|
||||||
if sub_map[operand] then return sub_map[operand] end
|
|
||||||
local dot = operand:find(".", 1, true)
|
|
||||||
if dot then
|
|
||||||
local head = operand:sub(1, dot - 1)
|
|
||||||
local mapped = sub_map[head]
|
|
||||||
if type(mapped) == "string" then
|
|
||||||
return mapped .. operand:sub(dot)
|
|
||||||
end
|
|
||||||
end
|
|
||||||
return operand
|
|
||||||
end
|
|
||||||
|
|
||||||
local function resolve_gpr_key(operand)
|
|
||||||
if type(operand) ~= "string" then return nil end
|
|
||||||
if operand:sub(1, 2) == "R_" then return operand end
|
|
||||||
if not (reg_use_schema and reg_use_param) then return nil end
|
|
||||||
local prefix = reg_use_param .. "."
|
|
||||||
if operand:sub(1, #prefix) ~= prefix then return nil end
|
|
||||||
local member_path = operand:sub(#prefix + 1)
|
|
||||||
local slot = reg_use_schema.alias_to_slot[member_path]
|
|
||||||
if not slot then return nil, member_path end
|
|
||||||
return "reguse:" .. atom_name .. ":" .. slot, nil, slot
|
|
||||||
end
|
|
||||||
|
|
||||||
local function open_invocation_ids_snapshot()
|
|
||||||
local ids = {}
|
|
||||||
for _, inv in ipairs(invocation_stack) do
|
|
||||||
ids[#ids + 1] = inv.id
|
|
||||||
end
|
|
||||||
return ids
|
|
||||||
end
|
|
||||||
|
|
||||||
local function emit_word(encoder, args, line, word_call_text,
|
|
||||||
def_source_now, def_line_now,
|
|
||||||
immediate_call_text, root_call_text_w, sub_map)
|
|
||||||
local inv_ids = open_invocation_ids_snapshot()
|
|
||||||
local outermost = inv_ids[1] or 0
|
|
||||||
-- For words emitted at the root atom body, `immediate_call_text` is nil and the walker's `word_call_text` (the word's own token, e.g. "nop") becomes the effective call_text.
|
|
||||||
-- For words emitted inside a component expansion, `immediate_call_text` is the immediate outer `mac_X(...)` token text;
|
|
||||||
-- The call that triggered the body expansion we're currently walking.
|
|
||||||
local eff_call_text = immediate_call_text or word_call_text
|
|
||||||
local eff_root_call_text = root_call_text_w
|
|
||||||
local gpr_keys = nil
|
|
||||||
if reg_use_schema or sub_map then
|
|
||||||
gpr_keys = {}
|
|
||||||
for pos, arg in ipairs(args or {}) do
|
|
||||||
local effective = apply_sub(sub_map, arg)
|
|
||||||
local key, unresolved, slot = resolve_gpr_key(effective)
|
|
||||||
gpr_keys[pos] = key
|
|
||||||
if unresolved then
|
|
||||||
errors[#errors + 1] = {
|
|
||||||
kind = "reguse_unresolved",
|
|
||||||
line = line,
|
|
||||||
msg = string.format("RegUse operand %q does not resolve in schema %q",
|
|
||||||
effective, (reg_use_schema and reg_use_schema.name) or "?"),
|
|
||||||
}
|
|
||||||
end
|
|
||||||
if key and slot and slot_readonly[slot] then
|
|
||||||
local row = M.instr(encoder)
|
|
||||||
if row and row.writes then
|
|
||||||
for _, wpos in ipairs(row.writes) do
|
|
||||||
if wpos == pos then
|
|
||||||
errors[#errors + 1] = {
|
|
||||||
kind = "reguse_const_write",
|
|
||||||
line = line,
|
|
||||||
msg = string.format("RegUse slot %q is Reg const; %s writes it",
|
|
||||||
slot, encoder),
|
|
||||||
}
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
if not reg_use_schema then
|
|
||||||
gpr_keys = nil
|
|
||||||
end
|
|
||||||
items[#items + 1] = {
|
|
||||||
kind = "word",
|
|
||||||
encoder = encoder,
|
|
||||||
args = args,
|
|
||||||
i = word_idx,
|
|
||||||
word_count = 1,
|
|
||||||
line = line,
|
|
||||||
call_text = eff_call_text,
|
|
||||||
root_call_text = eff_root_call_text,
|
|
||||||
invocation_ids = inv_ids,
|
|
||||||
outermost_invocation_id = outermost,
|
|
||||||
gpr_keys = gpr_keys,
|
|
||||||
}
|
|
||||||
word_events[#word_events + 1] = {
|
|
||||||
i = word_idx,
|
|
||||||
encoder = encoder,
|
|
||||||
args = args,
|
|
||||||
def_path = def_source_now or "",
|
|
||||||
def_line = def_line_now or 0,
|
|
||||||
call_text = eff_call_text,
|
|
||||||
root_call_text = eff_root_call_text,
|
|
||||||
invocation_ids = inv_ids,
|
|
||||||
outermost_invocation_id = outermost,
|
|
||||||
word_count = 1,
|
|
||||||
gpr_keys = gpr_keys,
|
|
||||||
}
|
|
||||||
word_idx = word_idx + 1
|
|
||||||
end
|
|
||||||
|
|
||||||
local function emit_marker(kind, name, target, line,
|
|
||||||
immediate_call_text, root_call_text_w,
|
|
||||||
consuming_encoder, consuming_arg_pos)
|
|
||||||
local inv_ids = open_invocation_ids_snapshot()
|
|
||||||
local outermost = inv_ids[1] or 0
|
|
||||||
-- Markers carry the open invocation stack snapshot. `call_text` / `root_call_text` belong to words, not markers — markers are zero-width and skip per-word call-site attribution.
|
|
||||||
-- `consuming_encoder` + `consuming_arg_pos` carry the surrounding control-transfer instruction context
|
|
||||||
-- (e.g. `branch_le_zero` consuming its 3rd argument, or `jump` / `call_addr` consuming their only argument).
|
|
||||||
-- `passes/offsets.lua` reads these to dispatch per-consuming-instruction offset encoding.
|
|
||||||
-- nil for top-level markers (where the marker is the entire token — no surrounding consuming instruction).
|
|
||||||
local it = {
|
|
||||||
kind = kind,
|
|
||||||
name = name,
|
|
||||||
line = line,
|
|
||||||
word_index = word_idx,
|
|
||||||
invocation_ids = inv_ids,
|
|
||||||
outermost_invocation_id = outermost,
|
|
||||||
}
|
|
||||||
if target ~= nil then it.target = target end
|
|
||||||
if consuming_encoder then it.consuming_encoder = consuming_encoder end
|
|
||||||
if consuming_arg_pos then it.consuming_arg_pos = consuming_arg_pos end
|
|
||||||
items[#items + 1] = it
|
|
||||||
markers[#markers + 1] = {
|
|
||||||
kind = kind,
|
|
||||||
name = name,
|
|
||||||
line = line,
|
|
||||||
word_index = word_idx,
|
|
||||||
target = target,
|
|
||||||
consuming_encoder = consuming_encoder,
|
|
||||||
consuming_arg_pos = consuming_arg_pos,
|
|
||||||
}
|
|
||||||
end
|
|
||||||
|
|
||||||
-- Count top-level commas in `tok` between position `from_pos` (inclusive) and `to_pos` (exclusive).
|
|
||||||
-- Tracks paren depth so commas inside nested () don't count. Skips string literals + comments.
|
|
||||||
-- Used by `emit_embedded_markers` to compute `consuming_arg_pos` for each embedded marker.
|
|
||||||
local function count_top_level_commas(tok, from_pos, to_pos)
|
|
||||||
local depth = 0
|
|
||||||
local count = 0
|
|
||||||
local i = from_pos
|
|
||||||
while i < to_pos do
|
|
||||||
local c = tok:sub(i, i)
|
|
||||||
if c == "'" or c == '"' then
|
|
||||||
local next_pos = M.skip_str_or_cmt(tok, i)
|
|
||||||
i = (next_pos > i) and next_pos or (i + 1)
|
|
||||||
elseif c == "/" and tok:sub(i + 1, i + 1) == "/" then
|
|
||||||
-- line comment: skip to end of line
|
|
||||||
local nl = tok:find("\n", i, true)
|
|
||||||
i = (nl and nl + 1) or (#tok + 1)
|
|
||||||
elseif c == "/" and tok:sub(i + 1, i + 1) == "*" then
|
|
||||||
-- block comment: skip to matching */
|
|
||||||
local close = tok:find("*/", i + 2, true)
|
|
||||||
i = (close and close + 2) or (#tok + 1)
|
|
||||||
elseif c == "(" then
|
|
||||||
depth = depth + 1
|
|
||||||
i = i + 1
|
|
||||||
elseif c == ")" then
|
|
||||||
depth = depth - 1
|
|
||||||
i = i + 1
|
|
||||||
elseif c == "," and depth == 0 then
|
|
||||||
count = count + 1
|
|
||||||
i = i + 1
|
|
||||||
else
|
|
||||||
i = i + 1
|
|
||||||
end
|
|
||||||
end
|
|
||||||
return count
|
|
||||||
end
|
|
||||||
|
|
||||||
-- Find the position of the consuming instruction's open paren (the `(` that starts the consuming instruction's argument list).
|
|
||||||
-- Returns nil if the token's leading text isn't an ident followed by `(` (e.g. the ident is at the start of a non-instruction token).
|
|
||||||
local function find_consuming_paren(tok)
|
|
||||||
local i = 1
|
|
||||||
while i <= #tok do
|
|
||||||
local c = tok:sub(i, i)
|
|
||||||
if c == "(" then return i end
|
|
||||||
if not c:match("[%w_]") and c ~= " " then return nil end
|
|
||||||
i = i + 1
|
|
||||||
end
|
|
||||||
return nil
|
|
||||||
end
|
|
||||||
|
|
||||||
local function emit_embedded_markers(tok, tok_line, consuming_encoder)
|
|
||||||
-- When called with a non-nil `consuming_encoder`, the marker is nested inside that instruction's argument list.
|
|
||||||
-- We compute each marker's arg position by counting top-level commas between the consuming instruction's `(` and the marker's start.
|
|
||||||
local consuming_paren = nil
|
|
||||||
if consuming_encoder then consuming_paren = find_consuming_paren(tok) end
|
|
||||||
local pos = 1
|
|
||||||
while pos <= #tok do
|
|
||||||
-- Trim leading whitespace and comments before each scan.
|
|
||||||
pos = M.skip_ws_and_cmt(tok, pos)
|
|
||||||
if pos > #tok then break end
|
|
||||||
local ident, after = M.read_ident(tok, pos)
|
|
||||||
if not ident then
|
|
||||||
-- Not an ident: token is a string or comment; skip or one-step.
|
|
||||||
local next_pos = M.skip_str_or_cmt(tok, pos)
|
|
||||||
pos = (next_pos > pos) and next_pos or (pos + 1)
|
|
||||||
goto continue_loop
|
|
||||||
end
|
|
||||||
if M.DELAY_MARKERS[ident] then
|
|
||||||
local arg_pos = nil
|
|
||||||
if consuming_encoder and consuming_paren then
|
|
||||||
arg_pos = count_top_level_commas(tok, consuming_paren + 1, pos) + 1
|
|
||||||
end
|
|
||||||
emit_marker("delay", ident, nil, tok_line, nil, nil, consuming_encoder, arg_pos)
|
|
||||||
pos = after
|
|
||||||
goto continue_loop
|
|
||||||
end
|
|
||||||
if ident ~= "atom_label" and ident ~= "atom_offset" then
|
|
||||||
-- Ordinary ident; nothing to emit, step past the ident only.
|
|
||||||
pos = after
|
|
||||||
goto continue_loop
|
|
||||||
end
|
|
||||||
-- Marker ident: parse the (...) arguments.
|
|
||||||
local open = M.skip_ws_and_cmt(tok, after)
|
|
||||||
local inner, after_paren = M.read_parens(tok, open)
|
|
||||||
if not inner then
|
|
||||||
-- (...) Unreadable: fall back to non-marker behavior.
|
|
||||||
pos = after
|
|
||||||
goto continue_loop
|
|
||||||
end
|
|
||||||
-- Commit: label takes 1 arg, offset takes 2.
|
|
||||||
-- For embedded markers, propagate the consuming_encoder + the marker's arg position
|
|
||||||
-- (1-based) so `passes/offsets.lua` can dispatch per-consuming-instruction offset encoding.
|
|
||||||
-- Top-level markers (no consuming_encoder) get nil for both — the offsets pass treats
|
|
||||||
-- them as branch-equivalent for backward compatibility.
|
|
||||||
local arg_pos = nil
|
|
||||||
if consuming_encoder and consuming_paren then
|
|
||||||
arg_pos = count_top_level_commas(tok, consuming_paren + 1, pos) + 1
|
|
||||||
end
|
|
||||||
local args = split_call_args(inner)
|
|
||||||
if ident == "atom_label" then emit_marker("label", args[1] or "", nil, tok_line, nil, nil, consuming_encoder, arg_pos)
|
|
||||||
else emit_marker("offset", args[1] or "", args[2] or "", tok_line, nil, nil, consuming_encoder, arg_pos)
|
|
||||||
end
|
|
||||||
pos = after_paren
|
|
||||||
::continue_loop::
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
local function emit_invoke_begin(inv_kind, component_name, call_text,
|
|
||||||
root_call_text, call_path, call_line)
|
|
||||||
next_inv_id = next_inv_id + 1
|
|
||||||
-- Invocation-level debug_skip stamp: Emission pass owns `atom.paths.invocations[*].debug_skip`.
|
|
||||||
-- The stamp is resolved from the `corpus.components[name]` registry (passed in via `ctx_table.components` by `emission_model.run`),
|
|
||||||
-- Unmarked components stamp `false` (not `nil`) so consumers can dispatch on the boolean without nil checks.
|
|
||||||
--
|
|
||||||
-- The walker has already found the component body in `ctx_table.component_index[component_name]`, so the matching entry MUST exist in `ctx_table.components[component_name]`
|
|
||||||
-- (both registries are populated from the same source by the components pass).
|
|
||||||
-- A missing entry is a corpus-plumbing bug; we fail loudly here rather than silently stamp `false` and mask the regression.
|
|
||||||
local components = ctx_table.components
|
|
||||||
local component_def = components and components[component_name] or nil
|
|
||||||
if not component_def then
|
|
||||||
error("duffle.emit_invoke_begin: component " .. string.format("%q", component_name)
|
|
||||||
.. " is present in `component_index` (the walker matched a `mac_" .. component_name .. "()` call) but absent from `components` (the canonical corpus.components registry). "
|
|
||||||
.. "This is a corpus-plumbing bug — the components pass must populate corpus.components[name] for every component it puts in corpus.component_body_index[name]. "
|
|
||||||
.. "The emission pass refuses to silently stamp `debug_skip = false` for a missing registry entry."
|
|
||||||
, 0
|
|
||||||
)
|
|
||||||
end
|
|
||||||
local debug_skip_stamp = component_def.debug_skip == true
|
|
||||||
local inv = {
|
|
||||||
id = next_inv_id,
|
|
||||||
parent_id = 0, -- patched below by caller
|
|
||||||
kind = inv_kind,
|
|
||||||
component_name = component_name,
|
|
||||||
call_text = call_text,
|
|
||||||
root_call_text = root_call_text,
|
|
||||||
call_path = call_path,
|
|
||||||
call_line = call_line,
|
|
||||||
def_path = nil, -- patched below after component lookup
|
|
||||||
def_line = nil,
|
|
||||||
-- 0-based emitted-word position. `word_idx` is the monotonic 0-based counter of `word` items emitted so far in this walk —
|
|
||||||
-- BEFORE this invocation's first word is emitted, it equals the position of the first word inside the invocation.
|
|
||||||
-- `start_word` (1-based items index of `invoke_begin`) is kept for items-walking consumers (Annotation pass bounds checks),
|
|
||||||
-- but DWARF / provenance rows MUST read `start_pos` because those rows are 1-based over the dense `word_events` stream (which has no `invoke_begin` items).
|
|
||||||
start_pos = word_idx,
|
|
||||||
start_word = #items + 1, -- 1-based items index of invoke_begin
|
|
||||||
end_pos = nil, -- patched by emit_invoke_end
|
|
||||||
end_word = nil, -- patched by emit_invoke_end
|
|
||||||
word_count = 0,
|
|
||||||
debug_skip = debug_skip_stamp,
|
|
||||||
errors = {},
|
|
||||||
}
|
|
||||||
invocations[#invocations + 1] = inv
|
|
||||||
items [#items + 1] = {
|
|
||||||
kind = "invoke_begin",
|
|
||||||
invocation_id = inv.id,
|
|
||||||
word_index = word_idx,
|
|
||||||
invocation_ids = open_invocation_ids_snapshot(),
|
|
||||||
}
|
|
||||||
invocation_stack[#invocation_stack + 1] = inv
|
|
||||||
return inv
|
|
||||||
end
|
|
||||||
|
|
||||||
local function emit_invoke_end(inv)
|
|
||||||
-- 0-based emitted-word position of the LAST word inside this invocation.
|
|
||||||
-- After the last body word was emitted, `word_idx` was incremented past it, so `word_idx - 1` is the 0-based position of the last word.
|
|
||||||
inv.end_pos = word_idx - 1
|
|
||||||
inv.end_word = #items + 1 -- 1-based items index of invoke_end
|
|
||||||
items[#items + 1] = {
|
|
||||||
kind = "invoke_end",
|
|
||||||
invocation_id = inv.id,
|
|
||||||
word_index = word_idx,
|
|
||||||
invocation_ids = open_invocation_ids_snapshot(),
|
|
||||||
}
|
|
||||||
for i = #invocation_stack, 1, -1 do
|
|
||||||
if invocation_stack[i] == inv then
|
|
||||||
table.remove(invocation_stack, i)
|
|
||||||
break
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
-- Resolve the per-token word count. If unresolved, surface ONE warning
|
|
||||||
-- and fall back to 1 opaque word so the cycle budget still accounts for the slot.
|
|
||||||
local function resolve_count(ident, tok_line)
|
|
||||||
local wc = ctx_table.word_counts
|
|
||||||
if wc and wc[ident] then return wc[ident] end
|
|
||||||
local canon = M.gte_canon(ident)
|
|
||||||
if canon ~= ident and wc and wc[canon] then return wc[canon] end
|
|
||||||
warnings[#warnings + 1] = {
|
|
||||||
kind = "uncounted",
|
|
||||||
line = tok_line,
|
|
||||||
msg = string.format("project_emission: opaque word emitted for %q (no entry in word_counts or component_index)",
|
|
||||||
ident),
|
|
||||||
}
|
|
||||||
return 1
|
|
||||||
end
|
|
||||||
|
|
||||||
-- Recursive walker: walk one body entry, possibly descending into components.
|
|
||||||
-- walk_parent_inv_id: Invocation ID of the enclosing call (0 for the root call).
|
|
||||||
-- walk_root_call_text: Outermost `mac_X(...)` token text (preserved across recursion).
|
|
||||||
-- walk_immediate_call_text: IMMEDIATE outer `mac_X(...)` token text for words emitted in this body — nil for the root atom body.
|
|
||||||
-- Two trackers are propagated as separate parameters so words deep inside nested expansions correctly identify both their immediate call site and the outermost call site.
|
|
||||||
local function walk_body_entry(body_entry, walk_parent_inv_id,
|
|
||||||
walk_root_call_text, walk_immediate_call_text)
|
|
||||||
local tokens = body_entry.body_tokens or {}
|
|
||||||
local body_off = body_entry.body_off or 0
|
|
||||||
local line_of = body_entry.line_of or M.LineIndex("")
|
|
||||||
local def_source = body_entry.source or ""
|
|
||||||
local def_line = body_entry.declaration or 0
|
|
||||||
local sub_map = body_entry.sub_map
|
|
||||||
-- Per-token dispatch: each matched branch returns; only the fall-through
|
|
||||||
-- "opaque word" emit handles direct encoders + mac_X-without-component.
|
|
||||||
local function process_token(bt)
|
|
||||||
local tok = M.trim(bt.tok or "")
|
|
||||||
if tok == "" then return end
|
|
||||||
local ident, after = M.read_ident(tok, 1)
|
|
||||||
if not ident then ident = "?" end
|
|
||||||
local _, args = token_ident_and_args(tok)
|
|
||||||
local tok_line = line_of(body_off + bt.rel) or 0
|
|
||||||
if M.DELAY_MARKERS[ident] then
|
|
||||||
emit_marker("delay", ident, nil, tok_line)
|
|
||||||
local rest = M.trim(tok:sub(after or (#tok + 1)))
|
|
||||||
if rest ~= "" then
|
|
||||||
process_token({ tok = rest, rel = bt.rel })
|
|
||||||
end
|
|
||||||
return
|
|
||||||
end
|
|
||||||
-- embedded markers live only in non-marker tokens.
|
|
||||||
-- Pass `ident` as the consuming instruction so `emit_embedded_markers` can compute each marker's arg position + record the consuming_encoder for the offsets pass.
|
|
||||||
-- Canonicalize `jump_rel` to `branch_equal` (its preprocessor-expanded form) so the `consuming_encoder` metadata in marker records is canonical.
|
|
||||||
-- `jump_rel`: unconditional jump alias from `code/duffle/mips.h`.
|
|
||||||
local consuming_encoder_for_markers = (ident == "jump_rel") and "branch_equal" or ident
|
|
||||||
if ident ~= "atom_label" and ident ~= "atom_offset" then
|
|
||||||
emit_embedded_markers(tok, tok_line, consuming_encoder_for_markers)
|
|
||||||
end
|
|
||||||
-- atom_label / atom_offset: terminal markers, no further descent.
|
|
||||||
-- Top-level markers (the marker IS the entire token) have no consuming instruction;
|
|
||||||
-- nil for both `consuming_encoder` and `consuming_arg_pos`.
|
|
||||||
-- The offsets pass treats these as branch-equivalent for backward compatibility.
|
|
||||||
-- TODO(Ed): Review this don't want legacy cruft here..
|
|
||||||
if ident == "atom_label" then emit_marker("label", args[1] or "", nil, tok_line); return
|
|
||||||
elseif ident == "atom_offset" then emit_marker("offset", args[1] or "", args[2] or "", tok_line); return
|
|
||||||
end
|
|
||||||
if ident:sub(1, 4) == "mac_" then
|
|
||||||
local bare = ident:sub(5)
|
|
||||||
local comp = ctx_table.component_index[bare]
|
|
||||||
if comp then
|
|
||||||
local invocation_root_call_text = walk_root_call_text or tok
|
|
||||||
if ctx_table.visiting[bare] then
|
|
||||||
-- Cycle: still allocate inv_id, emit zero-width begin/end, record the cycle error; do NOT recurse.
|
|
||||||
local inv = emit_invoke_begin(comp.kind or "comp_bare", bare, tok, invocation_root_call_text, def_source, tok_line)
|
|
||||||
inv.parent_id = walk_parent_inv_id
|
|
||||||
inv.call_text = tok
|
|
||||||
local err = {
|
|
||||||
kind = "cycle",
|
|
||||||
msg = string.format("project_emission: component cycle detected: %q", bare),
|
|
||||||
source = def_source,
|
|
||||||
line = tok_line,
|
|
||||||
}
|
|
||||||
inv.errors[#inv.errors + 1] = err
|
|
||||||
errors [#errors + 1] = err
|
|
||||||
emit_invoke_end(inv)
|
|
||||||
return
|
|
||||||
end
|
|
||||||
-- First visit: descend + count + count_mismatch-check below.
|
|
||||||
ctx_table.visiting[bare] = true
|
|
||||||
local inv = emit_invoke_begin(comp.kind or "comp_bare", bare, tok, invocation_root_call_text, def_source, tok_line)
|
|
||||||
inv.parent_id = walk_parent_inv_id
|
|
||||||
inv.call_text = tok
|
|
||||||
inv.def_path = comp.source
|
|
||||||
inv.def_line = comp.declaration
|
|
||||||
-- Propagate trackers into the recursive walk:
|
|
||||||
-- immediate_call_text = this call's tok (the IMMEDIATE outer call for words emitted in this body)
|
|
||||||
-- root_call_text = the OUTERMOST call (immutable across the recursion)
|
|
||||||
local formal_names = ctx_table.component_index[bare]
|
|
||||||
and ctx_table.component_index[bare].arg_names
|
|
||||||
local child_map = nil
|
|
||||||
if formal_names then
|
|
||||||
child_map = {}
|
|
||||||
for i, fname in ipairs(formal_names) do
|
|
||||||
child_map[fname] = apply_sub(sub_map, args[i])
|
|
||||||
end
|
|
||||||
end
|
|
||||||
walk_body_entry({
|
|
||||||
body_tokens = comp.body_tokens or {},
|
|
||||||
body_off = comp.body_off or 0,
|
|
||||||
line_of = comp.line_of,
|
|
||||||
source = comp.source,
|
|
||||||
declaration = comp.declaration,
|
|
||||||
sub_map = child_map,
|
|
||||||
},
|
|
||||||
inv.id,
|
|
||||||
invocation_root_call_text,
|
|
||||||
tok)
|
|
||||||
ctx_table.visiting[bare] = nil
|
|
||||||
emit_invoke_end(inv)
|
|
||||||
-- Count `word` items inside [start_word, end_word].
|
|
||||||
local wc_inside = 0
|
|
||||||
for i = inv.start_word, inv.end_word do
|
|
||||||
local it = items[i]
|
|
||||||
if it and it.kind == "word" then
|
|
||||||
wc_inside = wc_inside + 1
|
|
||||||
end
|
|
||||||
end
|
|
||||||
inv.word_count = wc_inside
|
|
||||||
-- count_mismatch is a construction error: word_counts["mac_X"] is the declared count populated by the components pass;
|
|
||||||
-- We compare against the measured word count.
|
|
||||||
local declared = ctx_table.word_counts["mac_" .. bare]
|
|
||||||
if declared and wc_inside ~= declared then
|
|
||||||
local err = {
|
|
||||||
kind = "count_mismatch",
|
|
||||||
msg = string.format("project_emission: mac_%s declared=%d measured=%d", bare, declared, wc_inside),
|
|
||||||
source = def_source,
|
|
||||||
line = tok_line,
|
|
||||||
}
|
|
||||||
inv.errors[#inv.errors + 1] = err
|
|
||||||
errors [#errors + 1] = err
|
|
||||||
end
|
|
||||||
return
|
|
||||||
end
|
|
||||||
-- mac_X NOT in component_index: fall through to opaque emit.
|
|
||||||
end
|
|
||||||
-- Direct encoder, or mac_X-without-component: resolve count + emit n words.
|
|
||||||
-- Resolve_count may emit a warning if the count is unresolved.
|
|
||||||
local n = resolve_count(ident, tok_line)
|
|
||||||
local out_ident = (ident == "nop2") and "nop" or ident
|
|
||||||
for _ = 1, n do
|
|
||||||
emit_word(out_ident, args, tok_line, tok, def_source, def_line, walk_immediate_call_text, walk_root_call_text, sub_map)
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
for _, bt in ipairs(tokens) do
|
|
||||||
process_token(bt)
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
-- Initialize the per-walk mutable context.
|
|
||||||
-- `visiting` is the active DFS component stack; `root_call_path` / `root_call_line` are preserved across recursion so nested words always point at the
|
|
||||||
-- ORIGINAL root atom call site.
|
|
||||||
ctx_table.visiting = ctx_table.visiting or {}
|
|
||||||
ctx_table.root_call_path = ctx_table.root_call_path or ""
|
|
||||||
ctx_table.root_call_line = ctx_table.root_call_line or 0
|
|
||||||
|
|
||||||
-- Walk first; the pass caller stamps the root call site for direct words after the projection returns.
|
|
||||||
-- For nested words the def_path / def_line already point at the component source and MUST be preserved (the stamping helper checks for that).
|
|
||||||
walk_body_entry(root_body_entry, 0, nil, nil)
|
|
||||||
|
|
||||||
-- Boundary check: every invoke_begin must have a matching invoke_end.
|
|
||||||
-- If anything is still open, surface a hard error.
|
|
||||||
if #invocation_stack > 0 then
|
|
||||||
errors[#errors + 1] = {
|
|
||||||
kind = "unbalanced",
|
|
||||||
msg = string.format("project_emission: invocation boundaries not balanced (%d unclosed invocation(s) at end of walk)", #invocation_stack),
|
|
||||||
}
|
|
||||||
end
|
|
||||||
|
|
||||||
return {
|
|
||||||
items = items,
|
|
||||||
word_events = word_events,
|
|
||||||
markers = markers,
|
|
||||||
invocations = invocations,
|
|
||||||
errors = errors,
|
|
||||||
warnings = warnings,
|
|
||||||
}
|
|
||||||
end
|
|
||||||
|
|
||||||
--- Project a body string into the per-atom emission projection.
|
|
||||||
---
|
|
||||||
--- Semantics:
|
|
||||||
--- * Direct one-word tokens (`nop`, `add_ui`, ...): one `word` item, encoder = ident, word_count = 1.
|
|
||||||
--- * Metadata-backed N-word tokens (`nop2`, `mask_upper`, ...): N `word` items, all sharing the same encoder + word_count = 1.
|
|
||||||
--- `nop2` is normalized to encoder `nop` (per the spec).
|
|
||||||
--- * `atom_label(F)` markers: one `label` item with `name = "F"`, `word_index = current word_idx`; zero-width (does NOT advance word_idx).
|
|
||||||
--- * `atom_offset(B, T)` markers: one `offset` item with `name = "B"`, `target = "T"`, `word_index = current word_idx`; zero-width.
|
|
||||||
--- * Delay markers (`GteDelay_` / `LdSlot_` / `BdSlot_` / `DmaSlot_`): one `delay` item; zero-width. The following encoder is the next token.
|
|
||||||
--- * `mac_X(...)` calls: emit `invoke_begin` (zero-width), recurse into the component body, emit `invoke_end` (zero-width).
|
|
||||||
--- The component body's words land between the begin/end pair; one invocation record is allocated per call (monotonic ID per atom).
|
|
||||||
--- * Unknown uncounted macros emit 1 opaque word + one warning per occurrence.
|
|
||||||
--- * Tokens whose count cannot be resolved (e.g. `mac_unknown` not in word_counts and not in component_index) surface one
|
|
||||||
--- warning; cycle + count-mismatch + boundary violations are construction errors on `pass.errors`.
|
|
||||||
---
|
|
||||||
--- Every emitted `word` carries: `i` (0-based word index), `encoder`, `args` (top-level args), `def_path`, `def_line`,
|
|
||||||
--- `call_text` (the immediate token spelling), `root_call_text` (outermost `mac_X(...)` text), `word_count` (always 1),
|
|
||||||
--- `invocation_ids` (innermost last), `outermost_invocation_id`.
|
|
||||||
--- Markers carry: `kind`, `name`, `line`, `word_index`, `target` (only for offset kind), plus `invocation_ids` / `outermost_invocation_id`
|
|
||||||
--- for the open invocation stack at that word.
|
|
||||||
---
|
|
||||||
--- @param body_text string -- the raw atom body string
|
|
||||||
--- @param component_index table -- bare-name → component record (corpus.component_body_index)
|
|
||||||
--- @param word_counts table -- macro name → emitted word count
|
|
||||||
--- @param components table -- bare-name → component definition (corpus.components); REQUIRED — consumed at the invocation-construction site to stamp
|
|
||||||
--- `invocation.debug_skip`. A missing or non-table `components` raises a fail-loud error rather than silently falling back.
|
|
||||||
--- @return EmissionProjection
|
|
||||||
function M.project_emission(body_text, component_index, word_counts, components, reg_use_ctx)
|
|
||||||
-- The recursive walk delegates to `_project_emission_inner` so component bodies (which arrive as
|
|
||||||
-- `{body_tokens, body_off, line_of, source, declaration}` records from `corpus.component_body_index`)
|
|
||||||
-- re-enter the same walker with the same shared output state.
|
|
||||||
--
|
|
||||||
-- The walker is body-relative: it builds `line_of` from `body_text` and stamps body-relative line numbers (1..N)
|
|
||||||
-- into `item.line` and `invocation.call_line`. `passes/emission_model.lua::stamp_root_provenance` performs the single
|
|
||||||
-- conversion from body-relative to physical source line at the close site, using the source's `line_of` closure that
|
|
||||||
-- the pass forwarded. One owner of the line state.
|
|
||||||
if type(components) ~= "table" then
|
|
||||||
error("duffle.project_emission: `components` is required "
|
|
||||||
.. "(bare-name -> component definition, e.g. corpus.components); "
|
|
||||||
.. "got " .. type(components) .. ". "
|
|
||||||
.. "The emission pass MUST forward the corpus registry "
|
|
||||||
.. "so the invocation-construction site can stamp `debug_skip` "
|
|
||||||
.. "without a second pass, source parse, or parallel lookup.",
|
|
||||||
0)
|
|
||||||
end
|
|
||||||
|
|
||||||
if type(body_text) ~= "string" or body_text == "" then
|
|
||||||
-- Empty body: still return a valid (empty) projection.
|
|
||||||
return {
|
|
||||||
items = {},
|
|
||||||
word_events = {},
|
|
||||||
markers = {},
|
|
||||||
invocations = {},
|
|
||||||
errors = {},
|
|
||||||
warnings = {},
|
|
||||||
}
|
|
||||||
end
|
|
||||||
|
|
||||||
local tokens = M.tokenize_body(body_text)
|
|
||||||
return _project_emission_inner({
|
|
||||||
body_tokens = tokens,
|
|
||||||
body_off = 0,
|
|
||||||
line_of = M.LineIndex(body_text),
|
|
||||||
source = "",
|
|
||||||
declaration = 0,
|
|
||||||
},
|
|
||||||
{
|
|
||||||
component_index = component_index or {},
|
|
||||||
word_counts = word_counts or {},
|
|
||||||
components = components,
|
|
||||||
reg_use_schema = reg_use_ctx and reg_use_ctx.reg_use_schema,
|
|
||||||
reg_use_param = reg_use_ctx and reg_use_ctx.reg_use_param,
|
|
||||||
atom_name = reg_use_ctx and reg_use_ctx.atom_name,
|
|
||||||
schema_name = reg_use_ctx and reg_use_ctx.schema_name,
|
|
||||||
})
|
|
||||||
end
|
|
||||||
|
|
||||||
-------------------------------------------------------------------------------
|
|
||||||
-- find_function_decl_for — backward walk for MipsAtomComp_Proc_ name extraction.
|
|
||||||
--
|
|
||||||
-- After the `sym` arg was dropped from MipsAtomComp_Proc_, the component name is derived from the preceding
|
|
||||||
-- `FI_ Slice_MipsCode ac_X(args)` function declaration. This function walks backward from `before_pos` to find it.
|
|
||||||
--
|
|
||||||
-- Returns (raw_name, args_inner) or (nil, nil).
|
|
||||||
-- raw_name — e.g. "ac_load_word_imm"
|
|
||||||
-- args_inner — e.g. "AtomBuilder_R ab, Reg dst, U4 imm"
|
|
||||||
--
|
|
||||||
-- The walk finds the LAST "Slice_MipsCode" before before_pos, then skips whitespace + qualifiers
|
|
||||||
-- (FI_, atom_dbg_skip, comments) until it finds an ident followed by "(".
|
|
||||||
-- That ident is the function name; the parens contents are the args.
|
|
||||||
-------------------------------------------------------------------------------
|
|
||||||
function M.find_function_decl_for(source, before_pos, slice_mips_code_len)
|
|
||||||
local search_pos = 1
|
|
||||||
local last_match = nil
|
|
||||||
while true do
|
|
||||||
local found = source:find("Slice_MipsCode", search_pos, true)
|
|
||||||
if not found or found >= before_pos then break end
|
|
||||||
last_match = found
|
|
||||||
search_pos = found + slice_mips_code_len
|
|
||||||
end
|
|
||||||
if not last_match then return nil, nil end
|
|
||||||
|
|
||||||
local pos = last_match + slice_mips_code_len
|
|
||||||
while pos < before_pos do
|
|
||||||
-- skip whitespace
|
|
||||||
while pos <= #source do
|
|
||||||
local c = source:sub(pos, pos)
|
|
||||||
if c == " " or c == "\t" or c == "\n" or c == "\r" then
|
|
||||||
pos = pos + 1
|
|
||||||
else
|
|
||||||
break
|
|
||||||
end
|
|
||||||
end
|
|
||||||
if pos > #source then break end
|
|
||||||
-- skip line comments
|
|
||||||
if source:sub(pos, pos + 1) == "//" then
|
|
||||||
while pos <= #source and source:sub(pos, pos) ~= "\n" do pos = pos + 1 end
|
|
||||||
pos = pos + 1
|
|
||||||
goto continue
|
|
||||||
end
|
|
||||||
-- skip block comments
|
|
||||||
if source:sub(pos, pos + 1) == "/*" then
|
|
||||||
local close = source:find("*/", pos + 2, true)
|
|
||||||
if not close then break end
|
|
||||||
pos = close + 2
|
|
||||||
goto continue
|
|
||||||
end
|
|
||||||
-- try to read an ident
|
|
||||||
local ident, ident_end = M.read_ident(source, pos)
|
|
||||||
if not ident then break end
|
|
||||||
-- check if the next non-ws char after ident is "("
|
|
||||||
local next_pos = M.skip_ws_and_cmt(source, ident_end)
|
|
||||||
if source:sub(next_pos, next_pos) == "(" then
|
|
||||||
local inner = M.read_parens(source, next_pos)
|
|
||||||
if inner then
|
|
||||||
return ident, inner
|
|
||||||
end
|
|
||||||
end
|
|
||||||
-- ident not followed by "(" — it's a qualifier (FI_, atom_dbg_skip, etc); skip it
|
|
||||||
pos = ident_end
|
|
||||||
::continue::
|
|
||||||
end
|
|
||||||
return nil, nil
|
|
||||||
end
|
|
||||||
|
|
||||||
-------------------------------------------------------------------------------
|
|
||||||
-- find_atom_proc_decl_for — backward walk for MipsAtom_Proc_ name extraction.
|
|
||||||
--
|
|
||||||
-- The atom name is the preceding `MipsAtom* ident(args)` function ident.
|
|
||||||
-- This function walks backward from `before_pos` to find it.
|
|
||||||
--
|
|
||||||
-- Returns (raw_name, args_inner, func_ident, after_paren) or (nil, nil).
|
|
||||||
-- raw_name — the function ident as written
|
|
||||||
-- args_inner — e.g. "AtomArena_R aa, U4 r_scratch, ..."
|
|
||||||
-- after_paren — source position after the function `)`
|
|
||||||
--
|
|
||||||
-- The walk finds the LAST "MipsAtom*" before before_pos, then skips whitespace + qualifiers (internal, I_, FI_, comments)
|
|
||||||
-- until it finds an ident followed by "(".
|
|
||||||
-------------------------------------------------------------------------------
|
|
||||||
function M.find_atom_proc_decl_for(source, before_pos, mips_atom_ptr_len)
|
|
||||||
local search_pos = 1
|
|
||||||
local last_match = nil
|
|
||||||
while true do
|
|
||||||
-- plain=true: "*" is literal, no escaping needed
|
|
||||||
local found = source:find("MipsAtom*", search_pos, true)
|
|
||||||
if not found or found >= before_pos then break end
|
|
||||||
last_match = found
|
|
||||||
search_pos = found + mips_atom_ptr_len
|
|
||||||
end
|
|
||||||
if not last_match then return nil, nil end
|
|
||||||
|
|
||||||
local pos = last_match + mips_atom_ptr_len
|
|
||||||
while pos < before_pos do
|
|
||||||
-- skip whitespace
|
|
||||||
while pos <= #source do
|
|
||||||
local c = source:sub(pos, pos)
|
|
||||||
if c == " " or c == "\t" or c == "\n" or c == "\r" then
|
|
||||||
pos = pos + 1
|
|
||||||
else
|
|
||||||
break
|
|
||||||
end
|
|
||||||
end
|
|
||||||
if pos > #source then break end
|
|
||||||
-- skip line comments
|
|
||||||
if source:sub(pos, pos + 1) == "//" then
|
|
||||||
while pos <= #source and source:sub(pos, pos) ~= "\n" do pos = pos + 1 end
|
|
||||||
pos = pos + 1
|
|
||||||
goto continue
|
|
||||||
end
|
|
||||||
-- skip block comments
|
|
||||||
if source:sub(pos, pos + 1) == "/*" then
|
|
||||||
local close = source:find("*/", pos + 2, true)
|
|
||||||
if not close then break end
|
|
||||||
pos = close + 2
|
|
||||||
goto continue
|
|
||||||
end
|
|
||||||
-- try to read an ident
|
|
||||||
local ident, ident_end = M.read_ident(source, pos)
|
|
||||||
if not ident then break end
|
|
||||||
-- check if the next non-ws char after ident is "("
|
|
||||||
local next_pos = M.skip_ws_and_cmt(source, ident_end)
|
|
||||||
if source:sub(next_pos, next_pos) == "(" then
|
|
||||||
local inner, after_paren = M.read_parens(source, next_pos)
|
|
||||||
if inner then
|
|
||||||
return ident, inner, ident, after_paren
|
|
||||||
end
|
|
||||||
end
|
|
||||||
-- ident not followed by "(" — it's a qualifier; skip it
|
|
||||||
pos = ident_end
|
|
||||||
::continue::
|
|
||||||
end
|
|
||||||
return nil, nil
|
|
||||||
end
|
|
||||||
|
|
||||||
return M
|
|
||||||
@@ -1,725 +0,0 @@
|
|||||||
--- duffle_isa.lua — encoder / GTE / hardware tables.
|
|
||||||
local M = {}
|
|
||||||
|
|
||||||
-- Section 7: domain tables
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
|
||||||
|
|
||||||
-- atom_info sub-calls: atom_bind, atom_reads, atom_writes, atom_view, atom_reg_types, atom_ctx, atom_phase.
|
|
||||||
M.TAPE_ATOM_MACROS = {
|
|
||||||
["atom_info"] = { kind = "info", binds = false },
|
|
||||||
}
|
|
||||||
|
|
||||||
-- Empty C macros that prefix the next encoder. Zero words.
|
|
||||||
-- BdSlot_ nop is one nop word. The marker is not the BD instruction.
|
|
||||||
M.DELAY_MARKERS = {
|
|
||||||
["GteDelay_"] = true,
|
|
||||||
["LdSlot_"] = true,
|
|
||||||
["BdSlot_"] = true,
|
|
||||||
["DmaSlot_"] = true,
|
|
||||||
}
|
|
||||||
|
|
||||||
-- One row per encoder. Old table names are load-time views (build_isa_views).
|
|
||||||
M.INSTRUCTION = {
|
|
||||||
["BdSlot_"] = { cycles = 0, kind = "marker", },
|
|
||||||
["LdSlot_"] = { cycles = 0, kind = "marker", },
|
|
||||||
["add_s"] = { cycles = 1, kind = "alu", },
|
|
||||||
["add_si"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, },}, },
|
|
||||||
["add_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
|
||||||
["add_u_self"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, value = { dest = 1, op = "add_u", sources = { 1, 2 }, }, },
|
|
||||||
["add_ui"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, value = { dest = 1, immediate = 3, op = "add_ui", source = 2, }, },
|
|
||||||
["add_ui_self"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = { 1 }, imm = { { arg = 2, signed = true, width = 16, }, }, value = { dest = 1, immediate = 2, op = "add_ui", source = 1, }, },
|
|
||||||
["and"] = { cycles = 1, kind = "alu", },
|
|
||||||
["and_i"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, width = 16, }, }, value = { dest = 1, immediate = 3, op = "and_i", source = 2, }, },
|
|
||||||
["and_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
|
||||||
["atom_bind"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
|
||||||
["atom_info"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
|
||||||
["atom_label"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
|
||||||
["atom_offset"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
|
||||||
["atom_reads"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
|
||||||
["atom_writes"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
|
||||||
["branch_equal"] = { cycles = 2, kind = "branch", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
|
||||||
["branch_ge_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
|
|
||||||
["branch_gt_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
|
|
||||||
["branch_le_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
|
|
||||||
["branch_lt_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
|
|
||||||
["branch_ne"] = { cycles = 2, kind = "branch", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
|
||||||
["call_addr"] = { cycles = 2, kind = "call", reads = {}, writes = { 1 }, },
|
|
||||||
["call_reg"] = { cycles = 2, kind = "call", reads = { 1 }, writes = { 2 }, },
|
|
||||||
["div_s"] = { cycles = 35, kind = "alu", reads = { 1, 2 }, writes = {}, },
|
|
||||||
["div_u"] = { cycles = 35, kind = "alu", reads = { 1, 2 }, writes = {}, },
|
|
||||||
["gte_load_v0"] = { cycles = 2, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
|
|
||||||
["gte_load_v0v1v2"] = { cycles = 6, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
|
|
||||||
["gte_load_v1"] = { cycles = 2, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
|
|
||||||
["gte_load_v2"] = { cycles = 2, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
|
|
||||||
["gte_lw"] = { cycles = 1, kind = "load", reads = { 2 }, writes = {}, },
|
|
||||||
["gte_lwc2"] = { cycles = 1, kind = "load", },
|
|
||||||
["gte_mv_from_ctrl_r"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = { 1 }, },
|
|
||||||
["gte_mv_from_data_r"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = { 1 }, },
|
|
||||||
["gte_mv_to_ctrl_r"] = { cycles = 1, kind = "cop2_xfer", reads = { 1 }, writes = {}, },
|
|
||||||
["gte_mv_to_data_r"] = { cycles = 1, kind = "cop2_xfer", reads = { 1 }, writes = {}, },
|
|
||||||
["gte_stotz"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = {}, },
|
|
||||||
["gte_stsxy3"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = {}, },
|
|
||||||
["gte_sw"] = { cycles = 1, kind = "store", reads = { 2 }, writes = {}, },
|
|
||||||
["gte_swc2"] = { cycles = 1, kind = "store", },
|
|
||||||
["jump"] = { cycles = 2, kind = "jump", reads = {}, writes = {}, },
|
|
||||||
["jump_link"] = { cycles = 2, kind = "call", reads = { 1 }, writes = { 2 }, },
|
|
||||||
["jump_reg"] = { cycles = 2, kind = "jump", reads = { 1 }, writes = {}, suppress_arg1 = { R_AtomJmp = "fixed mac_yield handshake", }, },
|
|
||||||
["jump_rel"] = { cycles = 2, kind = "branch", delay_slot = true, },
|
|
||||||
["li_s"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, value = { dest = 1, immediate = 3, op = "add_ui", source = 2, }, },
|
|
||||||
["load_byte"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
|
||||||
["load_byte_u"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
|
||||||
["load_half"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
|
||||||
["load_half_u"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
|
||||||
["load_imm"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, },
|
|
||||||
["load_ui"] = { cycles = 1, kind = "alu", reads = {}, writes = { 1 }, },
|
|
||||||
["load_upper_i"] = { cycles = 1, kind = "alu", reads = {}, writes = { 1 }, imm = { { arg = 2, width = 16, }, }, value = { dest = 1, immediate = 2, op = "load_upper_i", }, },
|
|
||||||
["load_word"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
|
||||||
["mac_yield"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
|
||||||
["mask_upper"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, },
|
|
||||||
["mov_from_high"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, },
|
|
||||||
["mov_from_low"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, },
|
|
||||||
["mov_to_high"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = {}, },
|
|
||||||
["mov_to_low"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = {}, },
|
|
||||||
["mult_s"] = { cycles = 12, kind = "alu", reads = { 1, 2 }, writes = {}, },
|
|
||||||
["mult_u"] = { cycles = 12, kind = "alu", reads = { 1, 2 }, writes = {}, },
|
|
||||||
["nop"] = { cycles = 1, kind = "nop", reads = {}, writes = {}, },
|
|
||||||
["nop2"] = { cycles = 2, kind = "nop", reads = {}, writes = {}, },
|
|
||||||
["nor_u"] = { cycles = 1, kind = "alu", },
|
|
||||||
["or_i"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, width = 16, }, }, value = { dest = 1, immediate = 3, op = "or_i", source = 2, }, },
|
|
||||||
["or_i_self"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = { 1 }, imm = { { arg = 2, width = 16, }, }, value = { dest = 1, immediate = 2, op = "or_i", source = 1, }, },
|
|
||||||
["or_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
|
||||||
["or_u_self"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, value = { dest = 1, op = "or", sources = { 1, 2 }, }, },
|
|
||||||
["set_lt_s"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
|
||||||
["set_lt_si"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, },
|
|
||||||
["set_lt_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
|
||||||
["set_lt_ui"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, },
|
|
||||||
["shift_aright"] = { cycles = 1, kind = "alu", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
|
|
||||||
["shift_aright_var"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
|
|
||||||
["shift_lleft"] = { cycles = 1, kind = "alu", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
|
|
||||||
["shift_lleft_self"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = { 1 }, imm = { { arg = 2, width = 5, }, }, value = { dest = 1, immediate = 2, op = "shift_lleft", source = 1, }, },
|
|
||||||
["shift_lleft_var"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
|
||||||
["shift_lright"] = { cycles = 1, kind = "alu", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
|
|
||||||
["slt_s"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
|
||||||
["slt_si"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
|
||||||
["slt_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
|
||||||
["slt_ui"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16,}, }, },
|
|
||||||
["store_byte"] = { cycles = 1, kind = "store", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
|
||||||
["store_half"] = { cycles = 1, kind = "store", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
|
||||||
["store_word"] = { cycles = 1, kind = "store", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
|
||||||
["sub_s"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
|
||||||
["sub_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
|
||||||
["sys_mov_from_cop0"] = { cycles = 1, kind = "cop0_xfer", reads = {}, writes = { 1 }, },
|
|
||||||
["sys_mov_to_cop0"] = { cycles = 1, kind = "cop0_xfer", reads = { 1 }, writes = {}, },
|
|
||||||
["xor_i"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, width = 16, }, }, value = { dest = 1, immediate = 3, op = "xor_i", source = 2, }, },
|
|
||||||
["xor_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
|
||||||
}
|
|
||||||
|
|
||||||
-- One row per GTE command. Alias cycle numbers live here, not on INSTRUCTION.
|
|
||||||
M.GTE_COMMAND = {
|
|
||||||
["gte_cmdw_avsz3"] = {
|
|
||||||
aliases = { "gte_avg_sort_z3", "gte_avsz3", "gte_cmdw_avg_sort_z3" },
|
|
||||||
cycles = 5,
|
|
||||||
inputs = { "C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3", "gte_cr_ZSF3" },
|
|
||||||
outputs = {
|
|
||||||
{ register = "C2_OTZ", role = "otz", },
|
|
||||||
},
|
|
||||||
latch = {
|
|
||||||
{ register = "C2_OTZ", required = 4, },
|
|
||||||
},
|
|
||||||
},
|
|
||||||
["gte_cmdw_avsz4"] = {
|
|
||||||
aliases = { "gte_avg_sort_z4", "gte_avsz4", "gte_cmdw_avg_sort_z4" },
|
|
||||||
cycles = 6,
|
|
||||||
inputs = { "C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3", "gte_cr_ZSF4" },
|
|
||||||
outputs = {
|
|
||||||
{ register = "C2_OTZ", role = "otz", },
|
|
||||||
},
|
|
||||||
latch = {
|
|
||||||
{ register = "C2_OTZ", required = 4, },
|
|
||||||
},
|
|
||||||
},
|
|
||||||
["gte_cmdw_gpf"] = {
|
|
||||||
aliases = {},
|
|
||||||
cycles = 5,
|
|
||||||
inputs = { "C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3" },
|
|
||||||
outputs = {
|
|
||||||
{ register = "C2_MAC1", role = "mac_result", },
|
|
||||||
{ register = "C2_MAC2", role = "mac_result", },
|
|
||||||
{ register = "C2_MAC3", role = "mac_result", },
|
|
||||||
{ register = "C2_IR1", role = "latest_color", },
|
|
||||||
{ register = "C2_IR2", role = "latest_color", },
|
|
||||||
{ register = "C2_IR3", role = "latest_color", },
|
|
||||||
},
|
|
||||||
latch = {
|
|
||||||
{ register = "C2_MAC1", required = 4, },
|
|
||||||
{ register = "C2_MAC2", required = 4, },
|
|
||||||
{ register = "C2_MAC3", required = 4, },
|
|
||||||
{ register = "C2_IR1", required = 4, },
|
|
||||||
{ register = "C2_IR2", required = 4, },
|
|
||||||
{ register = "C2_IR3", required = 4, },
|
|
||||||
},
|
|
||||||
},
|
|
||||||
["gte_cmdw_mvmva"] = {
|
|
||||||
aliases = {},
|
|
||||||
cycles = 8,
|
|
||||||
inputs = {
|
|
||||||
"C2_VXY0", "C2_VZ0",
|
|
||||||
"C2_VXY1", "C2_VZ1",
|
|
||||||
"C2_VXY2", "C2_VZ2",
|
|
||||||
"C2_IR1", "C2_IR2", "C2_IR3",
|
|
||||||
"gte_cr_RT11", "gte_cr_RT12", "gte_cr_RT13",
|
|
||||||
"gte_cr_RT21", "gte_cr_RT22", "gte_cr_RT23",
|
|
||||||
"gte_cr_RT31", "gte_cr_RT32", "gte_cr_RT33",
|
|
||||||
"gte_cr_TRX", "gte_cr_TRY", "gte_cr_TRZ"
|
|
||||||
},
|
|
||||||
outputs = {
|
|
||||||
{ register = "C2_IR1", role = "latest_color", },
|
|
||||||
{ register = "C2_IR2", role = "latest_color", },
|
|
||||||
{ register = "C2_IR3", role = "latest_color", },
|
|
||||||
},
|
|
||||||
latch = {
|
|
||||||
{ register = "C2_IR1", required = 4, },
|
|
||||||
{ register = "C2_IR2", required = 4, },
|
|
||||||
{ register = "C2_IR3", required = 4, },
|
|
||||||
},
|
|
||||||
},
|
|
||||||
["gte_cmdw_nclip"] = {
|
|
||||||
aliases = { "gte_nclip" },
|
|
||||||
cycles = 8,
|
|
||||||
inputs = { "C2_SXY0", "C2_SXY1", "C2_SXY2" },
|
|
||||||
outputs = {
|
|
||||||
{ register = "C2_SZ3", role = "mac_result", },
|
|
||||||
},
|
|
||||||
latch = {
|
|
||||||
{ register = "C2_SZ3", required = 4, },
|
|
||||||
},
|
|
||||||
},
|
|
||||||
["gte_cmdw_op"] = {
|
|
||||||
aliases = { "gte_cmdw_outer_product", "gte_cmdw_wedge" },
|
|
||||||
cycles = 6,
|
|
||||||
inputs = {},
|
|
||||||
outputs = {
|
|
||||||
{ register = "C2_IR1", role = "latest_color", },
|
|
||||||
{ register = "C2_IR2", role = "latest_color", },
|
|
||||||
{ register = "C2_IR3", role = "latest_color", },
|
|
||||||
},
|
|
||||||
latch = {
|
|
||||||
{ register = "C2_IR1", required = 4, },
|
|
||||||
{ register = "C2_IR2", required = 4, },
|
|
||||||
{ register = "C2_IR3", required = 4, },
|
|
||||||
},
|
|
||||||
},
|
|
||||||
["gte_cmdw_rtps"] = {
|
|
||||||
aliases = { "gte_cmdw_rotate_translate_perspective_single", "gte_rtps" },
|
|
||||||
cycles = 15,
|
|
||||||
inputs = {
|
|
||||||
"C2_VXY0", "C2_VZ0",
|
|
||||||
"C2_VXY1", "C2_VZ1",
|
|
||||||
"C2_VXY2", "C2_VZ2",
|
|
||||||
"C2_RGB", "C2_OTZ",
|
|
||||||
"C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3",
|
|
||||||
"C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3",
|
|
||||||
"gte_cr_RT11", "gte_cr_RT12", "gte_cr_RT13",
|
|
||||||
"gte_cr_RT21", "gte_cr_RT22", "gte_cr_RT23",
|
|
||||||
"gte_cr_RT31", "gte_cr_RT32", "gte_cr_RT33",
|
|
||||||
"gte_cr_TRX", "gte_cr_TRY", "gte_cr_TRZ",
|
|
||||||
"gte_cr_OFX", "gte_cr_OFY",
|
|
||||||
"gte_cr_H",
|
|
||||||
"gte_cr_DQA", "gte_cr_DQB"
|
|
||||||
},
|
|
||||||
outputs = {
|
|
||||||
{ register = "C2_SXY2", role = "latest_screen_xy", },
|
|
||||||
{ register = "C2_SZ2", role = "latest_screen_z", },
|
|
||||||
{ register = "C2_OTZ", role = "otz", },
|
|
||||||
{ register = "C2_IR0", role = "latest_color", },
|
|
||||||
},
|
|
||||||
latch = {
|
|
||||||
{ register = "C2_SXY2", required = 4, },
|
|
||||||
{ register = "C2_SZ2", required = 4, },
|
|
||||||
{ register = "C2_OTZ", required = 4, },
|
|
||||||
{ register = "C2_IR0", required = 4, },
|
|
||||||
},
|
|
||||||
},
|
|
||||||
["gte_cmdw_rtpt"] = {
|
|
||||||
aliases = { "gte_cmdw_rotate_translate_perspective_triple", "gte_rtpt" },
|
|
||||||
cycles = 23,
|
|
||||||
inputs = {
|
|
||||||
"C2_VXY0", "C2_VZ0",
|
|
||||||
"C2_VXY1", "C2_VZ1",
|
|
||||||
"C2_VXY2", "C2_VZ2",
|
|
||||||
"C2_RGB", "C2_OTZ",
|
|
||||||
"C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3",
|
|
||||||
"C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3",
|
|
||||||
"gte_cr_RT11", "gte_cr_RT12", "gte_cr_RT13",
|
|
||||||
"gte_cr_RT21", "gte_cr_RT22", "gte_cr_RT23",
|
|
||||||
"gte_cr_RT31", "gte_cr_RT32", "gte_cr_RT33",
|
|
||||||
"gte_cr_TRX", "gte_cr_TRY", "gte_cr_TRZ",
|
|
||||||
"gte_cr_OFX", "gte_cr_OFY",
|
|
||||||
"gte_cr_H",
|
|
||||||
"gte_cr_DQA", "gte_cr_DQB"
|
|
||||||
},
|
|
||||||
outputs = {
|
|
||||||
{ register = "C2_SXY0", role = "screen_xy[0]", },
|
|
||||||
{ register = "C2_SXY1", role = "screen_xy[1]", },
|
|
||||||
{ register = "C2_SXY2", role = "latest_screen_xy", },
|
|
||||||
{ register = "C2_SZ3", role = "latest_screen_z", },
|
|
||||||
{ register = "C2_OTZ", role = "otz", },
|
|
||||||
},
|
|
||||||
latch = {
|
|
||||||
{ register = "C2_SXY0", required = 4, },
|
|
||||||
{ register = "C2_SXY1", required = 4, },
|
|
||||||
{ register = "C2_SXY2", required = 4, },
|
|
||||||
{ register = "C2_SZ3", required = 4, },
|
|
||||||
{ register = "C2_OTZ", required = 4, },
|
|
||||||
},
|
|
||||||
},
|
|
||||||
["gte_cmdw_sqr"] = {
|
|
||||||
aliases = {},
|
|
||||||
cycles = 5,
|
|
||||||
inputs = { "C2_IR1", "C2_IR2", "C2_IR3" },
|
|
||||||
outputs = {
|
|
||||||
{ register = "C2_MAC1", role = "mac_result", },
|
|
||||||
{ register = "C2_MAC2", role = "mac_result", },
|
|
||||||
{ register = "C2_MAC3", role = "mac_result", },
|
|
||||||
{ register = "C2_IR1", role = "latest_color", },
|
|
||||||
{ register = "C2_IR2", role = "latest_color", },
|
|
||||||
{ register = "C2_IR3", role = "latest_color", },
|
|
||||||
},
|
|
||||||
latch = {
|
|
||||||
{ register = "C2_MAC1", required = 4, },
|
|
||||||
{ register = "C2_MAC2", required = 4, },
|
|
||||||
{ register = "C2_MAC3", required = 4, },
|
|
||||||
{ register = "C2_IR1", required = 4, },
|
|
||||||
{ register = "C2_IR2", required = 4, },
|
|
||||||
{ register = "C2_IR3", required = 4, },
|
|
||||||
},
|
|
||||||
},
|
|
||||||
}
|
|
||||||
|
|
||||||
function M.instr (ident) return M.INSTRUCTION [ident] end
|
|
||||||
function M.gte_canon(ident) return M.ALIAS_TO_CANONICAL [ident] or ident end
|
|
||||||
function M.gte (ident) return M.GTE_COMMAND[M.gte_canon(ident)] end
|
|
||||||
|
|
||||||
local function build_isa_views()
|
|
||||||
M.ALIAS_TO_CANONICAL = {}
|
|
||||||
for canon, row in pairs(M.GTE_COMMAND) do
|
|
||||||
M.ALIAS_TO_CANONICAL[canon] = canon
|
|
||||||
for _, alias in ipairs(row.aliases or {}) do
|
|
||||||
M.ALIAS_TO_CANONICAL[alias] = canon
|
|
||||||
end
|
|
||||||
end
|
|
||||||
M.INSTRUCTION_LATENCY = {}
|
|
||||||
M.INSTRUCTION_GPR_EFFECTS = {}
|
|
||||||
M.IMMEDIATE_FIELD_WIDTHS = {}
|
|
||||||
M.GPR_VALUE_RULES = {}
|
|
||||||
M.CONTROL_TRANSFER_DELAY_SLOT_POLICIES = {}
|
|
||||||
for name, row in pairs(M.INSTRUCTION) do
|
|
||||||
M.INSTRUCTION_LATENCY[name] = row.cycles
|
|
||||||
if row.reads or row.writes then
|
|
||||||
M.INSTRUCTION_GPR_EFFECTS[name] = {
|
|
||||||
reads = row.reads or {},
|
|
||||||
writes = row.writes or {},
|
|
||||||
}
|
|
||||||
end
|
|
||||||
if row.imm then M.IMMEDIATE_FIELD_WIDTHS[name] = row.imm end
|
|
||||||
if row.value then M.GPR_VALUE_RULES [name] = row.value end
|
|
||||||
if (row.kind == "branch" or row.kind == "jump" or row.kind == "call")
|
|
||||||
and row.delay_slot ~= false then
|
|
||||||
M.CONTROL_TRANSFER_DELAY_SLOT_POLICIES[name] = {
|
|
||||||
family = row.kind,
|
|
||||||
suppress_arg1 = row.suppress_arg1,
|
|
||||||
}
|
|
||||||
end
|
|
||||||
end
|
|
||||||
M.GTE_COMMAND_ALIASES = {}
|
|
||||||
M.GTE_COMMAND_INPUTS = {}
|
|
||||||
M.GTE_COMMAND_OUTPUTS = {}
|
|
||||||
M.GTE_COMMAND_LATCH_WINDOWS = {}
|
|
||||||
for canon, row in pairs(M.GTE_COMMAND) do
|
|
||||||
M.GTE_COMMAND_ALIASES [canon] = canon
|
|
||||||
M.INSTRUCTION_LATENCY [canon] = row.cycles
|
|
||||||
M.INSTRUCTION_GPR_EFFECTS[canon] = { reads = {}, writes = {} }
|
|
||||||
for _, alias in ipairs(row.aliases or {}) do
|
|
||||||
M.GTE_COMMAND_ALIASES [alias] = canon
|
|
||||||
M.INSTRUCTION_LATENCY [alias] = row.cycles
|
|
||||||
M.INSTRUCTION_GPR_EFFECTS[alias] = { reads = {}, writes = {} }
|
|
||||||
end
|
|
||||||
M.GTE_COMMAND_INPUTS [canon] = row.inputs
|
|
||||||
M.GTE_COMMAND_OUTPUTS [canon] = row.outputs
|
|
||||||
M.GTE_COMMAND_LATCH_WINDOWS[canon] = row.latch
|
|
||||||
end
|
|
||||||
end
|
|
||||||
build_isa_views()
|
|
||||||
|
|
||||||
|
|
||||||
--- GTE control-register alias groups.
|
|
||||||
--- Aliases within a group write to the same C2 control-register slot (the HW double-maps some C2 slots across multiple PSX SDK / libgte conventions).
|
|
||||||
--- Aliases across groups write to distinct C2 slots.
|
|
||||||
---
|
|
||||||
--- Cross-alias writes inside one atom body, or across the wave-context boundary, silently clobber each other.
|
|
||||||
--- The `check_gte_cr_alias_writes` check warns about each pair per source. See `docs/gte_reference.md` §"Control-register alias table"
|
|
||||||
--- for the HW rationale and the libgte outer-product convention.
|
|
||||||
M.GTE_CR_ALIAS_GROUPS = {
|
|
||||||
{ 24, { "gte_cr_RBK", "gte_cr_OFX" } }, -- background R vs screen offset X
|
|
||||||
{ 25, { "gte_cr_GBK", "gte_cr_OFY" } }, -- background G vs screen offset Y
|
|
||||||
{ 26, { "gte_cr_BBK", "gte_cr_H" } }, -- background B vs projection plane distance H
|
|
||||||
}
|
|
||||||
|
|
||||||
-- Packed RT slots named by the gte.h packed-slot comment. First must be written before second.
|
|
||||||
M.GTE_PACKED_SLOT_RELATIONS = {
|
|
||||||
{ slot = 2, first = "gte_cr_RT13", second = "gte_cr_RT22" },
|
|
||||||
}
|
|
||||||
|
|
||||||
-- Operand-class table for the COP2->GPR load-delay check.
|
|
||||||
-- Maps each emitting-token ident to the set of GPR operand positions it reads.
|
|
||||||
-- Covers the current encoder vocabulary (`code/duffle/mips.h` + `code/duffle/gte.h`); add rows here as new encoders land.
|
|
||||||
--
|
|
||||||
-- Semantics:
|
|
||||||
-- * A "GPR operand position" is the textual slot in the macro's argument list, 1-based; e.g. `load_word(rt, base, off)` has positional operands 1 (rt), 2 (base), 3 (off).
|
|
||||||
-- The table reads operands 1 + 2 + 3 to find what GPRs the macro touches.
|
|
||||||
-- * The check tracks one entry per destination GPR per MFC2 / CFC2 event.
|
|
||||||
-- A subsequent event counts as a "use" iff any of its read operand positions reference that destination GPR's ident (e.g. `R_T0`).
|
|
||||||
-- * Branch delay slots are out of scope (MIPS control-flow; tracked separately).
|
|
||||||
M.OPERAND_READ_POSITIONS = {
|
|
||||||
-- CPU ALU with one or two GPR operands. Reads every GPR operand.
|
|
||||||
["add_ui"] = {1, 2},
|
|
||||||
["li_s"] = {1, 2}, -- rt (write), imm16 (immediate)
|
|
||||||
["add_ui_self"] = {1},
|
|
||||||
["add_si"] = {1, 2},
|
|
||||||
["add_u"] = {1, 2, 3},
|
|
||||||
["add_u_self"] = {1, 2},
|
|
||||||
["sub_s"] = {1, 2, 3},
|
|
||||||
["sub_u"] = {1, 2, 3},
|
|
||||||
["and_i"] = {1, 2},
|
|
||||||
["and"] = {1, 2, 3},
|
|
||||||
["or_i"] = {1, 2},
|
|
||||||
["or_i_self"] = {1},
|
|
||||||
["or"] = {1, 2, 3},
|
|
||||||
["or_self"] = {1, 2},
|
|
||||||
["xor_i"] = {1, 2},
|
|
||||||
["xor"] = {1, 2, 3},
|
|
||||||
["slt_s"] = {1, 2, 3},
|
|
||||||
["slt_u"] = {1, 2, 3},
|
|
||||||
["slt_si"] = {1, 2},
|
|
||||||
["slt_ui"] = {1, 2},
|
|
||||||
["mult_s"] = {1, 2},
|
|
||||||
["mult_u"] = {1, 2},
|
|
||||||
["div_s"] = {1, 2},
|
|
||||||
["div_u"] = {1, 2},
|
|
||||||
-- Shifts: shift_lleft(rd, rt, shamt); the rt operand is the value, rd is dest.
|
|
||||||
["shift_lleft"] = {1, 2},
|
|
||||||
["shift_lright"] = {1, 2},
|
|
||||||
["shift_aright"] = {1, 2},
|
|
||||||
["shift_lleft_self"] = {1},
|
|
||||||
-- Loads: load_word(rt, base, off); the rt operand is the destination (it's written, not read) and base + off are non-GPR operands.
|
|
||||||
-- The check treats the rt operand as a write, so the read-positions table for `load_*` is empty.
|
|
||||||
["load_word"] = {},
|
|
||||||
["load_half_u"] = {},
|
|
||||||
["load_byte_u"] = {},
|
|
||||||
["load_half"] = {},
|
|
||||||
["load_byte"] = {},
|
|
||||||
["load_upper_i"] = {},
|
|
||||||
["load_ui"] = {},
|
|
||||||
-- Stores write to memory; base + rt operands are non-read for load-delay purposes.
|
|
||||||
["store_word"] = {},
|
|
||||||
["store_half"] = {},
|
|
||||||
["store_byte"] = {},
|
|
||||||
-- Branches read rs (+ rt for beq/bne). The branch delay slot is out of scope.
|
|
||||||
["branch_equal"] = {1, 2},
|
|
||||||
["branch_ne"] = {1, 2},
|
|
||||||
["branch_le_zero"] = {1},
|
|
||||||
["branch_lt_zero"] = {1},
|
|
||||||
["branch_ge_zero"] = {1},
|
|
||||||
["branch_gt_zero"] = {1},
|
|
||||||
-- Jumps / link: jr / jalr read rs only (the target). RD is the destination link.
|
|
||||||
["jump_reg"] = {1},
|
|
||||||
["jump_link"] = {1},
|
|
||||||
["call_reg"] = {1},
|
|
||||||
["call_addr"] = {},
|
|
||||||
["jump"] = {},
|
|
||||||
-- mask_upper is a 2-word macro: shift_lleft then shift_lright. The first reads rt.
|
|
||||||
["mask_upper"] = {1, 2},
|
|
||||||
-- move from/to HI/LO.
|
|
||||||
["mov_from_high"] = {},
|
|
||||||
["mov_from_low"] = {},
|
|
||||||
["mov_to_high"] = {1},
|
|
||||||
["mov_to_low"] = {1},
|
|
||||||
-- GTE transfers / loads / stores / commands: the relevant table values live in the check itself.
|
|
||||||
-- `gte_mv_to_*` writes its rt operand; `gte_mv_from_*` writes its rt operand; `gte_*` commands are atomic-from-the-CPU-POV
|
|
||||||
-- once they issue (the CPU holds until the command completes, so load-delay violations don't surface here).
|
|
||||||
["gte_mv_from_data_r"] = {},
|
|
||||||
["gte_mv_from_ctrl_r"] = {},
|
|
||||||
["gte_mv_to_data_r"] = {},
|
|
||||||
["gte_mv_to_ctrl_r"] = {},
|
|
||||||
["gte_lw"] = {},
|
|
||||||
["gte_sw"] = {},
|
|
||||||
["shift_lleft_var"] = {1, 2, 3}, -- rd, rt, rs (variable shift amount)
|
|
||||||
["shift_aright_var"] = {1, 2, 3},
|
|
||||||
}
|
|
||||||
|
|
||||||
-- GP0 packet sizes (total words including the 1-word tag) per GP0 cmd byte.
|
|
||||||
-- Per PSX-SPX `docs/psx-spx/docs/graphicsprocessingunitgpu.md` §"GPU Render Polygon Commands":
|
|
||||||
-- Each polygon command's word count = 1 (tag/cmd) + per-vertex (vertex + optional color + optional UV).
|
|
||||||
-- F3: cmd + 3 vertices = 4 words; +1 tag = 5
|
|
||||||
-- F4: cmd + 4 vertices = 5 words; +1 tag = 6
|
|
||||||
-- G3: cmd + 3×(color + vertex) = 6 words; +1 tag = 7
|
|
||||||
-- G4: cmd + 4×(color + vertex) = 8 words; +1 tag = 9
|
|
||||||
-- FT3: cmd + tpage + clut + 3×(vertex + UV) = 7 words; +1 tag = 8
|
|
||||||
-- FT4: cmd + tpage + clut + 4×(vertex + UV) = 9 words; +1 tag = 10
|
|
||||||
-- GT3: cmd + tpage + clut + 3×(color + vertex + UV) = 9 words; +1 tag = 10
|
|
||||||
-- GT4: cmd + tpage + clut + 4×(color + vertex + UV) = 12 words; +1 tag = 13
|
|
||||||
--
|
|
||||||
-- Cross-checked against code/duffle/gp.h struct sizes + the set_poly_* macros
|
|
||||||
-- (which encode "len" = "words after tag"):
|
|
||||||
-- set_poly_f3(p) -> set_len(p, 4) -> 5 total GP0 0x20
|
|
||||||
-- set_poly_ft3(p) -> set_len(p, 7) -> 8 total GP0 0x24
|
|
||||||
-- set_poly_f4(p) -> set_len(p, 5) -> 6 total GP0 0x28
|
|
||||||
-- set_poly_ft4(p) -> set_len(p, 9) -> 10 total GP0 0x2C
|
|
||||||
-- set_poly_g3(p) -> set_len(p, 6) -> 7 total GP0 0x30
|
|
||||||
-- set_poly_gt3(p) -> set_len(p, 9) -> 10 total GP0 0x34
|
|
||||||
-- set_poly_g4(p) -> set_len(p, 8) -> 9 total GP0 0x38
|
|
||||||
-- set_poly_gt4(p) -> set_len(p, 12) -> 13 total GP0 0x3C
|
|
||||||
M.GP0_CMD_SIZE = {
|
|
||||||
[0x20] = 5, -- Poly_F3
|
|
||||||
[0x24] = 8, -- Poly_FT3
|
|
||||||
[0x28] = 6, -- Poly_F4
|
|
||||||
[0x2C] = 10, -- Poly_FT4
|
|
||||||
[0x30] = 7, -- Poly_G3
|
|
||||||
[0x34] = 10, -- Poly_GT3
|
|
||||||
[0x38] = 9, -- Poly_G4
|
|
||||||
[0x3C] = 13, -- Poly_GT4
|
|
||||||
}
|
|
||||||
|
|
||||||
-- Shape suffix (after `ac_format_` / `mac_format_` prefix) -> GP0 cmd byte.
|
|
||||||
-- Lets the static-analysis check derive the cmd byte from a macro name like `mac_format_g4_color` -> `g4` -> 0x38 -> 9 expected words.
|
|
||||||
M.GP0_CMD_BY_SHAPE = {
|
|
||||||
["f3"] = 0x20, ["ft3"] = 0x24,
|
|
||||||
["f4"] = 0x28, ["ft4"] = 0x2C,
|
|
||||||
["g3"] = 0x30, ["gt3"] = 0x34,
|
|
||||||
["g4"] = 0x38, ["gt4"] = 0x3C,
|
|
||||||
}
|
|
||||||
|
|
||||||
M.UNKNOWN_INSTRUCTION_CYCLES = 1
|
|
||||||
|
|
||||||
-- Hardware-relation policy table.
|
|
||||||
--
|
|
||||||
-- The forward walker in `passes/static_analysis.lua::analyze_hardware_relations` reads every emitted word_event, matches its `encoder` against `row.token`, and:
|
|
||||||
-- * stages the event as a producer in `atom.paths.forward_state`; or
|
|
||||||
-- * matches it as a consumer against pending producers and records a hazard on `atom.paths.hazards` when the gap is below `visibility.required`.
|
|
||||||
--
|
|
||||||
-- Each row is the contract for one CPU-to-coprocessor transfer semantic (the coprocessor-to-CPU path mirrors the same shape).
|
|
||||||
-- The `reads` / `writes` sub-tables carry the argument positions the analyzer inspects:
|
|
||||||
-- * `writes.arg` is the destination operand (the producer's effect); the analyzer stages this register as a pending producer.
|
|
||||||
-- * `reads` (when present) lists the operand positions the same token reads back from hardware; for MTC2 / CTC2 the producer reads the GPR source it is loading from.
|
|
||||||
-- The `fanout_to` field (MTC2-IRGB row only) tells the consumer-match logic which downstream COP2 registers are transitively updated by the write.
|
|
||||||
--
|
|
||||||
-- Visibility semantics:
|
|
||||||
-- * `kind = "post_producer_words"` means the consumer observes the producer's effect after `required` independent emitted words that are
|
|
||||||
-- strictly between the producer and the consumer. The producer's own emitted slot is implicit (it counts as the slot of issue, not toward `required`)
|
|
||||||
-- per the PSX-SPX rule: "Store delays are counted in numbers of clock cycles (not in numbers of opcodes).
|
|
||||||
-- For 3 cycle delay, one must usually insert 3 cached opcodes (or one uncached opcode)."
|
|
||||||
-- * `required` is the minimum count of intervening emitted words between producer and consumer.
|
|
||||||
-- `required = 0` permits the consumer on the very next slot; `required < 0` would place the consumer on the same slot as the producer
|
|
||||||
-- and is reserved for future "self-retires" relations.
|
|
||||||
--
|
|
||||||
-- Evidence:
|
|
||||||
-- * `evidence.confidence` is one of `"exact"`, `"conservative"`, `"unknown"`. The severity comes from `violation_kind`;
|
|
||||||
-- A hardware measurement that the vendor caveats may still classify as `"conservative"` even when the underlying timing is numerically known.
|
|
||||||
-- * `evidence.source` is the upstream reference (file + line range) the row is sourced from. New rows must carry this citation.
|
|
||||||
--
|
|
||||||
-- Consumers:
|
|
||||||
-- * passes/static_analysis.lua::analyze_hardware_relations (forward walker).
|
|
||||||
-- * passes/static_analysis.lua::transfer_hazards CHECK_RULES reader (renders hazards onto `findings`).
|
|
||||||
-- This table is consumed by the hardware-relation analyzer and hazard renderer.
|
|
||||||
M.HARDWARE_RELATIONS = {
|
|
||||||
-- CPU → COP2 data register (MTC2). The ordinary default is 2 cached words between producer and consumer (cpuspecifications.md:407-419).
|
|
||||||
{
|
|
||||||
id = "mtc2_gpr_visibility",
|
|
||||||
semantic = "MTC2",
|
|
||||||
token = "gte_mv_to_data_r",
|
|
||||||
direction = "gpr_to_cop2_data",
|
|
||||||
reads = { domain = "gpr", arg = 1 },
|
|
||||||
writes = { domain = "cop2.data", arg = 2 },
|
|
||||||
visibility = { kind = "post_producer_words", required = 2 },
|
|
||||||
evidence = {
|
|
||||||
confidence = "exact",
|
|
||||||
source = "cpuspecifications.md:407-419",
|
|
||||||
},
|
|
||||||
violation_kind = "error",
|
|
||||||
},
|
|
||||||
-- CPU → COP2 data register when the destination is C2_IRGB (data 28).
|
|
||||||
-- C2_IRGB drives the IR1/IR2/IR3 color-conversion fan-out, which extends the propagation delay to 3 cached words.
|
|
||||||
-- `destination_match = "C2_IRGB"` is the row's filter; the analyzer consults this when the producer's destination operand equals "C2_IRGB".
|
|
||||||
-- C2_ORGB (data 29) is read-only and is never classified as a writable fan-out destination.
|
|
||||||
{
|
|
||||||
id = "mtc2_irgb_visibility",
|
|
||||||
semantic = "MTC2",
|
|
||||||
token = "gte_mv_to_data_r",
|
|
||||||
direction = "gpr_to_cop2_data",
|
|
||||||
reads = { domain = "gpr", arg = 1 },
|
|
||||||
writes = { domain = "cop2.data", arg = 2 },
|
|
||||||
destination_match = "C2_IRGB",
|
|
||||||
fanout_to = { "C2_IR1", "C2_IR2", "C2_IR3" },
|
|
||||||
visibility = { kind = "post_producer_words", required = 3 },
|
|
||||||
evidence = {
|
|
||||||
confidence = "exact",
|
|
||||||
source = "cpuspecifications.md:407-419",
|
|
||||||
},
|
|
||||||
violation_kind = "error",
|
|
||||||
},
|
|
||||||
-- CPU → COP2 control register (CTC2). Ordinary minimum 2;
|
|
||||||
-- no IRGB-style fan-out exists for control registers (per spec §3.6: only C2_IRGB has the 3-cycle fan-out on the data side).
|
|
||||||
{
|
|
||||||
id = "ctc2_gpr_visibility",
|
|
||||||
semantic = "CTC2",
|
|
||||||
token = "gte_mv_to_ctrl_r",
|
|
||||||
direction = "gpr_to_cop2_control",
|
|
||||||
reads = { domain = "gpr", arg = 1 },
|
|
||||||
writes = { domain = "cop2.ctrl", arg = 2 },
|
|
||||||
visibility = { kind = "post_producer_words", required = 2 },
|
|
||||||
evidence = {
|
|
||||||
confidence = "exact",
|
|
||||||
source = "cpuspecifications.md:407-419",
|
|
||||||
},
|
|
||||||
violation_kind = "error",
|
|
||||||
},
|
|
||||||
-- COP2 data → GPR (MFC2). One cached slot between the transfer and the first GPR consumer;
|
|
||||||
-- the GPR is not updated until the instruction AFTER the MFC2 completes (geometrytransformationenginegte.md:29-32).
|
|
||||||
{
|
|
||||||
id = "mfc2_gpr_visibility",
|
|
||||||
semantic = "MFC2",
|
|
||||||
token = "gte_mv_from_data_r",
|
|
||||||
direction = "cop2_data_to_gpr",
|
|
||||||
reads = { domain = "cop2.data", arg = 2 },
|
|
||||||
writes = { domain = "gpr", arg = 1 },
|
|
||||||
visibility = { kind = "post_producer_words", required = 1 },
|
|
||||||
evidence = {
|
|
||||||
confidence = "exact",
|
|
||||||
source = "geometrytransformationenginegte.md:29-32",
|
|
||||||
},
|
|
||||||
violation_kind = "error",
|
|
||||||
},
|
|
||||||
-- COP2 control → GPR (CFC2). Same delay as MFC2 (cpuspecifications.md treats the two load-from-COP2 paths symmetrically).
|
|
||||||
{
|
|
||||||
id = "cfc2_gpr_visibility",
|
|
||||||
semantic = "CFC2",
|
|
||||||
token = "gte_mv_from_ctrl_r",
|
|
||||||
direction = "cop2_control_to_gpr",
|
|
||||||
reads = { domain = "cop2.ctrl", arg = 2 },
|
|
||||||
writes = { domain = "gpr", arg = 1 },
|
|
||||||
visibility = { kind = "post_producer_words", required = 1 },
|
|
||||||
evidence = {
|
|
||||||
confidence = "exact",
|
|
||||||
source = "cpuspecifications.md:382-419",
|
|
||||||
},
|
|
||||||
violation_kind = "error",
|
|
||||||
},
|
|
||||||
-- COP0 control → GPR (MFC0).
|
|
||||||
-- One cached slot; the analyzer treats `sys_mov_from_cop0(rt, 12)` (the SR/CU2 transfer) as the same shape as the COP2 load-delay path.
|
|
||||||
-- The semantic-level SR/CU2 transition models the load delay;
|
|
||||||
-- SR.CU2 bounded-value propagation is modeled separately).
|
|
||||||
{
|
|
||||||
id = "mfc0_gpr_visibility",
|
|
||||||
semantic = "MFC0",
|
|
||||||
token = "sys_mov_from_cop0",
|
|
||||||
direction = "cop0_control_to_gpr",
|
|
||||||
reads = { domain = "cop0.ctrl", arg = 2 },
|
|
||||||
writes = { domain = "gpr", arg = 1 },
|
|
||||||
visibility = { kind = "post_producer_words", required = 1 },
|
|
||||||
evidence = {
|
|
||||||
confidence = "exact",
|
|
||||||
source = "cpuspecifications.md:171-178",
|
|
||||||
},
|
|
||||||
violation_kind = "error",
|
|
||||||
},
|
|
||||||
-- Memory -> COP2 data register (LWC2).
|
|
||||||
-- The memory-side timing is not measured by the vendored GTE latch experiment, so this relation has no numeric retirement threshold.
|
|
||||||
-- The LWC2 destination has TWO retirement regimes (per PSX-SPX):
|
|
||||||
-- * GTE-command consumer (`gte_cmdw_*`): the GTE pipeline LATCHES the LWC2 result, so a `gte_cmdw_*`
|
|
||||||
-- in the very next slot uses the latched value. Gap = 0 is allowed. (Per `docs/psx-spx/docs/gtepipelinetimings.md:271-274`.)
|
|
||||||
-- * Any other consumer: standard MIPS load delay applies. Gap = 1 required. (Per `docs/psx-spx/docs/cpuspecifications.md:407-419`.)
|
|
||||||
-- Two separate relations so the walker can dispatch by consumer type and emit different severities
|
|
||||||
-- (the GTE-command path is `info` because the latch is intentional; the non-GTE-consumer path is `error` because the missing nop is a real bug).
|
|
||||||
{
|
|
||||||
id = "lwc2_to_gte_command",
|
|
||||||
semantic = "LWC2_to_GTE",
|
|
||||||
token = "gte_lw",
|
|
||||||
direction = "memory_to_cop2_data",
|
|
||||||
reads = { domain = "memory", arg = 2 },
|
|
||||||
writes = { domain = "cop2.data", arg = 1 },
|
|
||||||
required = 0, -- GTE-command consumer: gap = 0 OK (latched).
|
|
||||||
evidence = {
|
|
||||||
confidence = "measured",
|
|
||||||
source = "gtepipelinetimings.md:271-274",
|
|
||||||
},
|
|
||||||
violation_kind = "info",
|
|
||||||
clear_on_consumer = true,
|
|
||||||
},
|
|
||||||
{
|
|
||||||
id = "lwc2_to_other_consumer",
|
|
||||||
semantic = "LWC2_to_other",
|
|
||||||
token = "gte_lw",
|
|
||||||
direction = "memory_to_cop2_data",
|
|
||||||
reads = { domain = "memory", arg = 2 },
|
|
||||||
writes = { domain = "cop2.data", arg = 1 },
|
|
||||||
required = 1, -- Non-GTE-consumer: standard MIPS load delay.
|
|
||||||
evidence = {
|
|
||||||
confidence = "inferred",
|
|
||||||
source = "cpuspecifications.md:407-419",
|
|
||||||
},
|
|
||||||
violation_kind = "error",
|
|
||||||
clear_on_consumer = true,
|
|
||||||
},
|
|
||||||
-- COP2 data register -> memory (SWC2). A read of C2 state, not a CPU-to-COP2 write.
|
|
||||||
-- The policy row stays in for direction/provenance; staging it as a later command-input producer is suppressed.
|
|
||||||
{
|
|
||||||
id = "swc2_memory_write",
|
|
||||||
semantic = "SWC2",
|
|
||||||
token = "gte_sw",
|
|
||||||
direction = "cop2_data_to_memory",
|
|
||||||
reads = { domain = "cop2.data", arg = 1 },
|
|
||||||
writes = { domain = "memory", arg = 2 },
|
|
||||||
visibility = { kind = "none", required = 0 },
|
|
||||||
evidence = {
|
|
||||||
confidence = "exact",
|
|
||||||
source = "cpuspecifications.md:79",
|
|
||||||
},
|
|
||||||
violation_kind = "info",
|
|
||||||
stage = false,
|
|
||||||
},
|
|
||||||
-- MTC0 Status/SR.CU2. The ordinary COP0 store has no general store-delay relation;
|
|
||||||
-- this row feeds the dedicated CU2 transition logic in the same forward walk and is therefore not staged in `pending`.
|
|
||||||
{
|
|
||||||
id = "mtc0_cu2_visibility",
|
|
||||||
semantic = "MTC0",
|
|
||||||
token = "sys_mov_to_cop0",
|
|
||||||
direction = "gpr_to_cop0_status",
|
|
||||||
reads = { domain = "gpr", arg = 1 },
|
|
||||||
writes = { domain = "cop0.status", arg = 2 },
|
|
||||||
status_register = 12,
|
|
||||||
visibility = { kind = "post_producer_words", required = 2 },
|
|
||||||
evidence = {
|
|
||||||
confidence = "conservative",
|
|
||||||
source = "cpuspecifications.md:543,625-628",
|
|
||||||
},
|
|
||||||
violation_kind = "warning",
|
|
||||||
stage = false,
|
|
||||||
cu2_transition = true,
|
|
||||||
},
|
|
||||||
}
|
|
||||||
|
|
||||||
-- Bounded Status/SR.CU2 transition policy.
|
|
||||||
-- The value lattice and the transition consumer both read this immutable row; no second value pass is permitted.
|
|
||||||
-- The source says the enable/disable transition takes "2 clock cycles or so", so the boundary is conservative rather than exact.
|
|
||||||
M.CU2_TRANSITION_POLICY = {
|
|
||||||
status_register = 12,
|
|
||||||
enable_bit = 0x40000000,
|
|
||||||
required = 2,
|
|
||||||
visibility_kind = "post_producer_words",
|
|
||||||
evidence = {
|
|
||||||
confidence = "conservative",
|
|
||||||
source = "cpuspecifications.md:543,625-628",
|
|
||||||
},
|
|
||||||
}
|
|
||||||
|
|
||||||
return M
|
|
||||||
+18
-11
@@ -22,10 +22,12 @@ local M = {}
|
|||||||
local CACHE_KEY = "__duffle_repo_root__"
|
local CACHE_KEY = "__duffle_repo_root__"
|
||||||
|
|
||||||
--- Resolve the repo root from this script's own path. Zero shell spawn.
|
--- Resolve the repo root from this script's own path. Zero shell spawn.
|
||||||
--- `duffle_paths.lua` always lives at `<repo>/scripts/duffle_paths.lua`, so the repo root is the parent of the directory containing this script.
|
--- `duffle_paths.lua` always lives at `<repo>/scripts/duffle_paths.lua`, so the repo root is the
|
||||||
--- We derive it directly from `debug.getinfo(1, "S").source` (returns `@<path>` for the currently-running chunk).
|
--- parent of the directory containing this script. We derive it directly from `debug.getinfo(1, "S").source`
|
||||||
|
--- (returns `@<path>` for the currently-running chunk).
|
||||||
---
|
---
|
||||||
--- If `debug.getinfo` can't parse this script's path (shouldn't happen — dofile always populates source), return nil and let `M.setup()` fail loud.
|
--- If `debug.getinfo` can't parse this script's path (shouldn't happen — dofile always populates source),
|
||||||
|
--- return nil and let `M.setup()` fail loud.
|
||||||
--- @return string|nil
|
--- @return string|nil
|
||||||
local function find_repo_root()
|
local function find_repo_root()
|
||||||
if package.loaded[CACHE_KEY] then return package.loaded[CACHE_KEY] end
|
if package.loaded[CACHE_KEY] then return package.loaded[CACHE_KEY] end
|
||||||
@@ -45,17 +47,22 @@ local function find_repo_root()
|
|||||||
return root
|
return root
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Set `package.path` (for `require("duffle")` + `require("passes.X")`) and `package.cpath` (for `lpeg.dll`).
|
--- Set `package.path` (for `require("duffle")` + `require("passes.X")`) and
|
||||||
|
--- `package.cpath` (for `lpeg.dll`).
|
||||||
---
|
---
|
||||||
--- This script does NOT touch the OS environment: no `os.setenv`, no `os.putenv`, no `$PATH` mods.
|
--- This script does NOT touch the OS environment: no `os.setenv`, no `os.putenv`, no `$PATH` mods.
|
||||||
--- It just sets `package.path` and `package.cpath` (the standard Lua way to register module search dirs).
|
--- It just sets `package.path` and `package.cpath` (the standard Lua way to register module search dirs).
|
||||||
--- lpeg is built by `update_deps.ps1` to `toolchain/lpeg/`, which we wire into `package.cpath` here (so `require("lpeg")` from `duffle.lua` resolves without any global state).
|
--- lpeg is built by `update_deps.ps1` to `toolchain/lpeg/`,
|
||||||
|
--- which we wire into `package.cpath` here (so `require("lpeg")` from `duffle.lua` resolves without any global state).
|
||||||
function M.setup()
|
function M.setup()
|
||||||
local repo_root = find_repo_root()
|
local repo_root = find_repo_root()
|
||||||
if not repo_root then
|
if not repo_root then
|
||||||
-- Unreachable in practice: find_repo_root() derives the repo root from this script's own source path via debug.getinfo(1, "S").source (no subprocess, no git CLI, <1ms).
|
-- Unreachable in practice: find_repo_root() derives the repo root from this script's
|
||||||
-- A nil return means the source path did not match the expected <repo>/scripts/duffle_paths.lua layout — a packaging bug, not a "missing git repo" condition.
|
-- own source path via debug.getinfo(1, "S").source (no subprocess, no git CLI, <1ms).
|
||||||
-- os.exit(2) is retained so a real failure surfaces loud rather than silently producing an unconfigured module table.
|
-- A nil return means the source path did not match the expected
|
||||||
|
-- <repo>/scripts/duffle_paths.lua layout — a packaging bug, not a "missing git repo"
|
||||||
|
-- condition. os.exit(2) is retained so a real failure surfaces loud rather than
|
||||||
|
-- silently producing an unconfigured module table.
|
||||||
os.exit(2)
|
os.exit(2)
|
||||||
end
|
end
|
||||||
|
|
||||||
@@ -80,6 +87,6 @@ end
|
|||||||
-- Run the setup as a side effect.
|
-- Run the setup as a side effect.
|
||||||
M.setup()
|
M.setup()
|
||||||
|
|
||||||
-- Now that package.path includes scripts/, `require("duffle")` resolves.
|
-- Now that package.path includes scripts/, `require("duffle")` resolves. Return the duffle module
|
||||||
-- Return the duffle module so callers can do `local duffle = dofile(...duffle_paths.lua)` in one line.
|
-- so callers can do `local duffle = dofile(...duffle_paths.lua)` in one line.
|
||||||
return require("duffle")
|
return require("duffle")
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -1,418 +0,0 @@
|
|||||||
-- elf32.lua — Pure-Lua ELF32 format helpers with no lfs / no lpeg dependency.
|
|
||||||
-- The reload helper's `parse_manifest` (scripts/pcsx_debug_helper/reload.lua)
|
|
||||||
-- and the metaprogram's `read_elf_sections` + `read_nm` (scripts/elf_dwarf.lua)
|
|
||||||
-- both parsed ELF32 headers from wire bytes.
|
|
||||||
--
|
|
||||||
-- This module contains the format constants and the byte-level walker.
|
|
||||||
--- The metaprogram side keeps `read_u32_le` / `read_u16_le` as local forwarders; the helper side calls `E.*` directly.
|
|
||||||
--
|
|
||||||
-- **Adapter contract (explicit pass style):**
|
|
||||||
-- The helper VM's `Support.File` exposes byte-read methods that require `self` (fileffi.lua:225-227),
|
|
||||||
-- so callers wrap once in a 1-line adapter that strips `self`.
|
|
||||||
-- The parsers here operate on the unwrapped form.
|
|
||||||
-- Reads are flat function calls — `E.read_u8(adapter, off)`, `E.read_u32(adapter, off)`, `E.size(adapter)`.
|
|
||||||
-- read_u8(adapter, off) -> integer | nil
|
|
||||||
-- read_u16(adapter, off) -> integer | nil
|
|
||||||
-- read_u32(adapter, off) -> integer | nil
|
|
||||||
-- size(adapter) -> integer
|
|
||||||
--
|
|
||||||
-- **Convention:** every offset in the constants tables is a zero-based wire offset.
|
|
||||||
-- The `+ 1` conversion happens only at the `string.byte` boundary inside the readers.
|
|
||||||
--
|
|
||||||
-- spec: System V ABI gABI v1.2 §"ELF Header" (Table 1) + §"Section Header Table"
|
|
||||||
-- spec: System V ABI gABI v1.2 §"Symbol Table" (Elf32_Sym layout)
|
|
||||||
|
|
||||||
local M = {}
|
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
|
||||||
-- Little-endian readers (bit-weighted accumulator, math.floor only)
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
|
||||||
|
|
||||||
--- Read a 4-byte little-endian unsigned integer from `adapter` at zero-based wire offset `off`.
|
|
||||||
---
|
|
||||||
--- Bit weights are written as `0x100`, `0x10000`, `0x1000000` (i.e. 2^8, 2^16, 2^24) so the LE byte positions are visually explicit:
|
|
||||||
--- byte 0 contributes its value directly;
|
|
||||||
--- byte 1 is shifted left by 8; byte 2 by 16; byte 3 by 24.
|
|
||||||
---
|
|
||||||
--- math.floor (not LuaJIT's `>>`) keeps the body portable across LuaJIT 2.0/2.1 and plain Lua 5.x. `string.byte` receives `+ 1` at the boundary.
|
|
||||||
---
|
|
||||||
--- **Call form:** explicit-pass. The reader receives `adapter` as the first positional argument and the offset as the second; no `self` is passed.
|
|
||||||
--- Test fixtures declare `function(offset) ... end` and the parsers call them via dot syntax `adapter.read_u8_at(off)`.
|
|
||||||
--- The colon form `adapter:read_u8_at(off)` would prepend the adapter table as `offset` and break the contract.
|
|
||||||
--- @param adapter table
|
|
||||||
--- @param off integer -- zero-based wire offset
|
|
||||||
--- @return integer|nil
|
|
||||||
function M.read_u32(adapter, off)
|
|
||||||
return adapter.read_u8_at(off)
|
|
||||||
+ adapter.read_u8_at(off + 0x01) * 0x00000100
|
|
||||||
+ adapter.read_u8_at(off + 0x02) * 0x00010000
|
|
||||||
+ adapter.read_u8_at(off + 0x03) * 0x01000000
|
|
||||||
end
|
|
||||||
|
|
||||||
--- Read a 2-byte little-endian unsigned integer from `adapter` at zero-based wire offset `off`.
|
|
||||||
--- @param adapter table
|
|
||||||
--- @param off integer -- zero-based wire offset
|
|
||||||
--- @return integer|nil
|
|
||||||
function M.read_u16(adapter, off)
|
|
||||||
return adapter.read_u8_at(off)
|
|
||||||
+ adapter.read_u8_at(off + 0x01) * 0x00000100
|
|
||||||
end
|
|
||||||
|
|
||||||
--- Read a 1-byte unsigned integer from `adapter` at zero-based wire offset `off`.
|
|
||||||
--- @param adapter table
|
|
||||||
--- @param off integer -- zero-based wire offset
|
|
||||||
--- @return integer|nil
|
|
||||||
function M.read_u8(adapter, off)
|
|
||||||
return adapter.read_u8_at(off)
|
|
||||||
end
|
|
||||||
|
|
||||||
--- Total adapter byte length.
|
|
||||||
--- @param adapter table
|
|
||||||
--- @return integer
|
|
||||||
function M.size(adapter)
|
|
||||||
return adapter.read_size()
|
|
||||||
end
|
|
||||||
|
|
||||||
--- Forwarders kept for backward compat with scripts/elf_dwarf.lua.
|
|
||||||
--- The metaprogram side keeps `read_u32_le` / `read_u16_le`;
|
|
||||||
--- both layers now use the same byte-level helpers under the hood.
|
|
||||||
function M.read_u32_le(buf, off)
|
|
||||||
local byte_off = off + 1
|
|
||||||
return buf:byte(byte_off)
|
|
||||||
+ buf:byte(byte_off + 0x01) * 0x00000100
|
|
||||||
+ buf:byte(byte_off + 0x02) * 0x00010000
|
|
||||||
+ buf:byte(byte_off + 0x03) * 0x01000000
|
|
||||||
end
|
|
||||||
|
|
||||||
--- Read a 2-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
|
|
||||||
--- @param buf string
|
|
||||||
--- @param off integer -- zero-based wire offset
|
|
||||||
--- @return integer
|
|
||||||
function M.read_u16_le(buf, off)
|
|
||||||
local byte_off = off + 1
|
|
||||||
return buf:byte(byte_off) + buf:byte(byte_off + 0x01) * 0x00000100
|
|
||||||
end
|
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
|
||||||
-- Format constants
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
|
||||||
|
|
||||||
-- ELF format constants (System V ABI gABI v1.2).
|
|
||||||
M.ELFCLASS32 = 1 -- spec: gABI v1.2 §"ELF Header" — EI_CLASS byte
|
|
||||||
M.ELFDATA2LSB = 1 -- spec: gABI v1.2 §"ELF Header" — EI_DATA byte
|
|
||||||
M.EM_MIPS = 8 -- spec: gABI v1.2 §"Machine Information" — MIPS architecture
|
|
||||||
|
|
||||||
-- Section type constants (System V ABI gABI v1.2 §"Section Header Table").
|
|
||||||
M.SHT_SYMTAB = 2 -- spec: gABI v1.2 §"Section Types" — symbol table
|
|
||||||
M.SHT_STRTAB = 3 -- spec: gABI v1.2 §"Section Types" — string table
|
|
||||||
M.SHT_NOBITS = 8 -- spec: gABI v1.2 §"Section Types" — no space in file
|
|
||||||
|
|
||||||
-- Section flag constants (System V ABI gABI v1.2 §"Section Header Table").
|
|
||||||
M.SHF_WRITE = 0x1 -- spec: gABI v1.2 §"Section Attributes" — writable
|
|
||||||
M.SHF_ALLOC = 0x2 -- spec: gABI v1.2 §"Section Attributes" — occupies memory
|
|
||||||
M.SHF_EXECINSTR = 0x4 -- spec: gABI v1.2 §"Section Attributes" — executable
|
|
||||||
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
-- ELF32 header layout (System V ABI gABI v1.2 §"ELF Header" Table 1)
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
-- All offsets are zero-based wire offsets. The header is 52 bytes total (header_bytes = 0x34 = 52).
|
|
||||||
M.ELF32_HEADER = {
|
|
||||||
magic_offset = 0x00, -- 4 bytes; expected "\127ELF"
|
|
||||||
magic = "\127ELF",
|
|
||||||
class_offset = 0x04, -- 1 byte; 1 = ELF32, 2 = ELF64
|
|
||||||
endian_offset = 0x05, -- 1 byte; 1 = little-endian, 2 = big-endian
|
|
||||||
header_bytes = 0x34, -- ELF32 header is 52 bytes total
|
|
||||||
e_entry_offset = 0x18, -- 4-byte LE; entry-point virtual address
|
|
||||||
e_shoff_offset = 0x20, -- 4-byte LE; section-header table file offset
|
|
||||||
e_shentsize_offset = 0x2E, -- 2-byte LE; section-header entry size in bytes
|
|
||||||
e_shnum_offset = 0x30, -- 2-byte LE; number of section headers
|
|
||||||
e_shstrndx_offset = 0x32, -- 2-byte LE; index of section-name string table
|
|
||||||
}
|
|
||||||
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
-- ELF32 section-header layout (System V ABI gABI v1.2 §"Section Header Table")
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
-- Each entry is 40 bytes (sh_entsize_bytes = 0x28 = 40);
|
|
||||||
-- zero-based, field offsets relative to the start of the entry.
|
|
||||||
M.ELF32_SECTION = {
|
|
||||||
sh_name_offset = 0x00, -- 4-byte LE; offset into .shstrtab
|
|
||||||
sh_type_offset = 0x04, -- 4-byte LE; section type (SHT_*)
|
|
||||||
sh_flags_offset = 0x08, -- 4-byte LE; section flags (SHF_*)
|
|
||||||
sh_addr_offset = 0x0C, -- 4-byte LE; virtual address at execution
|
|
||||||
sh_offset_offset = 0x10, -- 4-byte LE; section's file offset
|
|
||||||
sh_size_offset = 0x14, -- 4-byte LE; section's size in bytes
|
|
||||||
sh_link_offset = 0x18, -- 4-byte LE; link to a related section
|
|
||||||
sh_entsize_bytes = 0x28, -- spec: gABI v1.2 §"Section Header Table" — 40 bytes per entry
|
|
||||||
}
|
|
||||||
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
-- ELF32 symbol-table entry layout (System V ABI gABI v1.2 §"Symbol Table")
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
-- Each entry is 16 bytes (sym_entry_bytes = 0x10 = 16);
|
|
||||||
-- zero-based, field offsets relative to the start of the entry.
|
|
||||||
M.ELF32_SYM = {
|
|
||||||
st_name = 0x00, -- 4-byte LE; offset into the linked string table
|
|
||||||
st_value = 0x04, -- 4-byte LE; symbol value (address / absolute)
|
|
||||||
st_size = 0x08, -- 4-byte LE; symbol size in bytes
|
|
||||||
st_info = 0x0C, -- 1 byte; binding (high nibble) + type (low nibble)
|
|
||||||
sym_entry_bytes = 0x10, -- spec: gABI v1.2 §"Symbol Table" — 16 bytes per entry
|
|
||||||
}
|
|
||||||
|
|
||||||
-- DWARF32 initial-length terminator (DWARF4 §7.4) — kept here so the metaprogram's elf_dwarf.lua can drop its own copy of the same constant.
|
|
||||||
M.dw_dwarf32_terminator = 0xFFFFFFFF
|
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
|
||||||
-- Adapter validation
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
|
||||||
|
|
||||||
--- Validate that `adapter` exposes the byte-read surface.
|
|
||||||
--- Returns true on success, false + a stable error code on failure.
|
|
||||||
--- The helper side calls this before parse_manifest to reject callers before any byte is read.
|
|
||||||
--- @param adapter any
|
|
||||||
--- @return boolean, string|nil
|
|
||||||
function M.validate_adapter(adapter)
|
|
||||||
if type(adapter) ~= "table" then return false, "bad_file_adapter" end
|
|
||||||
if type(adapter.read_u8_at) ~= "function" then return false, "bad_file_adapter" end
|
|
||||||
if type(adapter.read_u16_at) ~= "function" then return false, "bad_file_adapter" end
|
|
||||||
if type(adapter.read_u32_at) ~= "function" then return false, "bad_file_adapter" end
|
|
||||||
if type(adapter.read_size) ~= "function" then return false, "bad_file_adapter" end
|
|
||||||
return true, nil
|
|
||||||
end
|
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
|
||||||
-- String-table reader
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
|
||||||
|
|
||||||
--- Extract a NUL-terminated C string from `strtab` at zero-based offset `off`.
|
|
||||||
--- Returns nil if `off` is out of range or the string is not NUL-terminated.
|
|
||||||
--- @param strtab string
|
|
||||||
--- @param off integer
|
|
||||||
--- @return string|nil
|
|
||||||
function M.get_str(strtab, off)
|
|
||||||
if off < 0 or off >= #strtab then return nil end
|
|
||||||
local end_pos = strtab:find("\0", off + 1, true)
|
|
||||||
if not end_pos then return nil end
|
|
||||||
return strtab:sub(off + 1, end_pos - 1)
|
|
||||||
end
|
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
|
||||||
-- Header / section / symbol walkers
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
|
||||||
|
|
||||||
--- Read the ELF32 header through `adapter` and validate the magic, class, and data encoding.
|
|
||||||
--- Returns a table on success:
|
|
||||||
--- { e_entry, e_shoff, e_shentsize, e_shnum, e_shstrndx, error = nil }
|
|
||||||
--- On failure returns nil + a stable error code:
|
|
||||||
--- bad_magic, unsupported_elf_class, unsupported_elf_data, truncated_header
|
|
||||||
--- The header's machine field is NOT validated here — callers (e.g. the helper's prime path) decide whether to require EM_MIPS before symbol reads.
|
|
||||||
--- @param adapter table
|
|
||||||
--- @return table|nil, string|nil
|
|
||||||
function M.parse_elf32_headers(adapter)
|
|
||||||
local ok, err = M.validate_adapter(adapter)
|
|
||||||
if not ok then return nil, err end
|
|
||||||
|
|
||||||
-- 4-byte magic: 0x7F 'E' 'L' 'F'.
|
|
||||||
-- The byte readers take the adapter explicitly.
|
|
||||||
-- The production `Support.File` adapter is wrapped by the caller to drop its implicit `self` so the parser shape is flat pass-style.
|
|
||||||
local b1 = M.read_u8(adapter, 0)
|
|
||||||
local b2 = M.read_u8(adapter, 1)
|
|
||||||
local b3 = M.read_u8(adapter, 2)
|
|
||||||
local b4 = M.read_u8(adapter, 3)
|
|
||||||
if not (b1 and b2 and b3 and b4)
|
|
||||||
or not (b1 == 0x7f and b2 == 0x45 and b3 == 0x4c and b4 == 0x46) then
|
|
||||||
return nil, "bad_magic"
|
|
||||||
end
|
|
||||||
|
|
||||||
local class = M.read_u8(adapter, M.ELF32_HEADER.class_offset)
|
|
||||||
if class ~= M.ELFCLASS32 then
|
|
||||||
return nil, "unsupported_elf_class"
|
|
||||||
end
|
|
||||||
|
|
||||||
local data = M.read_u8(adapter, M.ELF32_HEADER.endian_offset)
|
|
||||||
if data ~= M.ELFDATA2LSB then
|
|
||||||
return nil, "unsupported_elf_data"
|
|
||||||
end
|
|
||||||
|
|
||||||
local e_entry = M.read_u32(adapter, M.ELF32_HEADER.e_entry_offset)
|
|
||||||
local e_shoff = M.read_u32(adapter, M.ELF32_HEADER.e_shoff_offset)
|
|
||||||
local e_shentsize = M.read_u16(adapter, M.ELF32_HEADER.e_shentsize_offset)
|
|
||||||
local e_shnum = M.read_u16(adapter, M.ELF32_HEADER.e_shnum_offset)
|
|
||||||
local e_shstrndx = M.read_u16(adapter, M.ELF32_HEADER.e_shstrndx_offset)
|
|
||||||
if not (e_entry and e_shoff and e_shentsize and e_shnum and e_shstrndx) then
|
|
||||||
return nil, "truncated_header"
|
|
||||||
end
|
|
||||||
|
|
||||||
return {
|
|
||||||
e_entry = e_entry,
|
|
||||||
e_shoff = e_shoff,
|
|
||||||
e_shentsize = e_shentsize,
|
|
||||||
e_shnum = e_shnum,
|
|
||||||
e_shstrndx = e_shstrndx,
|
|
||||||
error = nil,
|
|
||||||
}
|
|
||||||
end
|
|
||||||
|
|
||||||
--- Read one section-header entry from `adapter` at `sh_off`.
|
|
||||||
--- Returns a table with the wire fields plus a (yet-unresolved) `name` field.
|
|
||||||
--- @param adapter table
|
|
||||||
--- @param sh_off integer
|
|
||||||
--- @return table|nil, string|nil -- entry, error
|
|
||||||
local function read_section_entry(adapter, sh_off)
|
|
||||||
local entry = {
|
|
||||||
sh_name = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_name_offset),
|
|
||||||
sh_type = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_type_offset),
|
|
||||||
sh_flags = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_flags_offset),
|
|
||||||
sh_addr = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_addr_offset),
|
|
||||||
sh_offset = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_offset_offset),
|
|
||||||
sh_size = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_size_offset),
|
|
||||||
sh_link = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_link_offset),
|
|
||||||
name = "",
|
|
||||||
}
|
|
||||||
if not (entry.sh_name and entry.sh_type and entry.sh_flags and entry.sh_addr
|
|
||||||
and entry.sh_offset and entry.sh_size and entry.sh_link) then
|
|
||||||
return nil, "truncated_section_headers"
|
|
||||||
end
|
|
||||||
return entry, nil
|
|
||||||
end
|
|
||||||
|
|
||||||
--- Walk every section header in `hdr` and return a 1-based array of entries
|
|
||||||
--- (the section at logical index 0 is at array position 1, etc.).
|
|
||||||
--- Each entry has the wire fields plus a resolved `name` derived from `.shstrtab`.
|
|
||||||
--- Returns nil + a stable error code on failure: truncated_section_headers, missing_shstrtab, truncated_strtab
|
|
||||||
--- @param adapter table
|
|
||||||
--- @param hdr table -- the table returned by parse_elf32_headers
|
|
||||||
--- @return table|nil, string|nil
|
|
||||||
function M.walk_sections(adapter, hdr)
|
|
||||||
if not hdr or hdr.error then return nil, hdr and hdr.error or "truncated_section_headers" end
|
|
||||||
|
|
||||||
local file_size = M.size(adapter)
|
|
||||||
if hdr.e_shoff + hdr.e_shnum * hdr.e_shentsize > file_size then
|
|
||||||
return nil, "truncated_section_headers"
|
|
||||||
end
|
|
||||||
|
|
||||||
-- Read every section header first; we need .shstrtab to resolve names.
|
|
||||||
local sections = {}
|
|
||||||
for i = 0, hdr.e_shnum - 1 do
|
|
||||||
local sh_off = hdr.e_shoff + i * hdr.e_shentsize
|
|
||||||
local entry, err = read_section_entry(adapter, sh_off)
|
|
||||||
if not entry then return nil, err end
|
|
||||||
sections[i + 1] = entry
|
|
||||||
end
|
|
||||||
|
|
||||||
if hdr.e_shstrndx >= hdr.e_shnum then
|
|
||||||
return nil, "missing_shstrtab"
|
|
||||||
end
|
|
||||||
|
|
||||||
local shstrtab = sections[hdr.e_shstrndx + 1]
|
|
||||||
if not shstrtab or shstrtab.sh_type ~= M.SHT_STRTAB then
|
|
||||||
return nil, "missing_shstrtab"
|
|
||||||
end
|
|
||||||
if shstrtab.sh_offset + shstrtab.sh_size > file_size then
|
|
||||||
return nil, "truncated_section_headers"
|
|
||||||
end
|
|
||||||
local shstrtab_bytes = M.read_section_bytes(adapter, shstrtab)
|
|
||||||
if not shstrtab_bytes then return nil, "truncated_section_headers" end
|
|
||||||
|
|
||||||
for _, s in ipairs(sections) do
|
|
||||||
s.name = M.get_str(shstrtab_bytes, s.sh_name) or ""
|
|
||||||
end
|
|
||||||
|
|
||||||
return sections, nil
|
|
||||||
end
|
|
||||||
|
|
||||||
--- Read the bytes of one section. Returns a string, or nil if the adapter returns nil for any byte (out-of-bounds).
|
|
||||||
--- The caller is responsible fors sizing the buffer (the section's sh_offset + sh_size must fit in adapter.size).
|
|
||||||
--- @param adapter table
|
|
||||||
--- @param section table -- one entry from walk_sections
|
|
||||||
--- @return string|nil
|
|
||||||
function M.read_section_bytes(adapter, section)
|
|
||||||
local size = section.sh_size
|
|
||||||
if size == 0 then return "" end
|
|
||||||
local out = {}
|
|
||||||
for i = 0, size - 1 do
|
|
||||||
local b = M.read_u8(adapter, section.sh_offset + i)
|
|
||||||
if b == nil then return nil end
|
|
||||||
out[#out + 1] = string.char(b)
|
|
||||||
end
|
|
||||||
return table.concat(out)
|
|
||||||
end
|
|
||||||
|
|
||||||
--- Convenience: walk sections, then look up the named section, then read its bytes.
|
|
||||||
--- Returns nil + a stable error code if the section is absent or out-of-bounds.
|
|
||||||
--- @param adapter table
|
|
||||||
--- @param sections table -- 1-based array from walk_sections
|
|
||||||
--- @param name string
|
|
||||||
--- @return string|nil, string|nil
|
|
||||||
function M.read_named_section(adapter, sections, name)
|
|
||||||
if not sections then return nil, "missing_section" end
|
|
||||||
for _, s in ipairs(sections) do
|
|
||||||
if s.name == name then
|
|
||||||
local bytes = M.read_section_bytes(adapter, s)
|
|
||||||
if not bytes then return nil, "truncated_section_data" end
|
|
||||||
return bytes, nil
|
|
||||||
end
|
|
||||||
end
|
|
||||||
return nil, "missing_section"
|
|
||||||
end
|
|
||||||
|
|
||||||
--- Walk every SHT_SYMTAB section in `sections` and accumulate symbols by name.
|
|
||||||
--- Each stored entry is `{ value = st_value, size = st_size, info = st_info, shndx = st_shndx }`.
|
|
||||||
--- Both STB_LOCAL and STB_GLOBAL symbols are included; the live ELF stores `smem` as a local symbol.
|
|
||||||
--- Returns nil + a stable error code on failure: missing_symtab_strtab, truncated_section_headers
|
|
||||||
--- @param adapter table
|
|
||||||
--- @param sections table
|
|
||||||
--- @return table|nil, string|nil
|
|
||||||
function M.collect_symbols(adapter, sections)
|
|
||||||
if not sections then return nil, "missing_sections" end
|
|
||||||
local symbols = {}
|
|
||||||
local file_size = M.size(adapter)
|
|
||||||
for _, s in ipairs(sections) do
|
|
||||||
if s.sh_type == M.SHT_SYMTAB then
|
|
||||||
local strtab = sections[s.sh_link + 1]
|
|
||||||
if not strtab or strtab.sh_type ~= M.SHT_STRTAB then
|
|
||||||
return nil, "missing_symtab_strtab"
|
|
||||||
end
|
|
||||||
if strtab.sh_offset + strtab.sh_size > file_size then
|
|
||||||
return nil, "truncated_section_headers"
|
|
||||||
end
|
|
||||||
local strtab_bytes = M.read_section_bytes(adapter, strtab)
|
|
||||||
if not strtab_bytes then return nil, "truncated_section_headers" end
|
|
||||||
if s.sh_offset + s.sh_size > file_size then
|
|
||||||
return nil, "truncated_section_headers"
|
|
||||||
end
|
|
||||||
local symtab_bytes = M.read_section_bytes(adapter, s)
|
|
||||||
if not symtab_bytes then return nil, "truncated_section_headers" end
|
|
||||||
local n = #symtab_bytes / M.ELF32_SYM.sym_entry_bytes
|
|
||||||
for j = 0, n - 1 do
|
|
||||||
local e = s.sh_offset + j * M.ELF32_SYM.sym_entry_bytes
|
|
||||||
local st_name = M.read_u32(adapter, e + M.ELF32_SYM.st_name)
|
|
||||||
if st_name then
|
|
||||||
local st_value = M.read_u32(adapter, e + M.ELF32_SYM.st_value)
|
|
||||||
local st_size = M.read_u32(adapter, e + M.ELF32_SYM.st_size)
|
|
||||||
local st_info = M.read_u8(adapter, e + M.ELF32_SYM.st_info)
|
|
||||||
-- st_shndx is at offset 14 (2 bytes) — derived from the layout
|
|
||||||
-- the metaprogram reads too. Inline the read to keep the
|
|
||||||
-- adapter as the only I/O surface.
|
|
||||||
local b1 = M.read_u8(adapter, e + 14)
|
|
||||||
local b2 = M.read_u8(adapter, e + 15)
|
|
||||||
if not (b1 and b2) then
|
|
||||||
return nil, "truncated_section_headers"
|
|
||||||
end
|
|
||||||
local st_shndx = b1 + b2 * 0x100
|
|
||||||
local name = M.get_str(strtab_bytes, st_name) or ""
|
|
||||||
if name ~= "" then
|
|
||||||
symbols[name] = {
|
|
||||||
value = st_value,
|
|
||||||
size = st_size,
|
|
||||||
info = st_info,
|
|
||||||
shndx = st_shndx,
|
|
||||||
}
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
return symbols, nil
|
|
||||||
end
|
|
||||||
|
|
||||||
return M
|
|
||||||
+161
-173
@@ -11,11 +11,6 @@
|
|||||||
-- lfs is wired into package.cpath by `duffle_paths.lua` (vendored under `toolchain/lfs/lfs.dll`).
|
-- lfs is wired into package.cpath by `duffle_paths.lua` (vendored under `toolchain/lfs/lfs.dll`).
|
||||||
local lfs = require("lfs")
|
local lfs = require("lfs")
|
||||||
|
|
||||||
-- scripts/elf32.lua contains format-constant tables + the byte-level walker.
|
|
||||||
-- The this file re-exports `read_u32_le` / `read_u16_le` (and the DWARF32 terminator).
|
|
||||||
-- TODO(Ed): Remove re-export.
|
|
||||||
local E = require("elf32")
|
|
||||||
|
|
||||||
local M = {}
|
local M = {}
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
@@ -107,13 +102,27 @@ M.MIPS_BYTES_PER_WORD = 0x04
|
|||||||
--- **Wire-offset contract:** format offsets, fixed-width reader offsets, LEB/parser cursors, and section-relative values are zero-based wire offsets.
|
--- **Wire-offset contract:** format offsets, fixed-width reader offsets, LEB/parser cursors, and section-relative values are zero-based wire offsets.
|
||||||
--- Only Lua string APIs receive a `+ 1` conversion at their boundary (`byte`, `sub`, and `find`).
|
--- Only Lua string APIs receive a `+ 1` conversion at their boundary (`byte`, `sub`, and `find`).
|
||||||
--- ELF/DWARF field offsets are expressed in hex so they map directly to the zero-based byte positions in the binary file.
|
--- ELF/DWARF field offsets are expressed in hex so they map directly to the zero-based byte positions in the binary file.
|
||||||
---
|
|
||||||
--- The ELF32 header / section / sym layout tables are within scripts/elf32.lua.
|
|
||||||
--- The metaprogram re-exports the DWARF32 initial-length terminator.
|
|
||||||
|
|
||||||
--- spec: DWARF4 spec §7.4 — 32-bit DWARF initial-length terminator
|
--- spec: System V ABI gABI v1.2 §"ELF Header" (Table 1) + §"Section Header Table"
|
||||||
M.dw_dwarf32_terminator = E.dw_dwarf32_terminator
|
M.ELF32 = {
|
||||||
-- TODO(Ed): Remove re-export.
|
magic_offset = 0x00, -- 4-byte magic "\127ELF" at file offset 0x00
|
||||||
|
magic = "\127ELF",
|
||||||
|
class_offset = 0x04, -- 1-byte; 1 = ELF32, 2 = ELF64
|
||||||
|
class_elf32 = 1,
|
||||||
|
endian_offset = 0x05, -- 1-byte; 1 = little-endian, 2 = big-endian
|
||||||
|
endian_little = 1,
|
||||||
|
header_bytes = 0x34, -- spec: gABI v1.2 §"ELF Header" — ELF32 header is 52 bytes total
|
||||||
|
e_shoff_offset = 0x20, -- 4-byte LE; section-header table file offset
|
||||||
|
e_shentsize_offset = 0x2E, -- 2-byte LE; section-header entry size in bytes
|
||||||
|
e_shnum_offset = 0x30, -- 2-byte LE; number of section headers
|
||||||
|
e_shstrndx_offset = 0x32, -- 2-byte LE; index of section-name string table
|
||||||
|
sh_size_bytes = 0x28, -- spec: gABI v1.2 §"Section Header Table" — each entry is 40 bytes
|
||||||
|
sh_name_offset = 0x00, -- 4-byte LE; offset into .shstrtab
|
||||||
|
sh_type_offset = 0x04, -- 4-byte LE; section type (SHT_*)
|
||||||
|
sh_offset_offset = 0x10, -- 4-byte LE; section's file offset
|
||||||
|
sh_size_offset = 0x14, -- 4-byte LE; section's size in bytes
|
||||||
|
dw_dwarf32_terminator = 0xFFFFFFFF, -- spec: DWARF4 spec §7.4 — 32-bit DWARF initial-length terminator
|
||||||
|
}
|
||||||
|
|
||||||
-- ----------------------------------------------------------------------------
|
-- ----------------------------------------------------------------------------
|
||||||
-- DWARF4 .debug_aranges (per DWARF5 spec §7.4 — Address Range Table)
|
-- DWARF4 .debug_aranges (per DWARF5 spec §7.4 — Address Range Table)
|
||||||
@@ -232,24 +241,27 @@ M.DWARF5_DEBUG_LINE = {
|
|||||||
--- (which has partial `string.unpack` coverage).
|
--- (which has partial `string.unpack` coverage).
|
||||||
--- **Convention:** `off` is a zero-based wire offset; `+ 1` is applied only at the `string.byte` boundary.
|
--- **Convention:** `off` is a zero-based wire offset; `+ 1` is applied only at the `string.byte` boundary.
|
||||||
---
|
---
|
||||||
--- Thin forwarder: the canonical implementation lives in scripts/elf32.lua.
|
--- **Byte weights** are written as `0x100`, `0x10000`, `0x1000000` (i.e. 2^8, 2^16, 2^24) so the LE byte positions are visually explicit:
|
||||||
--- The "second caller lifts" pattern keeps the metaprogram side fluent
|
--- byte 0 contributes its value directly; byte 1 is shifted left by 8 (= 0x100); byte 2 by 16 (= 0x10000); byte 3 by 24 (= 0x1000000).
|
||||||
--- (`M.read_u32_le(buf, off)`) while the body is deduped.
|
|
||||||
--- @param buf string
|
--- @param buf string
|
||||||
--- @param off integer -- zero-based wire offset
|
--- @param off integer -- zero-based wire offset
|
||||||
--- @return integer
|
--- @return integer
|
||||||
function M.read_u32_le(buf, off)
|
function M.read_u32_le(buf, off)
|
||||||
return E.read_u32_le(buf, off)
|
local byte_off = off + 1
|
||||||
|
return buf:byte(byte_off)
|
||||||
|
+ buf:byte(byte_off + 0x01) * 0x00000100
|
||||||
|
+ buf:byte(byte_off + 0x02) * 0x00010000
|
||||||
|
+ buf:byte(byte_off + 0x03) * 0x01000000
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Read a 2-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
|
--- Read a 2-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
|
||||||
--- (`off` is zero-based; `+ 1` is applied only at the `string.byte` boundary.)
|
--- (`off` is zero-based; `+ 1` is applied only at the `string.byte` boundary.)
|
||||||
--- Thin forwarder — see `M.read_u32_le` for the rationale.
|
|
||||||
--- @param buf string
|
--- @param buf string
|
||||||
--- @param off integer -- zero-based wire offset
|
--- @param off integer -- zero-based wire offset
|
||||||
--- @return integer
|
--- @return integer
|
||||||
function M.read_u16_le(buf, off)
|
function M.read_u16_le(buf, off)
|
||||||
return E.read_u16_le(buf, off)
|
local byte_off = off + 1
|
||||||
|
return buf:byte(byte_off) + buf:byte(byte_off + 0x01) * 0x00000100
|
||||||
end
|
end
|
||||||
|
|
||||||
-- Pure-Lua 5.3 LEB128 readers (no `bit` library). `2^shift` arithmetic matches the existing parser.
|
-- Pure-Lua 5.3 LEB128 readers (no `bit` library). `2^shift` arithmetic matches the existing parser.
|
||||||
@@ -422,28 +434,28 @@ end
|
|||||||
--- should use this directly rather than going through `read_form_value`,
|
--- should use this directly rather than going through `read_form_value`,
|
||||||
--- which only exposes the low 4 bytes to preserve its existing (value, next_pos) return shape.
|
--- which only exposes the low 4 bytes to preserve its existing (value, next_pos) return shape.
|
||||||
--- @param buf string
|
--- @param buf string
|
||||||
--- @param pos integer -- zero-based wire offset
|
--- @param pos integer -- zero-based wire offset
|
||||||
--- @return integer -- low 4 bytes (LE), the type signature
|
--- @return integer -- low 4 bytes (LE), the type signature
|
||||||
--- @return integer -- high 4 bytes (LE), the offset within the matching type unit
|
--- @return integer -- high 4 bytes (LE), the offset within the matching type unit
|
||||||
--- @return integer -- cursor after the 8-byte value
|
--- @return integer -- cursor after the 8-byte value
|
||||||
function M.read_ref_sig8(buf, pos)
|
function M.read_ref_sig8(buf, pos)
|
||||||
return M.read_u32_le(buf, pos), M.read_u32_le(buf, pos + 4), pos + 8
|
return M.read_u32_le(buf, pos), M.read_u32_le(buf, pos + 4), pos + 8
|
||||||
end
|
end
|
||||||
|
|
||||||
--- DWARF5 §7.5.6 (Type Entries).
|
-- DWARF5 §7.5.6 (Type Entries).
|
||||||
--- Walk all units in `info` and return the 0-based offset of the first unit whose `DW_AT_type_signature`
|
-- Walk all units in `info` and return the 0-based offset of the first unit
|
||||||
--- (8-byte value at the end of the unit header) equals `target_sig`.
|
-- whose `DW_AT_type_signature` (8-byte value at the end of the unit header) equals `target_sig`.
|
||||||
--- The signature is interpreted as two 32-bit halves (low/high) per the read_ref_sig8 contract;
|
-- The signature is interpreted as two 32-bit halves (low/high) per the read_ref_sig8 contract;
|
||||||
--- we match both halves (i.e. the 8-byte value as a whole). Returns nil if no matching unit exists.
|
-- we match both halves (i.e. the 8-byte value as a whole). Returns nil if no matching unit exists.
|
||||||
---
|
--
|
||||||
--- Unit header layout (from pos 0):
|
-- Unit header layout (from pos 0):
|
||||||
--- unit_length(4) + version(2) + unit_type(1) + address_size(1) + debug_abbrev_offset(4)
|
-- unit_length(4) + version(2) + unit_type(1) + address_size(1) + debug_abbrev_offset(4)
|
||||||
--- followed by type_unit_specific fields: type_signature(8) + type_offset(4)
|
-- followed by type_unit_specific fields: type_signature(8) + type_offset(4)
|
||||||
--- The type_signature is at byte offset 8 of the body (right after debug_abbrev_offset).
|
-- The type_signature is at byte offset 8 of the body (right after debug_abbrev_offset).
|
||||||
--- @param info string -- the .debug_info section bytes
|
-- @param info string -- the .debug_info section bytes
|
||||||
--- @param target_sig_lo integer -- low 4 bytes (LE) of the desired signature
|
-- @param target_sig_lo integer -- low 4 bytes (LE) of the desired signature
|
||||||
--- @param target_sig_hi integer -- high 4 bytes (LE) of the desired signature
|
-- @param target_sig_hi integer -- high 4 bytes (LE) of the desired signature
|
||||||
--- @return integer|nil, integer|nil -- unit offset, type_offset within the unit
|
-- @return integer|nil, integer|nil -- unit offset, type_offset within the unit
|
||||||
function M.find_type_unit_by_signature(info, target_sig_lo, target_sig_hi)
|
function M.find_type_unit_by_signature(info, target_sig_lo, target_sig_hi)
|
||||||
local pos = 0
|
local pos = 0
|
||||||
local section_len = #info
|
local section_len = #info
|
||||||
@@ -459,23 +471,24 @@ function M.find_type_unit_by_signature(info, target_sig_lo, target_sig_hi)
|
|||||||
return nil, nil -- malformed
|
return nil, nil -- malformed
|
||||||
end
|
end
|
||||||
-- Per DWARF5 §7.5.6, the type_unit (DW_UT_type = 0x02) body layout is:
|
-- Per DWARF5 §7.5.6, the type_unit (DW_UT_type = 0x02) body layout is:
|
||||||
-- 0: version (2)
|
-- 0: version (2)
|
||||||
-- 2: unit_type (1) -- DW_UT_type = 0x02
|
-- 2: unit_type (1) -- DW_UT_type = 0x02
|
||||||
-- 3: address_size (1)
|
-- 3: address_size (1)
|
||||||
-- 4: debug_abbrev_offset (4)
|
-- 4: debug_abbrev_offset (4)
|
||||||
-- 8: type_signature (8)
|
-- 8: type_signature (8)
|
||||||
-- 16: type_offset (4)
|
-- 16: type_offset (4)
|
||||||
-- 20: <children>
|
-- 20: <children>
|
||||||
if body_end - body_start >= 20 then
|
if body_end - body_start >= 20 then
|
||||||
-- read_ref_sig8 / write_u32_le / etc. are 1-indexed (string:byte);
|
-- read_ref_sig8 / write_u32_le / etc. are 1-indexed (string:byte);
|
||||||
-- pos / body_start / body_end are 0-based wire offsets, so the 1-indexed byte at 0-based wire offset X is string:byte(X + 1).
|
-- pos / body_start / body_end are 0-based wire offsets, so the
|
||||||
|
-- 1-indexed byte at 0-based wire offset X is string:byte(X + 1).
|
||||||
-- Per DWARF5 §7.5.6, the type_unit body is laid out as:
|
-- Per DWARF5 §7.5.6, the type_unit body is laid out as:
|
||||||
-- byte 0-1: version (2)
|
-- byte 0-1: version (2)
|
||||||
-- byte 2: unit_type (1) -- DW_UT_type = 0x02
|
-- byte 2: unit_type (1) -- DW_UT_type = 0x02
|
||||||
-- byte 3: address_size (1)
|
-- byte 3: address_size (1)
|
||||||
-- byte 4-7: debug_abbrev_offset (4)
|
-- byte 4-7: debug_abbrev_offset (4)
|
||||||
-- byte 8-15: type_signature (8)
|
-- byte 8-15: type_signature (8)
|
||||||
-- byte 16-19: type_offset (4)
|
-- byte 16-19: type_offset (4)
|
||||||
local unit_type = info:byte(body_start + 2 + 1) -- 0-based +2 = unit_type in 1-indexed
|
local unit_type = info:byte(body_start + 2 + 1) -- 0-based +2 = unit_type in 1-indexed
|
||||||
if unit_type == 0x02 then -- DW_UT_type
|
if unit_type == 0x02 then -- DW_UT_type
|
||||||
local sig_lo, sig_hi, _ = M.read_ref_sig8(info, body_start + 8) -- 0-based +8 = type_signature in 1-indexed
|
local sig_lo, sig_hi, _ = M.read_ref_sig8(info, body_start + 8) -- 0-based +8 = type_signature in 1-indexed
|
||||||
@@ -526,7 +539,7 @@ end
|
|||||||
--- (we walk all `e_shnum` headers regardless of how many names are requested, to find the .shstrtab first).
|
--- (we walk all `e_shnum` headers regardless of how many names are requested, to find the .shstrtab first).
|
||||||
--- For frequent callers, pass the union of all needed sections in one call.
|
--- For frequent callers, pass the union of all needed sections in one call.
|
||||||
-- Can add `.debug_info` + `.debug_loc` + `.debug_str_offsets` to the list without writing a 2nd ELF walker.
|
-- Can add `.debug_info` + `.debug_loc` + `.debug_str_offsets` to the list without writing a 2nd ELF walker.
|
||||||
--- @param elf_path Path
|
--- @param elf_path Path
|
||||||
--- @param section_names string[] -- list of section names to read
|
--- @param section_names string[] -- list of section names to read
|
||||||
--- @return table<string, string>
|
--- @return table<string, string>
|
||||||
function M.read_elf_sections(elf_path, section_names)
|
function M.read_elf_sections(elf_path, section_names)
|
||||||
@@ -551,58 +564,69 @@ function M.read_elf_sections(elf_path, section_names)
|
|||||||
return result
|
return result
|
||||||
end
|
end
|
||||||
|
|
||||||
local file_size
|
-- Read the ELF32 header.
|
||||||
do
|
local header = f:read(M.ELF32.header_bytes)
|
||||||
f:seek("end", 0)
|
if not header or #header < M.ELF32.header_bytes then
|
||||||
file_size = f:seek("cur", 0)
|
io.stderr:write("[elf_dwarf.read_elf_sections] ELF too small for ELF32 header\n")
|
||||||
end
|
|
||||||
local adapter = {
|
|
||||||
read_u8_at = function(offset)
|
|
||||||
f:seek("set", offset)
|
|
||||||
local b = f:read(1)
|
|
||||||
if not b then return nil end
|
|
||||||
return b:byte()
|
|
||||||
end,
|
|
||||||
read_u16_at = function(offset)
|
|
||||||
f:seek("set", offset)
|
|
||||||
local b1 = f:read(1)
|
|
||||||
local b2 = f:read(1)
|
|
||||||
if not b1 or not b2 then return nil end
|
|
||||||
return b1:byte() + b2:byte() * 0x100
|
|
||||||
end,
|
|
||||||
read_u32_at = function(offset)
|
|
||||||
f:seek("set", offset)
|
|
||||||
local b1 = f:read(1)
|
|
||||||
local b2 = f:read(1)
|
|
||||||
local b3 = f:read(1)
|
|
||||||
local b4 = f:read(1)
|
|
||||||
if not b1 or not b2 or not b3 or not b4 then return nil end
|
|
||||||
return b1:byte() + b2:byte() * 0x100
|
|
||||||
+ b3:byte() * 0x10000 + b4:byte() * 0x1000000
|
|
||||||
end,
|
|
||||||
read_size = function() return file_size end,
|
|
||||||
}
|
|
||||||
|
|
||||||
-- Delegate the header parse + section walk to E.*.
|
|
||||||
local hdr, hdr_err = E.parse_elf32_headers(adapter)
|
|
||||||
if not hdr then
|
|
||||||
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] header parse failed: %s\n", tostring(hdr_err)))
|
|
||||||
f:close()
|
f:close()
|
||||||
return result
|
return result
|
||||||
end
|
end
|
||||||
|
|
||||||
local sections, walk_err = E.walk_sections(adapter, hdr)
|
-- Sanity-check magic + class + endianness.
|
||||||
if not sections then
|
if header:sub(M.ELF32.magic_offset + 1, M.ELF32.magic_offset + 0x04) ~= M.ELF32.magic then
|
||||||
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] section walk failed: %s\n", tostring(walk_err)))
|
io.stderr:write("[elf_dwarf.read_elf_sections] not an ELF file\n")
|
||||||
|
f:close()
|
||||||
|
return result
|
||||||
|
end
|
||||||
|
if header:byte(M.ELF32.class_offset + 1) ~= M.ELF32.class_elf32 then
|
||||||
|
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] not ELF32 (class=%d)\n", header:byte(M.ELF32.class_offset + 1)))
|
||||||
|
f:close()
|
||||||
|
return result
|
||||||
|
end
|
||||||
|
if header:byte(M.ELF32.endian_offset + 1) ~= M.ELF32.endian_little then
|
||||||
|
io.stderr:write("[elf_dwarf.read_elf_sections] not little-endian; unsupported\n")
|
||||||
f:close()
|
f:close()
|
||||||
return result
|
return result
|
||||||
end
|
end
|
||||||
|
|
||||||
-- Resolve the requested sections.
|
-- Parse section-header table location + dimensions from the header.
|
||||||
for _, s in ipairs(sections) do
|
local e_shoff = M.read_u32_le(header, M.ELF32.e_shoff_offset)
|
||||||
if wanted[s.name] then
|
local e_shentsize = M.read_u16_le(header, M.ELF32.e_shentsize_offset)
|
||||||
local bytes = E.read_section_bytes(adapter, s)
|
local e_shnum = M.read_u16_le(header, M.ELF32.e_shnum_offset)
|
||||||
if bytes then result[s.name] = bytes end
|
local e_shstrndx = M.read_u16_le(header, M.ELF32.e_shstrndx_offset)
|
||||||
|
|
||||||
|
-- Read the section-header string table (.shstrtab) so we can resolve section names from their `sh_name` offsets.
|
||||||
|
f:seek("set", e_shoff + e_shstrndx * e_shentsize)
|
||||||
|
local strtab_hdr = f:read(e_shentsize)
|
||||||
|
if not strtab_hdr or #strtab_hdr < e_shentsize then
|
||||||
|
io.stderr:write("[elf_dwarf.read_elf_sections] could not read .shstrtab header\n")
|
||||||
|
f:close()
|
||||||
|
return result
|
||||||
|
end
|
||||||
|
local strtab_offset = M.read_u32_le(strtab_hdr, M.ELF32.sh_offset_offset)
|
||||||
|
local strtab_size = M.read_u32_le(strtab_hdr, M.ELF32.sh_size_offset)
|
||||||
|
f:seek("set", strtab_offset)
|
||||||
|
local strtab = f:read(strtab_size) or ""
|
||||||
|
|
||||||
|
-- Walk all section headers; collect (offset, size) for the wanted names.
|
||||||
|
local function read_section_bytes(sh_offset, sh_size)
|
||||||
|
f:seek("set", sh_offset)
|
||||||
|
return f:read(sh_size) or ""
|
||||||
|
end
|
||||||
|
|
||||||
|
for sh_idx = 0, e_shnum - 1 do
|
||||||
|
f:seek("set", e_shoff + sh_idx * e_shentsize)
|
||||||
|
local sh = f:read(e_shentsize)
|
||||||
|
if not sh or #sh < e_shentsize then break end
|
||||||
|
local sh_name = M.read_u32_le(sh, M.ELF32.sh_name_offset)
|
||||||
|
local sh_offset = M.read_u32_le(sh, M.ELF32.sh_offset_offset)
|
||||||
|
local sh_size = M.read_u32_le(sh, M.ELF32.sh_size_offset)
|
||||||
|
|
||||||
|
-- Extract the name (null-terminated C string in strtab).
|
||||||
|
local name_end = strtab:find("\0", sh_name + 1, true) or (sh_name + 1)
|
||||||
|
local name = strtab:sub(sh_name + 1, name_end - 1)
|
||||||
|
if wanted[name] then
|
||||||
|
result[name] = read_section_bytes(sh_offset, sh_size)
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
@@ -611,95 +635,56 @@ function M.read_elf_sections(elf_path, section_names)
|
|||||||
end
|
end
|
||||||
|
|
||||||
--- Read ELF symbol addresses by walking the `.symtab` + `.strtab` sections directly (no `nm` subprocess).
|
--- Read ELF symbol addresses by walking the `.symtab` + `.strtab` sections directly (no `nm` subprocess).
|
||||||
--- Returns a map `{name -> {addr, size_bytes}}` for every defined symbol.
|
--- Returns a map `{name -> {addr, size_bytes}}` for every `code_<name>` symbol.
|
||||||
---
|
---
|
||||||
--- **Conventions:**
|
--- **Conventions:**
|
||||||
--- - ELF32 symtab entry = 16 bytes (`st_name:4 + st_value:4 + st_size:4 + st_info:1 + st_other:1 + st_shndx:2`); offsets within each entry are zero-based wire offsets.
|
--- - ELF32 symtab entry = 16 bytes (`st_name:4 + st_value:4 + st_size:4 + st_info:1 + st_other:1 + st_shndx:2`); offsets within each entry are zero-based wire offsets.
|
||||||
--- - Direct Lua `string.byte`/`string.sub`/`string.find` boundaries receive `+ 1`.
|
--- - Direct Lua `string.byte`/`string.sub`/`string.find` boundaries receive `+ 1`.
|
||||||
--- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded.
|
--- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded.
|
||||||
--- - Keys are the ELF symbol names as written (the C ident).
|
--- - The `code_` prefix is stripped (MipsAtom_ macros emit bare atom names, no `code_` prefix).
|
||||||
--- - `st_size > 0` filter excludes undefined/imported symbols.
|
--- - `st_size > 0` filter excludes undefined/imported symbols.
|
||||||
---
|
|
||||||
--- @param elf_path Path
|
--- @param elf_path Path
|
||||||
--- @return table<string, {integer, integer}>
|
--- @return table<string, {integer, integer}>
|
||||||
function M.read_nm(elf_path)
|
function M.read_nm(elf_path)
|
||||||
local addrs = {}
|
local addrs = {}
|
||||||
|
|
||||||
-- Existence check first; an empty or missing ELF returns an empty map.
|
-- Read .symtab + .strtab via the existing ELF walker (no subprocess).
|
||||||
if lfs.attributes(elf_path, "mode") ~= "file" then
|
local sections = M.read_elf_sections(elf_path, {".symtab", ".strtab"})
|
||||||
|
local symtab = sections[".symtab"]
|
||||||
|
local strtab = sections[".strtab"]
|
||||||
|
if not symtab or not strtab or #symtab == 0 or #strtab == 0 then
|
||||||
|
-- No symbol table (e.g. stripped ELF). Return empty.
|
||||||
return addrs
|
return addrs
|
||||||
end
|
end
|
||||||
|
|
||||||
local f = io.open(elf_path, "rb")
|
-- Iterate the 16-byte ELF32 symtab entries.
|
||||||
if not f then
|
-- Each entry (zero-based): st_name at 0, st_value at 4, st_size at 8, st_info at 12, st_other at 13, st_shndx at 14.
|
||||||
return addrs
|
local SYM_ENTRY_BYTES = 0x10
|
||||||
end
|
local SYM_ST_NAME = 0x00
|
||||||
|
local SYM_ST_VALUE = 0x04
|
||||||
-- Build the file adapter for E.*.
|
local SYM_ST_SIZE = 0x08
|
||||||
local file_size
|
local SYM_ST_INFO = 0x0C
|
||||||
do
|
local n_syms = #symtab / SYM_ENTRY_BYTES
|
||||||
f:seek("end", 0)
|
for i = 0, n_syms - 1 do
|
||||||
file_size = f:seek("cur", 0)
|
local entry_off = i * SYM_ENTRY_BYTES
|
||||||
end
|
local st_info = symtab:byte(entry_off + SYM_ST_INFO + 1)
|
||||||
local adapter = {
|
-- High nibble = binding (STB_LOCAL=0, STB_GLOBAL=1, STB_WEAK=2).
|
||||||
read_u8_at = function(offset)
|
-- Use math.floor(/16) instead of bit.rshift for LuaJIT 2.1 compat (LuaJIT's `>>` is 5.3+, but math.floor(x/16) works on all versions).
|
||||||
f:seek("set", offset)
|
local binding = math.floor(st_info / 16)
|
||||||
local b = f:read(1)
|
if binding == 0 or binding == 1 then -- STB_LOCAL or STB_GLOBAL
|
||||||
if not b then return nil end
|
local st_size = M.read_u32_le(symtab, entry_off + SYM_ST_SIZE)
|
||||||
return b:byte()
|
if st_size > 0 then
|
||||||
end,
|
local st_name_off = M.read_u32_le(symtab, entry_off + SYM_ST_NAME)
|
||||||
read_u16_at = function(offset)
|
-- Extract the name from .strtab (null-terminated C string).
|
||||||
f:seek("set", offset)
|
local name_end = strtab:find("\0", st_name_off + 1, true) or (st_name_off + 1)
|
||||||
local b1 = f:read(1)
|
local name = strtab:sub(st_name_off + 1, name_end - 1)
|
||||||
local b2 = f:read(1)
|
-- Filter: keep all symbol-table symbols (atoms emit their name as the bare `<name>` — MipsAtom_ macros strip the `code_` prefix).
|
||||||
if not b1 or not b2 then return nil end
|
-- The atoms_source_map pass already filters out non-atom symbols via the source-map.txt cross-ref.
|
||||||
return b1:byte() + b2:byte() * 0x100
|
if name and #name > 0 then
|
||||||
end,
|
local st_value = M.read_u32_le(symtab, entry_off + SYM_ST_VALUE)
|
||||||
read_u32_at = function(offset)
|
addrs[name] = { st_value, st_size }
|
||||||
f:seek("set", offset)
|
end
|
||||||
local b1 = f:read(1)
|
end
|
||||||
local b2 = f:read(1)
|
|
||||||
local b3 = f:read(1)
|
|
||||||
local b4 = f:read(1)
|
|
||||||
if not b1 or not b2 or not b3 or not b4 then return nil end
|
|
||||||
return b1:byte() + b2:byte() * 0x100
|
|
||||||
+ b3:byte() * 0x10000 + b4:byte() * 0x1000000
|
|
||||||
end,
|
|
||||||
read_size = function() return file_size end,
|
|
||||||
}
|
|
||||||
|
|
||||||
-- Delegate the header + section walk to E.*.
|
|
||||||
local hdr, hdr_err = E.parse_elf32_headers(adapter)
|
|
||||||
if not hdr then
|
|
||||||
io.stderr:write(string.format("[elf_dwarf.read_nm] header parse failed: %s\n", tostring(hdr_err)))
|
|
||||||
f:close()
|
|
||||||
return addrs
|
|
||||||
end
|
|
||||||
|
|
||||||
local sections, walk_err = E.walk_sections(adapter, hdr)
|
|
||||||
if not sections then
|
|
||||||
io.stderr:write(string.format("[elf_dwarf.read_nm] section walk failed: %s\n", tostring(walk_err)))
|
|
||||||
f:close()
|
|
||||||
return addrs
|
|
||||||
end
|
|
||||||
|
|
||||||
-- E.collect_symbols returns every defined symbol (no binding filter).
|
|
||||||
-- The metaprogram then applies its STB_LOCAL / STB_GLOBAL + size>0 filter, matching `nm`'s default (external symbols only).
|
|
||||||
local symbols, sym_err = E.collect_symbols(adapter, sections)
|
|
||||||
if not symbols then
|
|
||||||
io.stderr:write(string.format("[elf_dwarf.read_nm] symbol collection failed: %s\n", tostring(sym_err)))
|
|
||||||
f:close()
|
|
||||||
return addrs
|
|
||||||
end
|
|
||||||
|
|
||||||
f:close()
|
|
||||||
|
|
||||||
for name, entry in pairs(symbols) do
|
|
||||||
-- High nibble of st_info = binding (STB_LOCAL=0, STB_GLOBAL=1, STB_WEAK=2).
|
|
||||||
-- math.floor(/16) is portable across LuaJIT 2.0/2.1 and plain Lua 5.x.
|
|
||||||
local binding = math.floor(entry.info / 16)
|
|
||||||
if (binding == 0 or binding == 1) and entry.size > 0 then
|
|
||||||
addrs[name] = { entry.value, entry.size }
|
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
@@ -837,11 +822,12 @@ end
|
|||||||
--- * The `.debug_line` section may contain MULTIPLE line-program units
|
--- * The `.debug_line` section may contain MULTIPLE line-program units
|
||||||
--- File indices are 1-based, **per unit**; we concatenate all units and the index ranges from 1..N₁ in unit 1, N₁+1..N₁+N₂ in unit 2, etc.
|
--- File indices are 1-based, **per unit**; we concatenate all units and the index ranges from 1..N₁ in unit 1, N₁+1..N₁+N₂ in unit 2, etc.
|
||||||
--- Per-unit indices (the way gcc emits them, and the way `DW_LNS_set_file` references them in the line program)
|
--- Per-unit indices (the way gcc emits them, and the way `DW_LNS_set_file` references them in the line program)
|
||||||
--- are returned via the `basename_to_index` map only when the unit boundary happens to align with the metaprogram's per-atom `inv.call_file`
|
--- are returned via the `basename_to_index` map only when the unit boundary happens to align with the metaprogram's per-atom
|
||||||
|
--- `inv.call_file` (true today for hello_joypad — the C unit is the LAST unit, and atom-side file indices fit 1-based).
|
||||||
--- * Per spec, the `.debug_line_str` section (DWARF5 §7.5.6) holds the strings referenced by `DW_FORM_line_strp`.
|
--- * Per spec, the `.debug_line_str` section (DWARF5 §7.5.6) holds the strings referenced by `DW_FORM_line_strp`.
|
||||||
--- The legacy DWARF3 format embeds strings directly with null terminators. This helper handles BOTH.
|
--- The legacy DWARF3 format embeds strings directly with null terminators. This helper handles BOTH.
|
||||||
--- * File entries may have multiple forms (gcc -gdwarf-5 with `DW_LNCT_directory_index` emits 2 forms: path + dir_index).
|
--- * File entries may have multiple forms (gcc -gdwarf-5 with `DW_LNCT_directory_index`
|
||||||
--- The helper supports:
|
--- emits 2 forms: path + dir_index). The helper supports:
|
||||||
--- - DW_FORM_line_strp (DWARF5; offset into .debug_line_str)
|
--- - DW_FORM_line_strp (DWARF5; offset into .debug_line_str)
|
||||||
--- - DW_FORM_string (DWARF4-compat; inline null-terminated in .debug_line)
|
--- - DW_FORM_string (DWARF4-compat; inline null-terminated in .debug_line)
|
||||||
--- - DW_FORM_udata (ULEB128)
|
--- - DW_FORM_udata (ULEB128)
|
||||||
@@ -851,7 +837,9 @@ end
|
|||||||
---
|
---
|
||||||
--- Behavior on failure: writes to stderr and returns nil.
|
--- Behavior on failure: writes to stderr and returns nil.
|
||||||
--- Helpers consumed by `passes/dwarf_injection.lua::init_file_index_lookup(elf_path)` calls this once at pass start to populate the module-level `basename_to_index` map;
|
--- Helpers consumed by `passes/dwarf_injection.lua::init_file_index_lookup(elf_path)` calls this once at pass start to populate the module-level `basename_to_index` map;
|
||||||
--- downstream `resolve_provenance_file_index(path)` consumers consult the map directly.
|
--- downstream `resolve_provenance_file_index(path)` consumers
|
||||||
|
--- (which replaced the former hardcoded `ATOM_SOURCE_FILE_INDEX` + `PROVENANCE_BASENAME_TO_FILE_INDEX` table per `conductor/tracks/dwarf_file_index_lookup_20260731/`)
|
||||||
|
--- consult the map directly.
|
||||||
---
|
---
|
||||||
--- @param elf_path string -- absolute path to the post-link ELF (typically the gcc-emitted `.elf` BEFORE dwarf_injector's splice; both shapes work since the splice preserves `.debug_line`)
|
--- @param elf_path string -- absolute path to the post-link ELF (typically the gcc-emitted `.elf` BEFORE dwarf_injector's splice; both shapes work since the splice preserves `.debug_line`)
|
||||||
--- @return table|nil, table|nil, table|nil
|
--- @return table|nil, table|nil, table|nil
|
||||||
|
|||||||
@@ -16,7 +16,7 @@ define tape_atoms
|
|||||||
echo "[gdb_tape_atoms] STUB: run .\\build_psyq.ps1 to regenerate, then re-source this file."
|
echo "[gdb_tape_atoms] STUB: run .\\build_psyq.ps1 to regenerate, then re-source this file."
|
||||||
end
|
end
|
||||||
document tape_atoms
|
document tape_atoms
|
||||||
List every tape atom symbol in the loaded ELF with its .rodata address and word count.
|
List every tape atom symbol in the loaded ELF (code_<name>) with its .rodata address and word count.
|
||||||
STUB state: runtime file not sourced. Run build_psyq.ps1 to regenerate.
|
STUB state: runtime file not sourced. Run build_psyq.ps1 to regenerate.
|
||||||
end
|
end
|
||||||
|
|
||||||
|
|||||||
@@ -1,40 +1,77 @@
|
|||||||
# scripts/launch_pcsx_debug.ps1
|
# scripts/launch_pcsx_debug.ps1
|
||||||
#
|
#
|
||||||
# One-shot launcher for debug sessions:
|
# One-shot launcher for debug sessions: starts pcsx-redux with the .ps-exe
|
||||||
# Starts pcsx-redux with the .ps-exe loaded, the gdb stub enabled,
|
# loaded, the gdb stub enabled, the web server enabled, AND the
|
||||||
# AND the pcsx_debug_helper Lua plugin loaded so external CLI tools (gdb's `shell` command, etc.)
|
# pcsx_debug_helper Lua plugin loaded so external CLI tools can drive
|
||||||
# can read GTE state via http://localhost:8080/api/v1/lua/gte (the gdb stub doesn't expose COP2 at all).
|
# reloads via http://localhost:8080/api/v1/lua/reload.
|
||||||
#
|
|
||||||
# usage:
|
|
||||||
# .\scripts\launch_pcsx_debug.ps1
|
|
||||||
# .\scripts\launch_pcsx_debug.ps1 -ExePath build\hello_gte.ps-exe
|
|
||||||
# .\scripts\launch_pcsx_debug.ps1 -HelperZip scripts\pcsx_debug_helper.zip
|
|
||||||
#
|
#
|
||||||
# After launch:
|
# After launch:
|
||||||
# - gdb: target remote localhost:3333
|
# - gdb: target remote localhost:3333
|
||||||
# - web: curl http://localhost:8080/api/v1/lua/gte
|
# - web: POST http://localhost:8080/api/v1/lua/reload?mode=prime&...
|
||||||
|
#
|
||||||
|
# usage:
|
||||||
|
# .\scripts\launch_pcsx_debug.ps1
|
||||||
|
# .\scripts\launch_pcsx_debug.ps1 -ExePath build\hello_camera.ps-exe
|
||||||
|
# .\scripts\launch_pcsx_debug.ps1 -Cpu dynarec
|
||||||
|
# .\scripts\launch_pcsx_debug.ps1 -ElfPath build\hello_camera.elf
|
||||||
#
|
#
|
||||||
# Companion: scripts/debug_psyq.ps1 (bare launch — no .ps-exe, no helper).
|
# Companion: scripts/debug_psyq.ps1 (bare launch — no .ps-exe, no helper).
|
||||||
|
|
||||||
[CmdletBinding()]
|
[CmdletBinding()]
|
||||||
param(
|
param(
|
||||||
[string]$PcsxPath = (Join-Path $PSScriptRoot '..\toolchain\pcsx-redux\vsprojects\x64\Release\pcsx-redux.exe'),
|
[string]$PcsxPath = (Join-Path $PSScriptRoot '..\toolchain\pcsx-redux\vsprojects\x64\Release\pcsx-redux.exe'),
|
||||||
[string]$ExePath = (Join-Path $PSScriptRoot '..\build\hello_gte.ps-exe'),
|
[string]$ExePath = (Join-Path $PSScriptRoot '..\build\hello_camera.ps-exe'),
|
||||||
|
[string]$ElfPath = '',
|
||||||
[string]$HelperZip = (Join-Path $PSScriptRoot 'pcsx_debug_helper.zip'),
|
[string]$HelperZip = (Join-Path $PSScriptRoot 'pcsx_debug_helper.zip'),
|
||||||
[int] $GdbPort = 3333,
|
[int] $GdbPort = 3333,
|
||||||
[int] $WebPort = 8080
|
[int] $WebPort = 8080,
|
||||||
|
[ValidateSet('interpreter', 'dynarec')][string]$Cpu = 'interpreter'
|
||||||
)
|
)
|
||||||
|
|
||||||
$ErrorActionPreference = 'Stop'
|
$ErrorActionPreference = 'Stop'
|
||||||
|
|
||||||
|
# ── Derive -ElfPath when absent ──
|
||||||
|
# Convention: the .elf sits beside the .ps-exe with the same stem.
|
||||||
|
if ([string]::IsNullOrEmpty($ElfPath)) {
|
||||||
|
$exeFull = [System.IO.Path]::GetFullPath($ExePath)
|
||||||
|
$stem = [System.IO.Path]::GetFileNameWithoutExtension($exeFull)
|
||||||
|
$exeDir = [System.IO.Path]::GetDirectoryName($exeFull)
|
||||||
|
$ElfPath = Join-Path $exeDir "$stem.elf"
|
||||||
|
}
|
||||||
|
|
||||||
# ── Pre-checks ──
|
# ── Pre-checks ──
|
||||||
foreach ($p in @($PcsxPath, $ExePath, $HelperZip)) {
|
foreach ($p in @($PcsxPath, $ExePath, $ElfPath, $HelperZip)) {
|
||||||
if (-not (Test-Path $p)) {
|
if (-not (Test-Path -LiteralPath $p)) {
|
||||||
Write-Error "Missing: $p"
|
Write-Error "Missing: $p"
|
||||||
exit 1
|
exit 1
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# ── Reject a stale helper zip (Task 8) ──
|
||||||
|
# The helper zip must be newer than every .lua source that contributes
|
||||||
|
# to it. A stale zip means the running plugin does not match the on-disk
|
||||||
|
# source, which makes the reload contract meaningless.
|
||||||
|
$helperDir = Join-Path $PSScriptRoot 'pcsx_debug_helper'
|
||||||
|
$elf32Src = Join-Path $PSScriptRoot 'elf32.lua'
|
||||||
|
$sourceLuas = @(
|
||||||
|
(Join-Path $helperDir 'autoexec.lua'),
|
||||||
|
(Join-Path $helperDir 'reload.lua'),
|
||||||
|
$elf32Src
|
||||||
|
) | Where-Object { Test-Path -LiteralPath $_ }
|
||||||
|
|
||||||
|
$zipTime = (Get-Item -LiteralPath $HelperZip).LastWriteTime
|
||||||
|
$stale = $false
|
||||||
|
foreach ($src in $sourceLuas) {
|
||||||
|
$srcTime = (Get-Item -LiteralPath $src).LastWriteTime
|
||||||
|
if ($srcTime -gt $zipTime) {
|
||||||
|
Write-Error "helper zip is older than source: $src (zip=$($zipTime.ToString('o')) src=$($srcTime.ToString('o')); rerun build_psyq.ps1 to regenerate."
|
||||||
|
$stale = $true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if ($stale) {
|
||||||
|
exit 1
|
||||||
|
}
|
||||||
|
|
||||||
# Kill any existing pcsx-redux so the archive file isn't locked.
|
# Kill any existing pcsx-redux so the archive file isn't locked.
|
||||||
Get-Process pcsx-redux -ErrorAction SilentlyContinue | Stop-Process -Force
|
Get-Process pcsx-redux -ErrorAction SilentlyContinue | Stop-Process -Force
|
||||||
Start-Sleep -Seconds 2
|
Start-Sleep -Seconds 2
|
||||||
@@ -43,17 +80,23 @@ Start-Sleep -Seconds 2
|
|||||||
$absExe = [System.IO.Path]::GetFullPath($ExePath)
|
$absExe = [System.IO.Path]::GetFullPath($ExePath)
|
||||||
$absZip = [System.IO.Path]::GetFullPath($HelperZip)
|
$absZip = [System.IO.Path]::GetFullPath($HelperZip)
|
||||||
|
|
||||||
|
$cpuFlag = if ($Cpu -eq 'dynarec') { '-dynarec' } else { '-interpreter' }
|
||||||
|
|
||||||
$args = @(
|
$args = @(
|
||||||
'-gdb', '-run'
|
'-gdb', '-run'
|
||||||
'-loadexe', "`"$absExe`""
|
'-loadexe', "`"$absExe`""
|
||||||
'-archive', "`"$absZip`""
|
'-archive', "`"$absZip`""
|
||||||
|
'-webserver'
|
||||||
|
$cpuFlag
|
||||||
)
|
)
|
||||||
|
|
||||||
Write-Host "Launching pcsx-redux..." -ForegroundColor Cyan
|
Write-Host "Launching pcsx-redux..." -ForegroundColor Cyan
|
||||||
Write-Host " ps-exe : $absExe"
|
Write-Host " ps-exe : $absExe"
|
||||||
|
Write-Host " elf : $ElfPath"
|
||||||
Write-Host " helper zip: $absZip"
|
Write-Host " helper zip: $absZip"
|
||||||
Write-Host " gdb : localhost:$GdbPort"
|
Write-Host " gdb : localhost:$GdbPort"
|
||||||
Write-Host " web : localhost:$WebPort/api/v1/lua/gte"
|
Write-Host " web : localhost:$WebPort/api/v1/lua/reload"
|
||||||
|
Write-Host " cpu : $Cpu ($cpuFlag)"
|
||||||
Write-Host ""
|
Write-Host ""
|
||||||
|
|
||||||
Start-Process -FilePath $PcsxPath -ArgumentList $args | Out-Null
|
Start-Process -FilePath $PcsxPath -ArgumentList $args | Out-Null
|
||||||
@@ -83,12 +126,49 @@ try {
|
|||||||
$r = Invoke-WebRequest -Uri "http://localhost:$WebPort/api/v1/lua/gte" -UseBasicParsing -TimeoutSec 5
|
$r = Invoke-WebRequest -Uri "http://localhost:$WebPort/api/v1/lua/gte" -UseBasicParsing -TimeoutSec 5
|
||||||
$firstLine = ([System.Text.Encoding]::UTF8.GetString($r.Content) -split "`n")[0]
|
$firstLine = ([System.Text.Encoding]::UTF8.GetString($r.Content) -split "`n")[0]
|
||||||
Write-Host "GTE handler OK: $firstLine" -ForegroundColor Green
|
Write-Host "GTE handler OK: $firstLine" -ForegroundColor Green
|
||||||
}
|
} catch {
|
||||||
catch {
|
|
||||||
Write-Warning "GTE handler NOT responding: $_"
|
Write-Warning "GTE handler NOT responding: $_"
|
||||||
Write-Host "Check the pcsx-redux Lua Console for debug cli messages." -ForegroundColor Yellow
|
Write-Host "Check the pcsx-redux Lua Console for debug cli messages." -ForegroundColor Yellow
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# ── Prime the reload handler (Task 8) ──
|
||||||
|
# The reload handler keeps an internal ACTIVE manifest of the running
|
||||||
|
# ELF; reload requests fail with reload_not_primed until prime succeeds.
|
||||||
|
# We retry until the response carries ok=true or the launch deadline
|
||||||
|
# expires — the helper may not have finished registering handlers in the
|
||||||
|
# first web-poll cycle after the gte handler comes up.
|
||||||
|
$absElf = [System.IO.Path]::GetFullPath($ElfPath)
|
||||||
|
$encodedPath = [uri]::EscapeDataString($absElf)
|
||||||
|
$primeUri = "http://localhost:${WebPort}/api/v1/lua/reload?mode=prime&target=hello_camera&path=${encodedPath}"
|
||||||
|
|
||||||
|
Write-Host "Priming reload handler: $primeUri" -ForegroundColor Cyan
|
||||||
|
$primeDeadline = (Get-Date).AddSeconds(15)
|
||||||
|
$primeOk = $false
|
||||||
|
while ((Get-Date) -lt $primeDeadline) {
|
||||||
|
try {
|
||||||
|
$resp = Invoke-WebRequest -Method Post -Uri $primeUri -UseBasicParsing -TimeoutSec 5
|
||||||
|
$body = if ($resp.Content -is [byte[]]) {
|
||||||
|
[System.Text.Encoding]::UTF8.GetString([byte[]]$resp.Content)
|
||||||
|
} else {
|
||||||
|
[string]$resp.Content
|
||||||
|
}
|
||||||
|
$obj = $body | ConvertFrom-Json
|
||||||
|
if ($obj.ok) {
|
||||||
|
Write-Host "Prime OK: $(($obj | ConvertTo-Json -Compress))" -ForegroundColor Green
|
||||||
|
$primeOk = $true
|
||||||
|
break
|
||||||
|
} else {
|
||||||
|
Write-Host "Prime not yet ready: error=$($obj.error)" -ForegroundColor Yellow
|
||||||
|
}
|
||||||
|
} catch {
|
||||||
|
Write-Host "Prime request failed: $($_.Exception.Message)" -ForegroundColor Yellow
|
||||||
|
}
|
||||||
|
Start-Sleep -Milliseconds 500
|
||||||
|
}
|
||||||
|
if (-not $primeOk) {
|
||||||
|
Write-Warning "Prime did not return ok=true before the launch deadline. Reload requests will fail until the user primes manually."
|
||||||
|
}
|
||||||
|
|
||||||
Write-Host ""
|
Write-Host ""
|
||||||
Write-Host "pcsx-redux running. PIDs:" -ForegroundColor Cyan
|
Write-Host "pcsx-redux running. PIDs:" -ForegroundColor Cyan
|
||||||
Get-Process pcsx-redux | Select-Object Id, ProcessName | Format-Table
|
Get-Process pcsx-redux | Select-Object Id, ProcessName | Format-Table
|
||||||
@@ -0,0 +1,82 @@
|
|||||||
|
# make_helper_zip.ps1
|
||||||
|
#
|
||||||
|
# Regenerate scripts/pcsx_debug_helper.zip from scripts/pcsx_debug_helper/.
|
||||||
|
# The archive contains exactly three entries at archive root:
|
||||||
|
#
|
||||||
|
# autoexec.lua
|
||||||
|
# elf32.lua (copied in from scripts/elf32.lua before packaging)
|
||||||
|
# reload.lua
|
||||||
|
#
|
||||||
|
# Determinism: CreateFromDirectory on the same set of files produces
|
||||||
|
# identical bytes. Verified by running the same command twice and
|
||||||
|
# asserting SHA-256 equality (see plan.md Task 6 Step 4).
|
||||||
|
#
|
||||||
|
# Performance: the implementation uses System.IO.Compression.ZipFile
|
||||||
|
# (BCL, in-process). Benchmarked: ~2 ms cold, ~2 ms warm on this
|
||||||
|
# workstation. Compress-Archive is rejected because its first call
|
||||||
|
# takes ~200 ms (assembly load) and subsequent calls take ~16 ms
|
||||||
|
# (process spawn per invocation). The 50 ms budget documented in
|
||||||
|
# plan.md Task 8 Step 3 excludes the compiler/assembler toolchain.
|
||||||
|
#
|
||||||
|
# Usage:
|
||||||
|
# pwsh -NoProfile -File scripts\make_helper_zip.ps1
|
||||||
|
#
|
||||||
|
# Optional -OutputPath switches the destination. Default is
|
||||||
|
# scripts/pcsx_debug_helper.zip next to the helper dir.
|
||||||
|
#
|
||||||
|
# Companion: scripts/pcsx_debug_helper/{autoexec,elf32,reload}.lua
|
||||||
|
# tests/reload_helper_zip_regen.ps1 (planned Task 8 verifier)
|
||||||
|
|
||||||
|
[CmdletBinding()]
|
||||||
|
param(
|
||||||
|
[string]$HelperDir = (Join-Path $PSScriptRoot 'pcsx_debug_helper'),
|
||||||
|
[string]$SourcesDir = $PSScriptRoot,
|
||||||
|
[string]$OutputPath = (Join-Path $PSScriptRoot 'pcsx_debug_helper.zip')
|
||||||
|
)
|
||||||
|
|
||||||
|
$ErrorActionPreference = 'Stop'
|
||||||
|
|
||||||
|
if (-not (Test-Path -LiteralPath $HelperDir)) {
|
||||||
|
throw "helper dir not found: $HelperDir"
|
||||||
|
}
|
||||||
|
|
||||||
|
# Stage elf32.lua into the helper dir so the in-process ZipFile walker
|
||||||
|
# picks it up alongside the helper-local files. elf32.lua is the shared
|
||||||
|
# ELF32 byte reader; the production reload.lua loads it through
|
||||||
|
# Support.extra.dofile("elf32.lua") at runtime.
|
||||||
|
$elf32Src = Join-Path $SourcesDir 'elf32.lua'
|
||||||
|
$elf32Dest = Join-Path $HelperDir 'elf32.lua'
|
||||||
|
if (-not (Test-Path -LiteralPath $elf32Src)) {
|
||||||
|
throw "elf32.lua not found at $elf32Src"
|
||||||
|
}
|
||||||
|
Copy-Item -LiteralPath $elf32Src -Destination $elf32Dest -Force
|
||||||
|
|
||||||
|
try {
|
||||||
|
# Remove any existing archive so CreateFromDirectory can write fresh.
|
||||||
|
# ZipFile.CreateFromDirectory throws if the destination exists.
|
||||||
|
if (Test-Path -LiteralPath $OutputPath) {
|
||||||
|
Remove-Item -LiteralPath $OutputPath -Force
|
||||||
|
}
|
||||||
|
|
||||||
|
# In-process zip; ~2 ms cold, ~2 ms warm. BCL compression matches
|
||||||
|
# Compress-Archive at CompressionLevel Optimal for these small files.
|
||||||
|
# Assembly is loaded once per pwsh.exe; the first run pays ~14 ms,
|
||||||
|
# subsequent runs pay ~0.2 ms.
|
||||||
|
Add-Type -AssemblyName System.IO.Compression.FileSystem
|
||||||
|
[System.IO.Compression.ZipFile]::CreateFromDirectory(
|
||||||
|
$HelperDir, $OutputPath,
|
||||||
|
[System.IO.Compression.CompressionLevel]::Optimal,
|
||||||
|
$false) | Out-Null
|
||||||
|
|
||||||
|
$sha = (Get-FileHash -LiteralPath $OutputPath -Algorithm SHA256).Hash
|
||||||
|
Write-Output ("[make_helper_zip] wrote {0} bytes, sha256={1}" -f `
|
||||||
|
(Get-Item -LiteralPath $OutputPath).Length, $sha)
|
||||||
|
Write-Output "[make_helper_zip] entries: autoexec.lua, elf32.lua, reload.lua"
|
||||||
|
}
|
||||||
|
finally {
|
||||||
|
# Remove the staged elf32.lua so the helper directory only contains
|
||||||
|
# the files the user expects to see there.
|
||||||
|
if (Test-Path -LiteralPath $elf32Dest) {
|
||||||
|
Remove-Item -LiteralPath $elf32Dest -Force
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -115,7 +115,7 @@ end
|
|||||||
--- Post-loop: Needs full-corpus `annot_counts` from pipe_ctx.
|
--- Post-loop: Needs full-corpus `annot_counts` from pipe_ctx.
|
||||||
--- @param pipe_ctx PipeCtx
|
--- @param pipe_ctx PipeCtx
|
||||||
--- @param findings Findings
|
--- @param findings Findings
|
||||||
local function check_unique_annotation(_item, pipe_ctx, findings)
|
local function check_unique_annotation(pipe_ctx, findings)
|
||||||
for name, n in pairs(pipe_ctx.annot_counts) do
|
for name, n in pairs(pipe_ctx.annot_counts) do
|
||||||
if n > 1 then
|
if n > 1 then
|
||||||
findings.errors[#findings.errors + 1] = {
|
findings.errors[#findings.errors + 1] = {
|
||||||
@@ -147,8 +147,7 @@ end
|
|||||||
--- @param m MacroEntry
|
--- @param m MacroEntry
|
||||||
--- @param wc table<string, integer> -- Shared word-count table (from ctx.shared.word_counts)
|
--- @param wc table<string, integer> -- Shared word-count table (from ctx.shared.word_counts)
|
||||||
--- @param findings Findings
|
--- @param findings Findings
|
||||||
local function check_macro_word_drift(m, pipe_ctx, findings)
|
local function check_macro_word_drift(m, wc, findings)
|
||||||
local wc = (pipe_ctx and pipe_ctx.word_counts) or {}
|
|
||||||
local declared = wc[m.name]
|
local declared = wc[m.name]
|
||||||
if not declared then
|
if not declared then
|
||||||
findings.errors[#findings.errors + 1] = {
|
findings.errors[#findings.errors + 1] = {
|
||||||
@@ -430,17 +429,40 @@ local CHECK_RULES = {
|
|||||||
--- @param ctx PassCtx
|
--- @param ctx PassCtx
|
||||||
--- @return PipeCtx
|
--- @return PipeCtx
|
||||||
local function build_corpus_pipe_ctx(ctx)
|
local function build_corpus_pipe_ctx(ctx)
|
||||||
local view = duffle.corpus_view(ctx)
|
local corpus = ctx.shared and ctx.shared.corpus
|
||||||
|
if not corpus then
|
||||||
|
error("annotation requires ctx.shared.corpus "
|
||||||
|
.. "(the canonical corpus is the source of truth; "
|
||||||
|
.. "no per-source fallback is supported)", 0)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- `corpus.atom_infos` preserves source order and duplicates; I precompute counts here for `check_unique_annotation` and the per-source checks.
|
||||||
local annot_counts = {}
|
local annot_counts = {}
|
||||||
for _, info in ipairs(view.atom_infos) do
|
for _, info in ipairs(corpus.atom_infos or {}) do
|
||||||
if info and info.atom_name then
|
if info and info.atom_name then
|
||||||
annot_counts[info.atom_name] = (annot_counts[info.atom_name] or 0) + 1
|
annot_counts[info.atom_name] = (annot_counts[info.atom_name] or 0) + 1
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
view.annot_counts = annot_counts
|
|
||||||
view.atom_infos_list = view.atom_infos
|
-- Every consumer of these fields observes mutations via the canonical corpus without independently mutable registry construction.
|
||||||
view.word_counts = ctx.shared.corpus.word_counts or {}
|
return {
|
||||||
return view
|
-- Cross-source lookup tables from corpus.
|
||||||
|
register_alias_registry = corpus.register_alias_registry or {},
|
||||||
|
type_name_registry = corpus.type_name_registry or {},
|
||||||
|
atom_views = corpus.atom_views or {},
|
||||||
|
atom_ctxs = corpus.atom_ctxs or {},
|
||||||
|
atom_phases = corpus.atom_phases or {},
|
||||||
|
binds_by_name = corpus.binds_by_name or {},
|
||||||
|
atoms_by_name = corpus.atoms_by_name or {},
|
||||||
|
-- Corpus-wide ordered list of atom_info records (source-order + duplicates).
|
||||||
|
atom_infos_list = corpus.atom_infos or {},
|
||||||
|
-- Corpus-wide annotation count aggregation (post-rule consumes this).
|
||||||
|
annot_counts = annot_counts,
|
||||||
|
-- Corpus-wide collisions (recorded by scan_source.merge_corpus_registries).
|
||||||
|
collisions = corpus.collisions or {},
|
||||||
|
-- `check_macro_word_drift` reads `corpus.word_counts`, populated by word_count_eval.run.
|
||||||
|
word_counts = corpus.word_counts or {},
|
||||||
|
}
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Validate one source against its pre-scanned SourceScan payload + the corpus-wide pipe_ctx.
|
--- Validate one source against its pre-scanned SourceScan payload + the corpus-wide pipe_ctx.
|
||||||
@@ -455,8 +477,8 @@ local function validate(ctx, src, corpus_pipe_ctx)
|
|||||||
-- Project the pre-scanned atoms to the AtomEntry shape this pass needs.
|
-- Project the pre-scanned atoms to the AtomEntry shape this pass needs.
|
||||||
local atoms = {}
|
local atoms = {}
|
||||||
for _, a in ipairs(scan.atoms) do
|
for _, a in ipairs(scan.atoms) do
|
||||||
if a.kind == "atom" or a.kind == "atom_proc" then
|
if a.kind == "atom" then
|
||||||
atoms[#atoms + 1] = { line = a.line, name = a.raw_name or a.name }
|
atoms[#atoms + 1] = { line = a.line, name = a.raw_name }
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
@@ -514,28 +536,38 @@ local function validate(ctx, src, corpus_pipe_ctx)
|
|||||||
|
|
||||||
-- THE per-annotation pipeline. ONE loop. CHECK_RULES dispatches per_annot rules.
|
-- THE per-annotation pipeline. ONE loop. CHECK_RULES dispatches per_annot rules.
|
||||||
for _, a in ipairs(annots) do
|
for _, a in ipairs(annots) do
|
||||||
duffle.run_check_rules(CHECK_RULES, "per_annot", a, pipe_ctx, findings)
|
for _, rule in ipairs(CHECK_RULES) do
|
||||||
|
if rule.per_annot then rule.per_annot(a, pipe_ctx, findings) end
|
||||||
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
-- Post-loop rules (one-shot checks that need full-corpus aggregation in pipe_ctx).
|
-- Post-loop rules (one-shot checks that need full-corpus aggregation in pipe_ctx).
|
||||||
duffle.run_check_rules(CHECK_RULES, "post", nil, pipe_ctx, findings)
|
for _, rule in ipairs(CHECK_RULES) do
|
||||||
|
if rule.post then rule.post(pipe_ctx, findings) end
|
||||||
|
end
|
||||||
|
|
||||||
-- scan_source records each marker in scan.debug_skip_markers; this loop validates each record independently and emits at most one error per marker.
|
-- scan_source records each marker in scan.debug_skip_markers; this loop validates each record independently and emits at most one error per marker.
|
||||||
-- Valid markers stamp `debug_skip = true` on the following atom or component declaration, which downstream consumers read directly.
|
-- Valid markers stamp `debug_skip = true` on the following atom or component declaration, which downstream consumers read directly.
|
||||||
local skip_markers = scan.debug_skip_markers or {}
|
local skip_markers = scan.debug_skip_markers or {}
|
||||||
for _, marker in ipairs(skip_markers) do
|
for _, marker in ipairs(skip_markers) do
|
||||||
duffle.run_check_rules(CHECK_RULES, "per_skip_marker", marker, pipe_ctx, findings)
|
for _, rule in ipairs(CHECK_RULES) do
|
||||||
|
if rule.per_skip_marker then rule.per_skip_marker(marker, pipe_ctx, findings) end
|
||||||
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
-- Per-macro rules (TAPE_WORDS vs WORD_COUNT drift).
|
-- Per-macro rules (TAPE_WORDS vs WORD_COUNT drift).
|
||||||
pipe_ctx.word_counts = corpus_pipe_ctx.word_counts
|
local wc = corpus_pipe_ctx.word_counts
|
||||||
for _, m in ipairs(scan.macros) do
|
for _, m in ipairs(scan.macros) do
|
||||||
duffle.run_check_rules(CHECK_RULES, "per_macro", m, pipe_ctx, findings)
|
for _, rule in ipairs(CHECK_RULES) do
|
||||||
|
if rule.per_macro then rule.per_macro(m, wc, findings) end
|
||||||
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
-- Per-source rules (reg defaults, atom_view layout, compute-register type overrides, Binds_* field uniqueness).
|
-- Per-source rules (reg defaults, atom_view layout, compute-register type overrides, Binds_* field uniqueness).
|
||||||
-- Each per_source rule sees the full scan payload via pipe_ctx.
|
-- Each per_source rule sees the full scan payload via pipe_ctx.
|
||||||
duffle.run_check_rules(CHECK_RULES, "per_source", src, pipe_ctx, findings)
|
for _, rule in ipairs(CHECK_RULES) do
|
||||||
|
if rule.per_source then rule.per_source(src, pipe_ctx, findings) end
|
||||||
|
end
|
||||||
|
|
||||||
-- Information summary (always emitted).
|
-- Information summary (always emitted).
|
||||||
findings.info[#findings.info + 1] = {
|
findings.info[#findings.info + 1] = {
|
||||||
|
|||||||
@@ -83,7 +83,6 @@ local function canonical_word_entries(atom)
|
|||||||
line = event.call_line or item.line or 0,
|
line = event.call_line or item.line or 0,
|
||||||
text = event.call_text or item.call_text or "",
|
text = event.call_text or item.call_text or "",
|
||||||
body_line = event.body_line or item.body_line or item.line or 0,
|
body_line = event.body_line or item.body_line or item.line or 0,
|
||||||
gpr_keys = event.gpr_keys,
|
|
||||||
invocation = (event.outermost_invocation_id
|
invocation = (event.outermost_invocation_id
|
||||||
and paths.invocations
|
and paths.invocations
|
||||||
and paths.invocations[event.outermost_invocation_id]) or nil,
|
and paths.invocations[event.outermost_invocation_id]) or nil,
|
||||||
@@ -261,12 +260,12 @@ local function append_gdb_commands(lines, matched)
|
|||||||
for _, a in ipairs(matched) do
|
for _, a in ipairs(matched) do
|
||||||
-- gdb 12.1 quirk: literals in printf args require an attached target.
|
-- gdb 12.1 quirk: literals in printf args require an attached target.
|
||||||
-- Use the per-atom convenience vars set above as printf args.
|
-- Use the per-atom convenience vars set above as printf args.
|
||||||
lines[#lines + 1] = string.format(' printf " %%-32s @ 0x%%08x %%4d words\\n", $__atom_name_%d, $__atom_addr_%d, $__atom_words_%d',
|
lines[#lines + 1] = string.format(' printf " code_%%-32s @ 0x%%08x %%4d words\\n", $__atom_name_%d, $__atom_addr_%d, $__atom_words_%d',
|
||||||
a.idx, a.idx, a.idx)
|
a.idx, a.idx, a.idx)
|
||||||
end
|
end
|
||||||
lines[#lines + 1] = "end"
|
lines[#lines + 1] = "end"
|
||||||
lines[#lines + 1] = "document tape_atoms"
|
lines[#lines + 1] = "document tape_atoms"
|
||||||
lines[#lines + 1] = " List every tape atom symbol in the loaded ELF with .rodata addr + word count."
|
lines[#lines + 1] = " List every tape atom symbol in the loaded ELF (code_<name>) with .rodata addr + word count."
|
||||||
lines[#lines + 1] = "end"
|
lines[#lines + 1] = "end"
|
||||||
lines[#lines + 1] = ""
|
lines[#lines + 1] = ""
|
||||||
|
|
||||||
@@ -285,10 +284,10 @@ local function append_gdb_commands(lines, matched)
|
|||||||
for _, a in ipairs(matched) do
|
for _, a in ipairs(matched) do
|
||||||
lines[#lines + 1] = string.format("define break_atom_%s", a.name)
|
lines[#lines + 1] = string.format("define break_atom_%s", a.name)
|
||||||
lines[#lines + 1] = string.format(" break *$__atom_addr_%d", a.idx)
|
lines[#lines + 1] = string.format(" break *$__atom_addr_%d", a.idx)
|
||||||
lines[#lines + 1] = string.format(' printf " Breakpoint set at %s (0x%%08x)\\n", $__atom_addr_%d', a.name, a.idx)
|
lines[#lines + 1] = string.format(' printf " Breakpoint set at code_%s (0x%%08x)\\n", $__atom_addr_%d', a.name, a.idx)
|
||||||
lines[#lines + 1] = "end"
|
lines[#lines + 1] = "end"
|
||||||
lines[#lines + 1] = string.format("document break_atom_%s", a.name)
|
lines[#lines + 1] = string.format("document break_atom_%s", a.name)
|
||||||
lines[#lines + 1] = string.format(" Set a breakpoint at %s.", a.name)
|
lines[#lines + 1] = string.format(" Set a breakpoint at code_%s.", a.name)
|
||||||
lines[#lines + 1] = "end"
|
lines[#lines + 1] = "end"
|
||||||
lines[#lines + 1] = ""
|
lines[#lines + 1] = ""
|
||||||
end
|
end
|
||||||
@@ -323,7 +322,7 @@ local function append_gdb_commands(lines, matched)
|
|||||||
-- Precompute end_addr (gdb 12.1's expression evaluator chokes on `addr + words*4`).
|
-- Precompute end_addr (gdb 12.1's expression evaluator chokes on `addr + words*4`).
|
||||||
lines[#lines + 1] = string.format(" set $__end_%d = $__atom_addr_%d + $__atom_words_%d * 4", a.idx, a.idx, a.idx)
|
lines[#lines + 1] = string.format(" set $__end_%d = $__atom_addr_%d + $__atom_words_%d * 4", a.idx, a.idx, a.idx)
|
||||||
lines[#lines + 1] = string.format(" if $__pc >= $__atom_addr_%d && $__pc < $__end_%d", a.idx, a.idx)
|
lines[#lines + 1] = string.format(" if $__pc >= $__atom_addr_%d && $__pc < $__end_%d", a.idx, a.idx)
|
||||||
lines[#lines + 1] = string.format(' printf "atom: %%s\\n", $__atom_name_%d', a.idx)
|
lines[#lines + 1] = string.format(' printf "atom: code_%%s\\n", $__atom_name_%d', a.idx)
|
||||||
lines[#lines + 1] = ' printf "addr: 0x%08x\\n", $__pc'
|
lines[#lines + 1] = ' printf "addr: 0x%08x\\n", $__pc'
|
||||||
lines[#lines + 1] = string.format(" set $__word = ($__pc - $__atom_addr_%d) / 4", a.idx)
|
lines[#lines + 1] = string.format(" set $__word = ($__pc - $__atom_addr_%d) / 4", a.idx)
|
||||||
lines[#lines + 1] = string.format(' printf "word: %%d/%%d\\n", $__word, $__atom_words_%d', a.idx)
|
lines[#lines + 1] = string.format(' printf "word: %%d/%%d\\n", $__word, $__atom_words_%d', a.idx)
|
||||||
@@ -492,19 +491,8 @@ function M.render_atom_source_map(atom)
|
|||||||
local lines = {}
|
local lines = {}
|
||||||
lines[#lines + 1] = string.format("ATOM %s %d", (atom.raw_name or atom.name), total)
|
lines[#lines + 1] = string.format("ATOM %s %d", (atom.raw_name or atom.name), total)
|
||||||
for _, entry in ipairs(entries) do
|
for _, entry in ipairs(entries) do
|
||||||
local word_line = string.format("WORD %d LINE %d TEXT %s",
|
lines[#lines + 1] = string.format("WORD %d LINE %d TEXT %s",
|
||||||
entry.pos, entry.line, entry.text)
|
entry.pos, entry.line, entry.text)
|
||||||
local keys = {}
|
|
||||||
for pos = 1, 16 do
|
|
||||||
local k = entry.gpr_keys and entry.gpr_keys[pos]
|
|
||||||
if type(k) == "string" and k:sub(1, 7) == "reguse:" then
|
|
||||||
keys[#keys + 1] = k
|
|
||||||
end
|
|
||||||
end
|
|
||||||
if #keys > 0 then
|
|
||||||
word_line = word_line .. " KEYS " .. table.concat(keys, ",")
|
|
||||||
end
|
|
||||||
lines[#lines + 1] = word_line
|
|
||||||
end
|
end
|
||||||
lines[#lines + 1] = "ENDATOM"
|
lines[#lines + 1] = "ENDATOM"
|
||||||
return table.concat(lines, "\n") .. "\n"
|
return table.concat(lines, "\n") .. "\n"
|
||||||
@@ -541,7 +529,7 @@ function M.render_atom_provenance(atom, wc, rel_path)
|
|||||||
return table.concat(lines, "\n") .. "\n"
|
return table.concat(lines, "\n") .. "\n"
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Pass entry. For each source that declares at least one tape atom,
|
--- Pass entry. For each source that declares at least one `MipsAtom_(name)` / `MipsCode code_<name>`,
|
||||||
--- emit two files in `<out_root>/`: `<basename>.atoms.sourcemap.txt` (per-word call-site map) and `<basename>.atoms.provenance.txt`
|
--- emit two files in `<out_root>/`: `<basename>.atoms.sourcemap.txt` (per-word call-site map) and `<basename>.atoms.provenance.txt`
|
||||||
--- (per-word definition + body line, resolved via the outermost `mac_X(...)` invocation).
|
--- (per-word definition + body line, resolved via the outermost `mac_X(...)` invocation).
|
||||||
--- When `ctx.flags.gdb_runtime` is true and `ctx.flags.elf_path` exists, also emit the post-link gdb script `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`.
|
--- When `ctx.flags.gdb_runtime` is true and `ctx.flags.elf_path` exists, also emit the post-link gdb script `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`.
|
||||||
|
|||||||
@@ -1,343 +0,0 @@
|
|||||||
--- passes/auto_reg.lua — Per-phase automatic GPR allocator + gen/auto_reg.h emitter.
|
|
||||||
---
|
|
||||||
--- Reads the per-source + corpus-level `atom_auto_regs` + `phase_auto_regs` registries populated by `passes/scan_source.lua`.
|
|
||||||
--- Runs a deterministic first-fit allocator in the `R_T0..R_T7 + R_V0..R_V1` pool (10 physical GPRs).
|
|
||||||
--- Emits one `#define R_<Sym>_Code R_Tn_Code` per marker into per-directory `gen/auto_reg.h`.
|
|
||||||
---
|
|
||||||
--- User-pinned GPRs : The corpus's `register_alias_registry` is consulted to exclude GPRs the user has pinned via
|
|
||||||
--- `atom_reg` + `_Code` defs (e.g. carriers like `R_ResolveScratch = R_T4 atom_reg`).
|
|
||||||
--- These GPRs are unavailable to EVERY atom's source pool.
|
|
||||||
--- Carriers are preserved across atoms by context discipline and must never be reallocated.
|
|
||||||
--- Per-atom body parsing also catches alias references (R_<Alias>) and hardcoded R_Tn references,
|
|
||||||
--- so the user can write either `R_T4` or `R_ResolveScratch` in an atom body and the pass will
|
|
||||||
--- exclude R_T4 from that atom's pool.
|
|
||||||
---
|
|
||||||
--- Conflict detection: If the user hardcodes `R_Tn` in an atom body that shares a phase with an auto-reg that picked `R_Tn`,
|
|
||||||
--- emit `phase_register_clash` as an info finding (no build stop).
|
|
||||||
--- Should be unreachable after the user-pinning + body-parsing fix above; kept as a defensive safety net.
|
|
||||||
---
|
|
||||||
--- Pool exhaustion: If a phase declares more `R_<Sym>` mappings than the 10-register pool can hold,
|
|
||||||
--- emit `phase_register_pool_exhausted` as a build-stopping error.
|
|
||||||
|
|
||||||
--- @class AutoRegResult
|
|
||||||
--- @field outputs table[] -- {kind=, path=} entries
|
|
||||||
--- @field errors table[] -- {line=, msg=} entries (build-stops)
|
|
||||||
--- @field warnings table[] -- {line=, msg=} entries (build-continues)
|
|
||||||
|
|
||||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
|
||||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
|
||||||
-- THE GPR ALLOCATION POOL — what's allocatable, and (more importantly) WHY
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
|
||||||
--
|
|
||||||
-- The auto-reg pass picks physical GPRs for `atom_auto_reg(...)` / `phase_auto_reg(...)` markers.
|
|
||||||
-- The 24-register pool covers R2-R25 (the user/atom allocatable surface):
|
|
||||||
-- R_T0..R_T7, R_V0..R_V1, R_A0..A3, R_S0..S7, R_T8..T9.
|
|
||||||
-- Excluded (and never added to the pool):
|
|
||||||
-- R_0 (code 0) — hardwired zero. Cannot be written.
|
|
||||||
-- R_AT (code 1) — assembler temporary. Reserved by the MIPS O32 ABI.
|
|
||||||
-- R_A0..A3 — explicitly omitted above even though their integer codes
|
|
||||||
-- map to POOL entries; the pool-construction loop below
|
|
||||||
-- only references the POOL string literals, never the
|
|
||||||
-- integer codes, so they are NOT auto-allocated by default.
|
|
||||||
-- (A0-A3 become available when the user adds them to
|
|
||||||
-- POOL or hardcodes an R_A0 reference in the atom body.)
|
|
||||||
-- R_K0/K1 (codes 26-27) — kernel / interrupt handler reserves. Never touched by user code.
|
|
||||||
-- R_GP/SP/FP/RA (codes 28-31) — R_SP/R_FP/R_RA are tape-runtime carriers between
|
|
||||||
-- tape_enter and tape_exit; R_GP stays the host global pointer.
|
|
||||||
---
|
|
||||||
local POOL = {
|
|
||||||
"R_T0", "R_T1", "R_T2", "R_T3",
|
|
||||||
"R_T4", "R_T5", "R_T6", "R_T7",
|
|
||||||
"R_V0", "R_V1",
|
|
||||||
"R_A0", "R_A1", "R_A2", "R_A3",
|
|
||||||
"R_S0", "R_S1", "R_S2", "R_S3",
|
|
||||||
"R_S4", "R_S5", "R_S6", "R_S7",
|
|
||||||
"R_T8", "R_T9",
|
|
||||||
}
|
|
||||||
|
|
||||||
-- Map from integer MIPS GPR code (the `code` field on AliasEntry) to the physical GPR ident in POOL.
|
|
||||||
-- The standard MIPS O32 ABI register numbering matches mips.h's R_*_Code #defines (mips.h).
|
|
||||||
-- Only the POOL entries matter for auto_reg — non-pool aliases
|
|
||||||
-- (R_AT=1, R_A0..A3=4..7, R_T8=24, R_T9=25, R_K0/K1=26..27, R_GP/SP/FP/RA=28..31)
|
|
||||||
-- are deliberately omitted — see the comment block above for the WHY of each exclusion.
|
|
||||||
local INT_CODE_TO_POOL_GPR = {
|
|
||||||
[2] = "R_V0", [3] = "R_V1",
|
|
||||||
[4] = "R_A0", [5] = "R_A1", [6] = "R_A2", [7] = "R_A3",
|
|
||||||
[8] = "R_T0", [9] = "R_T1", [10] = "R_T2", [11] = "R_T3",
|
|
||||||
[12] = "R_T4", [13] = "R_T5", [14] = "R_T6", [15] = "R_T7",
|
|
||||||
[16] = "R_S0", [17] = "R_S1", [18] = "R_S2", [19] = "R_S3",
|
|
||||||
[20] = "R_S4", [21] = "R_S5", [22] = "R_S6", [23] = "R_S7",
|
|
||||||
[24] = "R_T8", [25] = "R_T9",
|
|
||||||
}
|
|
||||||
|
|
||||||
-- Stable sort for deterministic allocation order.
|
|
||||||
local function stable_sort_keys(tbl)
|
|
||||||
local keys = {}
|
|
||||||
for k in pairs(tbl) do keys[#keys + 1] = k end
|
|
||||||
table.sort(keys)
|
|
||||||
return keys
|
|
||||||
end
|
|
||||||
|
|
||||||
-- Allocate one phase's auto-reg mappings.
|
|
||||||
-- Returns (allocated_map, errors). On pool exhaustion, errors is populated and the function halts.
|
|
||||||
local function allocate_phase(phase_label, decls)
|
|
||||||
-- Deep-copy POOL into a fresh sequence table. The original `table.unpack and table.unpack(POOL) or { unpack(POOL) }`
|
|
||||||
-- idiom wraps the unpacked values in a single inner table under LuaJIT 5.1 (`table.unpack` is nil; the `or` returns one value),
|
|
||||||
-- which corrupts the pool into `{ {R_T0, R_T1, ...} }` — making `table.remove(pool, 1)` return the inner table on iteration.
|
|
||||||
local pool = {}
|
|
||||||
for i = 1, #POOL do pool[i] = POOL[i] end
|
|
||||||
local result = {}
|
|
||||||
local errors = {}
|
|
||||||
for _, sym in ipairs(stable_sort_keys(decls)) do
|
|
||||||
local next_gpr = table.remove(pool, 1)
|
|
||||||
if not next_gpr then
|
|
||||||
errors[#errors + 1] = {
|
|
||||||
line = 0,
|
|
||||||
msg = string.format("phase_register_pool_exhausted: "
|
|
||||||
.. "phase '%s' requested symbol '%s' but the pool has no remaining registers "
|
|
||||||
.. "(max 24 per phase: R_T0..R_T7 + R_V0..R_V1 + R_A0..R_A3 + R_S0..R_S7 + R_T8..R_T9). Split the phase or use hardcoded GPRs."
|
|
||||||
, phase_label, sym),
|
|
||||||
}
|
|
||||||
return result, errors
|
|
||||||
end
|
|
||||||
result[sym] = next_gpr
|
|
||||||
end
|
|
||||||
return result, errors
|
|
||||||
end
|
|
||||||
|
|
||||||
-- Build two projections from corpus.register_alias_registry:
|
|
||||||
-- user_pinned -- { [physical_gpr_ident] = true } -- GPRs unavailable to auto_reg globally (wave-context carriers, file-scope pinned aliases)
|
|
||||||
-- alias_to_gpr -- { [alias_ident] = physical_gpr_ident } -- for body parsing
|
|
||||||
-- Both projections are derived from the same set of entries: every AliasEntry in register_alias_registry has `has_atom_reg = true`
|
|
||||||
-- (only those entries are added to the registry; see passes/scan_source.lua parse_enum_entry).
|
|
||||||
-- Each entry's `code` is the integer MIPS GPR number (0..31); INT_CODE_TO_POOL_GPR translates it back to the physical GPR ident.
|
|
||||||
-- Aliases whose `code` points to a non-POOL GPR (e.g. R_S0, R_T8, R_K1) are ignored —
|
|
||||||
-- they don't affect the auto_reg pool, and they're already excluded from POOL above.
|
|
||||||
local function build_user_pins(corpus)
|
|
||||||
local user_pinned = {}
|
|
||||||
local alias_to_gpr = {}
|
|
||||||
if not corpus.register_alias_registry then return user_pinned, alias_to_gpr end
|
|
||||||
for alias_name, alias_entry in pairs(corpus.register_alias_registry) do
|
|
||||||
if alias_entry.has_atom_reg and alias_entry.code then
|
|
||||||
local gpr = INT_CODE_TO_POOL_GPR[alias_entry.code]
|
|
||||||
if gpr then
|
|
||||||
user_pinned[gpr] = true
|
|
||||||
alias_to_gpr[alias_name] = gpr
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
return user_pinned, alias_to_gpr
|
|
||||||
end
|
|
||||||
|
|
||||||
-- Find every physical GPR referenced in the atom body, via EITHER:
|
|
||||||
-- (a) A hardcoded physical GPR ident (R_T\d+|R_V\d+|R_A\d+|R_S\d+) — the existing regex;
|
|
||||||
-- (b) An alias ident (R_<Alias>) resolved via alias_to_gpr back to its physical GPR ident.
|
|
||||||
-- Returns { [physical_gpr_ident] = count }. Clash-detection and source-pool-exclusion logic
|
|
||||||
-- only needs the presence of each GPR (boolean test), but keeping count preserves the
|
|
||||||
-- original find_hardcoded_rn shape so callers can switch without churn.
|
|
||||||
-- The alias pattern is sorted lexicographically to keep the regex deterministic.
|
|
||||||
local function find_used_gprs(body_text, alias_to_gpr)
|
|
||||||
local found = {}
|
|
||||||
-- (a) Hardcoded physical GPRs (R_T0..R_T7, R_V0..R_V1, R_A0..R_A3, R_S0..R_S7).
|
|
||||||
for gpr in body_text:gmatch("(R_T%d+|R_V%d+|R_A%d+|R_S%d+)") do
|
|
||||||
found[gpr] = (found[gpr] or 0) + 1
|
|
||||||
end
|
|
||||||
-- (b) Alias references (R_<Alias>) resolved to physical GPRs via the registry.
|
|
||||||
-- Sorted by name so the regex is byte-stable across runs.
|
|
||||||
if alias_to_gpr and next(alias_to_gpr) then
|
|
||||||
local aliases = {}
|
|
||||||
for alias_name in pairs(alias_to_gpr) do
|
|
||||||
aliases[#aliases + 1] = alias_name
|
|
||||||
end
|
|
||||||
table.sort(aliases)
|
|
||||||
local pattern = "(" .. table.concat(aliases, "|") .. ")"
|
|
||||||
for alias_name in body_text:gmatch(pattern) do
|
|
||||||
local gpr = alias_to_gpr[alias_name]
|
|
||||||
if gpr and not found[gpr] then
|
|
||||||
found[gpr] = 1
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
return found
|
|
||||||
end
|
|
||||||
|
|
||||||
-- Emit one gen/auto_reg.h header per directory.
|
|
||||||
local function emit_auto_reg_h(out_dir, dir, sources, mappings)
|
|
||||||
if not mappings or next(mappings) == nil then return end
|
|
||||||
local out_path = out_dir .. "/" .. "auto_reg.h"
|
|
||||||
duffle.ensure_dir(out_dir)
|
|
||||||
local lines = {
|
|
||||||
"#ifdef INTELLISENSE_DIRECTIVES",
|
|
||||||
"#pragma once",
|
|
||||||
"#endif",
|
|
||||||
"// Auto-generated by ps1_meta.lua (passes/auto_reg.lua) — DO NOT EDIT",
|
|
||||||
"// Directory: " .. dir:gsub("/", "\\"),
|
|
||||||
}
|
|
||||||
for _, src in ipairs(sources) do
|
|
||||||
lines[#lines + 1] = "// source: " .. src.path
|
|
||||||
end
|
|
||||||
lines[#lines + 1] = "// Per-phase register allocations resolved by the lua pass."
|
|
||||||
lines[#lines + 1] = "// R_<Sym>_Code = <chosen GPR's _Code constant> for every marker in this directory."
|
|
||||||
lines[#lines + 1] = ""
|
|
||||||
for _, sym in ipairs(stable_sort_keys(mappings)) do
|
|
||||||
local gpr = mappings[sym]
|
|
||||||
local gpr_code = gpr .. "_Code"
|
|
||||||
lines[#lines + 1] = "#define " .. sym .. "_Code " .. gpr_code
|
|
||||||
end
|
|
||||||
lines[#lines + 1] = ""
|
|
||||||
duffle.write_file_lf(out_path, table.concat(lines, "\n") .. "\n")
|
|
||||||
print(" -> " .. out_path)
|
|
||||||
return out_path
|
|
||||||
end
|
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
|
||||||
-- Pass entry
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
|
||||||
|
|
||||||
local M = {}
|
|
||||||
|
|
||||||
--- @param ctx PassCtx
|
|
||||||
--- @return AutoRegResult
|
|
||||||
function M.run(ctx)
|
|
||||||
local outputs = {}
|
|
||||||
local errors = {}
|
|
||||||
local warnings = {}
|
|
||||||
|
|
||||||
local corpus = ctx.shared and ctx.shared.corpus
|
|
||||||
if type(corpus) ~= "table" then
|
|
||||||
error("auto_reg.run requires ctx.shared.corpus", 0)
|
|
||||||
end
|
|
||||||
|
|
||||||
-- 0. Build the user-pinned GPR exclusion set + alias-to-GPR resolution map.
|
|
||||||
-- Wave-context carriers (e.g. `R_ResolveScratch = R_T4 atom_reg` in hello_camera.atom.c)
|
|
||||||
-- MUST NOT be allocated to any auto-reg marker — they're preserved across atoms by the wave-context discipline.
|
|
||||||
-- The corpus's register_alias_registry is the source of truth for these opt-in pins.
|
|
||||||
-- Body references to those aliases (via alias_to_gpr) are also excluded on a per-atom basis in step 2 below.
|
|
||||||
local user_pinned, alias_to_gpr = build_user_pins(corpus)
|
|
||||||
|
|
||||||
-- 1. Allocate phase pools first (phase declarations take precedence over per-atom declarations).
|
|
||||||
local phase_allocations = {}
|
|
||||||
for phase_label, decls in pairs(corpus.phase_auto_regs or {}) do
|
|
||||||
local mapping, errs = allocate_phase(phase_label, decls)
|
|
||||||
for sym, gpr in pairs(mapping) do
|
|
||||||
phase_allocations[phase_label] = phase_allocations[phase_label] or {}
|
|
||||||
phase_allocations[phase_label][sym] = gpr
|
|
||||||
end
|
|
||||||
for _, e in ipairs(errs) do
|
|
||||||
errors[#errors + 1] = e
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
-- 2. Allocate per-atom auto-regs. If the atom scope matches a phase, reuse the phase pool.
|
|
||||||
-- Otherwise, allocate a private pool for the atom.
|
|
||||||
-- The phase membership is in `corpus.atom_phases[phase_label].atoms` (an array of atom names declared via `atom_phase(<phase>)`
|
|
||||||
-- in the atom's `atom_info` line). Build a reverse map `atom_name -> phase_label` so the lookup is O(1) per atom scope.
|
|
||||||
local atom_name_to_phase = {}
|
|
||||||
for phase_label, entry in pairs(corpus.atom_phases or {}) do
|
|
||||||
for _, atom_name in ipairs(entry.atoms or {}) do
|
|
||||||
atom_name_to_phase[atom_name] = phase_label
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
local atom_allocations = {}
|
|
||||||
for atom_scope, decls in pairs(corpus.atom_auto_regs or {}) do
|
|
||||||
local phase_label = atom_name_to_phase[atom_scope]
|
|
||||||
-- Build the atom's source pool: start with the full POOL, subtract:
|
|
||||||
-- (a) every GPR already committed (phase allocations + prior atom allocations)
|
|
||||||
-- (b) every USER-PINNED GPR (wave-context carriers + file-scope pinned aliases)
|
|
||||||
-- (c) every GPR referenced in the atom's body — either hardcoded R_X or alias R_Xxx
|
|
||||||
-- (the latter resolved via alias_to_gpr; this catches cases where the user wrote R_ResolveScratch instead of R_T4 directly)
|
|
||||||
-- Atoms whose scope matches a phase share the global pool with the phase allocations;
|
|
||||||
-- the original `source_pool = phase_allocations[phase_label]` form used the phase
|
|
||||||
-- allocation MAP as a pool, but that map has no array part, so `table.remove(source_pool, 1)`
|
|
||||||
-- returned nil and every atom-with-phase marker errored with `phase_register_pool_exhausted`.
|
|
||||||
local used = {}
|
|
||||||
for _, m in pairs(phase_allocations) do for _, gpr in pairs(m) do used[gpr] = true end end
|
|
||||||
for _, m in pairs(atom_allocations) do for _, gpr in pairs(m) do used[gpr] = true end end
|
|
||||||
-- (c) Body references — scan the atom body for hardcoded + alias-resolved GPRs.
|
|
||||||
-- Folded into `used` so the source_pool exclusion is a single check.
|
|
||||||
local atom = corpus.atoms_by_name and corpus.atoms_by_name[atom_scope]
|
|
||||||
if atom and atom.body then
|
|
||||||
local body_used = find_used_gprs(atom.body, alias_to_gpr)
|
|
||||||
for gpr in pairs(body_used) do used[gpr] = true end
|
|
||||||
end
|
|
||||||
local source_pool = {}
|
|
||||||
for _, gpr in ipairs(POOL) do
|
|
||||||
-- Exclude (a) prior commitments, (b) USER-PINNED GPRs (wave-context carriers
|
|
||||||
-- declared via atom_reg + _Code defs, preserved across atoms globally).
|
|
||||||
if not used[gpr] and not user_pinned[gpr] then
|
|
||||||
source_pool[#source_pool + 1] = gpr
|
|
||||||
end
|
|
||||||
end
|
|
||||||
local result = {}
|
|
||||||
for _, sym in ipairs(stable_sort_keys(decls)) do
|
|
||||||
local next_gpr = table.remove(source_pool, 1)
|
|
||||||
if not next_gpr then
|
|
||||||
errors[#errors + 1] = {
|
|
||||||
line = 0,
|
|
||||||
msg = string.format("phase_register_pool_exhausted: atom '%s' requested symbol '%s' "
|
|
||||||
.. "but no free registers remain in its scope pool."
|
|
||||||
, atom_scope, sym),
|
|
||||||
}
|
|
||||||
else
|
|
||||||
result[sym] = next_gpr
|
|
||||||
end
|
|
||||||
end
|
|
||||||
atom_allocations[atom_scope] = result
|
|
||||||
end
|
|
||||||
|
|
||||||
-- 3. Conflict-with-hardcoded detection (defensive — should be unreachable now).
|
|
||||||
-- The source_pool exclusion in step 2 (b) + (c) already accounts for both user-pinned GPRs
|
|
||||||
-- and body-referenced GPRs (hardcoded R_Tn OR alias R_<Alias>).
|
|
||||||
-- An auto-reg allocation that matched an existing body reference would be impossible by construction.
|
|
||||||
-- This warning is kept as a defensive safety net for cases the body scanner might miss
|
|
||||||
-- (e.g. macros that expand to register references the scanner cannot resolve).
|
|
||||||
-- For each resolved (scope, sym) -> R_Tn mapping, scan the atom body source for used GPRs.
|
|
||||||
for atom_scope, decls in pairs(atom_allocations) do
|
|
||||||
local atom = corpus.atoms_by_name and corpus.atoms_by_name[atom_scope]
|
|
||||||
if atom and atom.body then
|
|
||||||
local used_in_body = find_used_gprs(atom.body, alias_to_gpr)
|
|
||||||
for sym, allocated_gpr in pairs(decls) do
|
|
||||||
if used_in_body[allocated_gpr] and used_in_body[allocated_gpr] > 0 then
|
|
||||||
warnings[#warnings + 1] = {
|
|
||||||
line = atom.line or 0,
|
|
||||||
msg = string.format("phase_register_clash: atom '%s' has hardcoded '%s' in its body AND an auto-reg marker '%s' "
|
|
||||||
.. "that was allocated to '%s' (same phase). Resolve by removing the hardcoded reference or renaming the auto-reg."
|
|
||||||
, atom_scope, allocated_gpr, sym, allocated_gpr),
|
|
||||||
}
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
-- 4. Emit per-directory gen/auto_reg.h.
|
|
||||||
-- For each source directory that has atom_auto_regs or phase_auto_regs entries, emit one header.
|
|
||||||
local sources_by_dir = corpus.sources_by_dir or {}
|
|
||||||
for dir, sources in pairs(sources_by_dir) do
|
|
||||||
local per_dir_mappings = {}
|
|
||||||
for _, src in ipairs(sources) do
|
|
||||||
-- Collect every (sym -> gpr) entry that originated from a source in this directory.
|
|
||||||
-- `src.scan.atom_auto_regs` is keyed by ATOM SCOPE NAME; `pairs(t)` iterates KEYS so `scope_name` here is the scope ident (e.g. "cube_g4_face").
|
|
||||||
-- The previous `for _, scan_atom_auto` form silently assigned the VALUE (a `{sym = sym}` table) to the variable,
|
|
||||||
-- which made `atom_allocations[scan_atom_auto]` a table-indexed lookup that never resolved.
|
|
||||||
for scope_name in pairs(src.scan and src.scan.atom_auto_regs or {}) do
|
|
||||||
for sym, gpr in pairs(atom_allocations[scope_name] or {}) do
|
|
||||||
per_dir_mappings[sym] = gpr
|
|
||||||
end
|
|
||||||
end
|
|
||||||
for scope_name in pairs(src.scan and src.scan.phase_auto_regs or {}) do
|
|
||||||
for sym, gpr in pairs(phase_allocations[scope_name] or {}) do
|
|
||||||
per_dir_mappings[sym] = gpr
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
local out_dir = dir .. "/gen"
|
|
||||||
local out_path = emit_auto_reg_h(out_dir, dir, sources, per_dir_mappings)
|
|
||||||
if out_path then outputs[#outputs + 1] = { auto_reg_h = out_path } end
|
|
||||||
end
|
|
||||||
return { outputs = outputs, errors = errors, warnings = warnings }
|
|
||||||
end
|
|
||||||
|
|
||||||
return M
|
|
||||||
+63
-218
@@ -3,12 +3,9 @@
|
|||||||
--- Ownership: `corpus.word_counts`, `corpus.components`, and `corpus.component_body_index`.
|
--- Ownership: `corpus.word_counts`, `corpus.components`, and `corpus.component_body_index`.
|
||||||
--- Scanner owns `declaration_comment` and `debug_skip` on each declaration record; this pass projects both forward.
|
--- Scanner owns `declaration_comment` and `debug_skip` on each declaration record; this pass projects both forward.
|
||||||
---
|
---
|
||||||
--- Reads the pre-scanned SourceScan payload from `duffle.scan_source` for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations (kind="comp_bare" / "comp_proc"),
|
--- Reads the pre-scanned SourceScan payload from `duffle.scan_source` for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations,
|
||||||
--- then resolves the function-args string from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration via a backward walk.
|
--- then resolves the function-args string from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration via a backward walk.
|
||||||
---
|
---
|
||||||
--- `MipsAtom_Proc_(X, ab, { body })` declarations (kind="atom_proc") are ATOMS, not components, and are deliberately excluded —
|
|
||||||
--- the ELF symbol is the C ident. Raw `MipsCode code_*` is leftover, not the atom rule.
|
|
||||||
---
|
|
||||||
--- Emits one `gen/macs.h` per *immediate source directory* with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
|
--- Emits one `gen/macs.h` per *immediate source directory* with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
|
||||||
--- All sources inside the same directory contribute to the same file (per-directory aggregation).
|
--- All sources inside the same directory contribute to the same file (per-directory aggregation).
|
||||||
--- The directory itself is the namespace, so the filename does not repeat the module name.
|
--- The directory itself is the namespace, so the filename does not repeat the module name.
|
||||||
@@ -79,7 +76,7 @@ local MACS_FILENAME = "macs.h"
|
|||||||
--- @field args string|nil -- Function-args string (function form only)
|
--- @field args string|nil -- Function-args string (function form only)
|
||||||
--- @field line integer -- Source line of the declaration
|
--- @field line integer -- Source line of the declaration
|
||||||
--- @field comment string|nil -- Scanner-owned `declaration_comment`; the components pass reads it from the scanner record
|
--- @field comment string|nil -- Scanner-owned `declaration_comment`; the components pass reads it from the scanner record
|
||||||
--- @field kind string -- "comp_bare" | "comp_proc" (atom_proc is NOT a component — see `project_components`)
|
--- @field kind string -- "comp_bare" | "comp_proc"
|
||||||
--- @field debug_skip boolean -- Mirror of `a.debug_skip` (scanner-owned); true iff a bare `atom_dbg_skip` marker immediately preceded the declaration
|
--- @field debug_skip boolean -- Mirror of `a.debug_skip` (scanner-owned); true iff a bare `atom_dbg_skip` marker immediately preceded the declaration
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
@@ -96,21 +93,48 @@ local M = {}
|
|||||||
-- so this file reads it forward rather than re-walking the source.
|
-- so this file reads it forward rather than re-walking the source.
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
--- Find the args of the function declaration that immediately precedes a `MipsAtomComp_Proc_` invocation.
|
--- Find the args of the function declaration that immediately precedes a `MipsAtomComp_Proc_` invocation of the given name.
|
||||||
--- Returns the args string (e.g., `"U4 off, U4 code, U1 r, U1 g, U1 b"`) or nil if no function declaration is found.
|
--- Returns the args string (e.g., `"U4 off, U4 code, U1 r, U1 g, U1 b"`) or nil if no function declaration is found.
|
||||||
---
|
---
|
||||||
--- After the `sym` arg was dropped from MipsAtomComp_Proc_, the component name
|
--- Convention: function form is
|
||||||
--- and the args both come from the preceding `FI_ Slice_MipsCode ac_X(args)`
|
--- `FI_ Slice_MipsCode ac_X(args) MipsAtomComp_Proc_(ac_X, { body })`
|
||||||
--- declaration. The shared `duffle.find_function_decl_for` helper does the
|
--- We find the LAST occurrence of `"ac_X("` before `before_pos` and extract the args from inside the parens.
|
||||||
--- backward walk; this function returns just the args.
|
--- We then verify the preceding context ends with `Slice_MipsCode`
|
||||||
|
--- (the function-decl keyword with possible qualifiers between).
|
||||||
---
|
---
|
||||||
--- @param source string
|
--- @param source string
|
||||||
--- @param name string (retained for signature stability; unused — the walk derives the name)
|
--- @param name string
|
||||||
--- @param before_pos integer
|
--- @param before_pos integer
|
||||||
--- @return string|nil
|
--- @return string|nil
|
||||||
local function find_function_args_for(source, name, before_pos)
|
local function find_function_args_for(source, name, before_pos)
|
||||||
local _, args_inner = duffle.find_function_decl_for(source, before_pos, #MIPS_ATOM)
|
-- Find the LAST occurrence of `name + "("` in `source[1..before_pos]`.
|
||||||
return args_inner
|
local name_open = name .. "("
|
||||||
|
local last_idx = nil
|
||||||
|
local scan_pos = 1
|
||||||
|
while true do
|
||||||
|
-- Pass `before_pos + 1` so string.find only returns positions < before_pos + 1
|
||||||
|
-- (string.find's 4th arg `plain` is true; we use the 3rd arg `init` for the upper bound).
|
||||||
|
local found = source:find(name_open, scan_pos, true)
|
||||||
|
if not found or found >= before_pos then break end
|
||||||
|
last_idx = found
|
||||||
|
scan_pos = found + #name_open
|
||||||
|
end
|
||||||
|
if not last_idx then return nil end
|
||||||
|
|
||||||
|
-- Verify the preceding context ends with "MipsAtom" (with possible qualifiers between).
|
||||||
|
local before = source:sub(1, last_idx - 1)
|
||||||
|
local trimmed = duffle.trim(before)
|
||||||
|
if trimmed:sub(-#MIPS_ATOM) ~= MIPS_ATOM then
|
||||||
|
-- Preceding context is not a function declaration.
|
||||||
|
return nil
|
||||||
|
end
|
||||||
|
|
||||||
|
local open_paren = last_idx + #name -- position of "("
|
||||||
|
-- scan: MipsAtom ac_X(
|
||||||
|
local inner = duffle.read_parens(source, open_paren)
|
||||||
|
-- scan: MipsAtom ac_X(<args>)
|
||||||
|
if not inner then return nil end
|
||||||
|
return inner
|
||||||
end
|
end
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
@@ -130,60 +154,6 @@ local function extract_arg_names(args_str)
|
|||||||
for _, tok in ipairs(tokens) do
|
for _, tok in ipairs(tokens) do
|
||||||
local trimmed = duffle.trim(tok)
|
local trimmed = duffle.trim(tok)
|
||||||
if trimmed ~= "" then
|
if trimmed ~= "" then
|
||||||
-- Strip trailing block comment (/* ... */) from the token, if present.
|
|
||||||
-- split_top_level_commas only skips block comments at TOP LEVEL (between commas),
|
|
||||||
-- not block comments embedded WITHIN a token between a parameter and a trailing comma.
|
|
||||||
-- Without this strip, the identifier-walk below stops at the `/` of `*/` and returns
|
|
||||||
-- the wrong name (or nothing). See `test_extract_arg_names_handles_trailing_block_comments`.
|
|
||||||
local trimmed_end = #trimmed
|
|
||||||
if trimmed_end >= 2 and trimmed:sub(trimmed_end - 1, trimmed_end) == "*/" then
|
|
||||||
-- Find the matching `/*` that opens the trailing comment.
|
|
||||||
-- Walk back from the `*/` looking for `/*` (whitespace + `/*`).
|
|
||||||
local close_pos = trimmed_end - 1 -- position of the second-to-last char
|
|
||||||
-- Walk back: skip trailing whitespace, then look for the `/*` opener.
|
|
||||||
while close_pos > 1 do
|
|
||||||
local ch = trimmed:sub(close_pos, close_pos)
|
|
||||||
if ch == " " or ch == "\t" or ch == "\n" or ch == "\r" then
|
|
||||||
close_pos = close_pos - 1
|
|
||||||
else
|
|
||||||
break
|
|
||||||
end
|
|
||||||
end
|
|
||||||
-- Now scan back from close_pos for the `/*` opener (slashes are at close_pos-1 and close_pos-2).
|
|
||||||
local opener_pos = nil
|
|
||||||
local scan = close_pos - 3
|
|
||||||
while scan >= 1 do
|
|
||||||
if trimmed:sub(scan, scan + 1) == "/*" then
|
|
||||||
opener_pos = scan
|
|
||||||
break
|
|
||||||
end
|
|
||||||
scan = scan - 1
|
|
||||||
end
|
|
||||||
if opener_pos then
|
|
||||||
-- Truncate everything from opener_pos onwards.
|
|
||||||
trimmed = duffle.trim(trimmed:sub(1, opener_pos - 1))
|
|
||||||
end
|
|
||||||
end
|
|
||||||
if trimmed == "" then goto continue end
|
|
||||||
-- Strip trailing array suffix `[N]` if present.
|
|
||||||
-- Example: `Reg r_data[4]` → identifier is `r_data`, not `4`.
|
|
||||||
trimmed_end = #trimmed
|
|
||||||
if trimmed_end >= 4 and trimmed:sub(trimmed_end, trimmed_end) == "]" then
|
|
||||||
-- Walk back: skip digits, expect `[`.
|
|
||||||
local bracket_pos = trimmed_end - 1
|
|
||||||
while bracket_pos > 1 do
|
|
||||||
local ch = trimmed:sub(bracket_pos, bracket_pos)
|
|
||||||
if ch >= "0" and ch <= "9" then
|
|
||||||
bracket_pos = bracket_pos - 1
|
|
||||||
else
|
|
||||||
break
|
|
||||||
end
|
|
||||||
end
|
|
||||||
if bracket_pos >= 1 and trimmed:sub(bracket_pos, bracket_pos) == "[" then
|
|
||||||
trimmed = duffle.trim(trimmed:sub(1, bracket_pos - 1))
|
|
||||||
end
|
|
||||||
end
|
|
||||||
if trimmed == "" then goto continue end
|
|
||||||
-- Find the identifier at the end: walk back over trailers (whitespace + `*` + `[]`),
|
-- Find the identifier at the end: walk back over trailers (whitespace + `*` + `[]`),
|
||||||
-- then walk back over the identifier chars (alnum + `_`).
|
-- then walk back over the identifier chars (alnum + `_`).
|
||||||
local ident_end = #trimmed
|
local ident_end = #trimmed
|
||||||
@@ -207,21 +177,12 @@ local function extract_arg_names(args_str)
|
|||||||
ident_start = ident_start + 1
|
ident_start = ident_start + 1
|
||||||
local name = trimmed:sub(ident_start, ident_end)
|
local name = trimmed:sub(ident_start, ident_end)
|
||||||
if name ~= "" then names[#names + 1] = name end
|
if name ~= "" then names[#names + 1] = name end
|
||||||
::continue::
|
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
if #names == 0 then return nil end
|
if #names == 0 then return nil end
|
||||||
return names
|
return names
|
||||||
end
|
end
|
||||||
|
|
||||||
local function formal_arg_names(args_str)
|
|
||||||
local names = extract_arg_names(args_str)
|
|
||||||
if not names then return nil end
|
|
||||||
if names[1] == "ab" then table.remove(names, 1) end
|
|
||||||
if #names == 0 then return nil end
|
|
||||||
return names
|
|
||||||
end
|
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- Component projection (read from pre-scanned SourceScan)
|
-- Component projection (read from pre-scanned SourceScan)
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
@@ -239,16 +200,7 @@ end
|
|||||||
local function project_components(source, scan)
|
local function project_components(source, scan)
|
||||||
local out = {}
|
local out = {}
|
||||||
for _, a in ipairs(scan.atoms) do
|
for _, a in ipairs(scan.atoms) do
|
||||||
-- Only `MipsAtomComp_(ac_X)` (kind="comp_bare") and `MipsAtomComp_Proc_(ac_X, ...)` (kind="comp_proc")
|
|
||||||
-- are COMPONENTS — they get inlined via `mac_<name>` aliases inside atom bodies.
|
|
||||||
-- `MipsAtom_Proc_` (kind="atom_proc") is an ATOM (ends with `mac_yield()`); it gets emitted via
|
|
||||||
-- `tb_emit` of the C ident, NOT inlined as a macro. Including `atom_proc` here
|
|
||||||
-- would incorrectly emit `mac_<name>` aliases for atoms, polluting `gen/macs.h`.
|
|
||||||
-- See `docs/duffle_dsl_primer.md` §"mac_* aliases" for the contract.
|
|
||||||
if a.kind == "comp_bare" or a.kind == "comp_proc" then
|
if a.kind == "comp_bare" or a.kind == "comp_proc" then
|
||||||
-- Function-args lookup is meaningful for `MipsAtomComp_Proc_` components
|
|
||||||
-- (the macro sits inside `FI_ Slice_MipsCode ac_X(...)`); the alias expansion
|
|
||||||
-- discards the `ab` (atom-builder) arg the same way both forms do.
|
|
||||||
local args = find_function_args_for(source, a.raw_name, a.ident_pos)
|
local args = find_function_args_for(source, a.raw_name, a.ident_pos)
|
||||||
-- Comment ownership: scan_source.lua stamps `declaration_comment` on the record by walking backward past any associated bare marker.
|
-- Comment ownership: scan_source.lua stamps `declaration_comment` on the record by walking backward past any associated bare marker.
|
||||||
-- The pass reads `declaration_comment` directly.
|
-- The pass reads `declaration_comment` directly.
|
||||||
@@ -260,7 +212,6 @@ local function project_components(source, scan)
|
|||||||
body_off = a.body_off,
|
body_off = a.body_off,
|
||||||
body_tokens = a.body_tokens,
|
body_tokens = a.body_tokens,
|
||||||
args = args,
|
args = args,
|
||||||
arg_names = formal_arg_names(args),
|
|
||||||
comment = comment,
|
comment = comment,
|
||||||
kind = a.kind, -- "comp_bare" | "comp_proc"; provenance emitter reads this.
|
kind = a.kind, -- "comp_bare" | "comp_proc"; provenance emitter reads this.
|
||||||
debug_skip = a.debug_skip == true,
|
debug_skip = a.debug_skip == true,
|
||||||
@@ -348,9 +299,7 @@ local function word_count_rec(name, comp_by_name, wc, cache)
|
|||||||
local trimmed = t.tok
|
local trimmed = t.tok
|
||||||
if trimmed ~= "" then
|
if trimmed ~= "" then
|
||||||
local lookup = strip_mac_prefix(duffle.read_ident(trimmed, 1))
|
local lookup = strip_mac_prefix(duffle.read_ident(trimmed, 1))
|
||||||
if lookup == "atom_label" or lookup == "atom_offset" then
|
if lookup and comp_by_name[lookup] then
|
||||||
-- Pure metaprogram anchors; emit zero words.
|
|
||||||
elseif lookup and comp_by_name[lookup] then
|
|
||||||
-- It's a `mac_X(...)` call. Recurse.
|
-- It's a `mac_X(...)` call. Recurse.
|
||||||
n = n + word_count_rec(lookup, comp_by_name, wc, cache)
|
n = n + word_count_rec(lookup, comp_by_name, wc, cache)
|
||||||
elseif lookup and wc and wc[lookup] then
|
elseif lookup and wc and wc[lookup] then
|
||||||
@@ -427,10 +376,8 @@ local function cycle_cost_rec(name, comp_by_name, latency, cache)
|
|||||||
local nested = ident:sub(MAC_PREFIX_LEN + 1)
|
local nested = ident:sub(MAC_PREFIX_LEN + 1)
|
||||||
n = n + cycle_cost_rec(nested, comp_by_name, latency, cache)
|
n = n + cycle_cost_rec(nested, comp_by_name, latency, cache)
|
||||||
else
|
else
|
||||||
-- Leaf instruction or pseudo-macro.
|
-- Leaf instruction or pseudo-macro. Look up in INSTRUCTION_LATENCY; default 1.
|
||||||
local isa = duffle.instr(ident)
|
n = n + (latency[ident] or 1)
|
||||||
local gte = duffle.gte(ident)
|
|
||||||
n = n + ((isa and isa.cycles) or (gte and gte.cycles) or latency[ident] or 1)
|
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
@@ -443,17 +390,14 @@ local function cycle_cost_rec(name, comp_by_name, latency, cache)
|
|||||||
end
|
end
|
||||||
|
|
||||||
--- (internal) Recursive GP0 prim-buffer contribution. Count `store_word` / `store_half` / `store_byte`
|
--- (internal) Recursive GP0 prim-buffer contribution. Count `store_word` / `store_half` / `store_byte`
|
||||||
--- calls in the component body that target `R_PrimCursor` (these are the RAM-side prim-buffer words the macro contributes), recursing through nested `mac_*` calls.
|
--- calls in the component body that target `R_PrimCursor` (these are the
|
||||||
|
--- RAM-side prim-buffer words the macro contributes), recursing through nested `mac_*` calls.
|
||||||
--- Only `R_PrimCursor`-targeting stores count. Stores targeting other registers (e.g. `R_OtBase`, heap pointers) are not prim-buffer contributions.
|
--- Only `R_PrimCursor`-targeting stores count. Stores targeting other registers (e.g. `R_OtBase`, heap pointers) are not prim-buffer contributions.
|
||||||
--- @param name string
|
--- @param name string
|
||||||
--- @param comp_by_name table<string, Component>
|
--- @param comp_by_name table<string, Component>
|
||||||
--- @param cache table<string, integer>
|
--- @param cache table<string, integer>
|
||||||
--- @return integer
|
--- @return integer
|
||||||
local function gp0_contrib_rec(name, comp_by_name, cache)
|
local function gp0_contrib_rec(name, comp_by_name, cache)
|
||||||
if name:match("^insert_ot_tag") then
|
|
||||||
cache[name] = 0
|
|
||||||
return 0
|
|
||||||
end
|
|
||||||
if cache[name] ~= nil then return cache[name] end
|
if cache[name] ~= nil then return cache[name] end
|
||||||
cache[name] = -1
|
cache[name] = -1
|
||||||
local cc = comp_by_name[name]
|
local cc = comp_by_name[name]
|
||||||
@@ -469,15 +413,8 @@ local function gp0_contrib_rec(name, comp_by_name, cache)
|
|||||||
-- Nested `mac_X(...)` call: recurse.
|
-- Nested `mac_X(...)` call: recurse.
|
||||||
local nested = ident:sub(MAC_PREFIX_LEN + 1)
|
local nested = ident:sub(MAC_PREFIX_LEN + 1)
|
||||||
n = n + gp0_contrib_rec(nested, comp_by_name, cache)
|
n = n + gp0_contrib_rec(nested, comp_by_name, cache)
|
||||||
elseif ident == "gte_sw" then
|
|
||||||
n = n + 1
|
|
||||||
elseif ident == "store_word" or ident == "store_half" or ident == "store_byte" then
|
elseif ident == "store_word" or ident == "store_half" or ident == "store_byte" then
|
||||||
if trimmed:find("R_PrimCursor", 1, true)
|
if trimmed:find("R_PrimCursor", 1, true) then
|
||||||
or trimmed:find("O_(Poly_", 1, true)
|
|
||||||
or trimmed:find("r_prim_cursor", 1, true)
|
|
||||||
or trimmed:find("r_primitive_cursor", 1, true)
|
|
||||||
or trimmed:find("r_base", 1, true)
|
|
||||||
then
|
|
||||||
n = n + 1
|
n = n + 1
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
@@ -536,16 +473,12 @@ local function split_comment_lines(s)
|
|||||||
end
|
end
|
||||||
|
|
||||||
--- Determine the macro signature: function-args list (function form) or variadic-ignored (bare form).
|
--- Determine the macro signature: function-args list (function form) or variadic-ignored (bare form).
|
||||||
--- For `MipsAtomComp_Proc_` components, the leading `ab` (atom-builder) arg is dropped:
|
|
||||||
--- the generated `mac_<name>` macros are inline-expansion aliases for baked atoms; their bodies don't reference `ab`
|
|
||||||
--- (the builder is only consumed by the procedural `atombuilder_unroll` line that `MipsAtomComp_Proc_` appends after the body).
|
|
||||||
--- Inline callers therefore don't need to thread a builder context.
|
|
||||||
--- @param args_str string|nil
|
--- @param args_str string|nil
|
||||||
--- @return string
|
--- @return string
|
||||||
local function signature_from_args(args_str)
|
local function signature_from_args(args_str)
|
||||||
local names = formal_arg_names(args_str)
|
local arg_names = extract_arg_names(args_str)
|
||||||
if names then
|
if arg_names and #arg_names > 0 then
|
||||||
return table.concat(names, ", ")
|
return table.concat(arg_names, ", ")
|
||||||
end
|
end
|
||||||
return "..."
|
return "..."
|
||||||
end
|
end
|
||||||
@@ -559,103 +492,16 @@ local function strip_trailing_continuation(lines)
|
|||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Classify a token as a "pure delay marker token" (a delay-marker identifier
|
|
||||||
--- with no following instruction — only whitespace and/or block comments).
|
|
||||||
--- Examples that match:
|
|
||||||
--- * `GteDelay_` → marker alone
|
|
||||||
--- * `GteDelay_ /* RT diagonal: D1 = a.x... */` → marker + block comment
|
|
||||||
--- * `GteDelay_ /* RT diagonal: ... */\n\t` → marker + comment + trailing whitespace
|
|
||||||
--- Examples that DO NOT match (these contain a real instruction after the marker
|
|
||||||
--- and must be preserved verbatim so the instruction still gets emitted):
|
|
||||||
--- * `GteDelay_ nop2`
|
|
||||||
--- * `GteDelay_ add_si(r.dst_ptr, r.scratch, dst_offset)`
|
|
||||||
---
|
|
||||||
--- Why this classification matters: the metaprogram emits tokens separated by `,`
|
|
||||||
--- and joins them with `\<newline>` line continuations. After C preprocessor
|
|
||||||
--- phase 2 (line splicing), the macro body collapses to a single logical line.
|
|
||||||
--- Each delay-marker identifier expands to empty (its definition
|
|
||||||
--- `#define GteDelay_ // ...` consumes the `//` line comment during preprocessing
|
|
||||||
--- of the definition itself, leaving an empty replacement list). When a token
|
|
||||||
--- is purely a delay marker with only a trailing comment, the `,` the metaprogram
|
|
||||||
--- normally adds before each token-after-the-first brackets empty content and
|
|
||||||
--- produces the syntax error `,,` (`expected expression before ',' token`) at
|
|
||||||
--- C compile. The metaprogram therefore emits such tokens WITHOUT the leading
|
|
||||||
--- `,` (see `token_skips_leading_comma`) — but the marker + trailing comment
|
|
||||||
--- are still emitted verbatim so the annotation is preserved in `gen/macs.h`.
|
|
||||||
--- @param tok string -- a single token from split_top_level_commas (already trimmed at the start, may contain trailing whitespace + block comment)
|
|
||||||
--- @return boolean
|
|
||||||
local function is_pure_delay_marker_token(tok)
|
|
||||||
local markers = duffle.DELAY_MARKERS
|
|
||||||
if type(markers) ~= "table" then return false end
|
|
||||||
|
|
||||||
-- Identify a leading delay-marker identifier (e.g. `GteDelay_`).
|
|
||||||
local ident_end = 1
|
|
||||||
while ident_end <= #tok do
|
|
||||||
local ch = tok:sub(ident_end, ident_end)
|
|
||||||
if ch:match("[%w_]") then
|
|
||||||
ident_end = ident_end + 1
|
|
||||||
else
|
|
||||||
break
|
|
||||||
end
|
|
||||||
end
|
|
||||||
local ident = tok:sub(1, ident_end - 1)
|
|
||||||
if not markers[ident] then return false end
|
|
||||||
|
|
||||||
-- Walk the remainder: only whitespace and block comments are allowed.
|
|
||||||
local scan = ident_end
|
|
||||||
while scan <= #tok do
|
|
||||||
local ch = tok:sub(scan, scan)
|
|
||||||
if ch:match("%s") then
|
|
||||||
scan = scan + 1
|
|
||||||
elseif ch == "/" and tok:sub(scan + 1, scan + 1) == "*" then
|
|
||||||
local close = tok:find("*/", scan + 2, true)
|
|
||||||
if not close then return false end
|
|
||||||
scan = close + 2
|
|
||||||
else
|
|
||||||
-- Non-whitespace, non-block-comment content: a real instruction
|
|
||||||
-- follows the marker (e.g. `GteDelay_ nop2`); keep this token intact.
|
|
||||||
return false
|
|
||||||
end
|
|
||||||
end
|
|
||||||
return true
|
|
||||||
end
|
|
||||||
|
|
||||||
--- Classify a token's "leading comma requirement".
|
|
||||||
--- Pure delay-marker tokens (`GteDelay_` / `LdSlot_` / `BdSlot_` / `DmaSlot_`
|
|
||||||
--- followed by whitespace + optional block comment and NOTHING ELSE) expand
|
|
||||||
--- to empty at C preprocessor time. Emitting them WITHOUT the leading `,`
|
|
||||||
--- separator that the metaprogram normally adds before each token after the
|
|
||||||
--- first keeps exactly one `,` between the surrounding real expressions in
|
|
||||||
--- the spliced macro body:
|
|
||||||
---
|
|
||||||
--- * before this rule: `<tok1> ,\t<gdelay> ,\t<tok3>` → after expansion
|
|
||||||
--- `<tok1> , /* comment */ , <tok3>` → `,,` syntax error.
|
|
||||||
--- * after this rule: `<tok1> \t<gdelay> ,\t<tok3>` → after expansion
|
|
||||||
--- `<tok1> /* comment */ , <tok3>` → `<tok1>, <tok3>` — valid.
|
|
||||||
---
|
|
||||||
--- Tokens like `GteDelay_ nop2` keep the leading `,` (the marker is followed
|
|
||||||
--- by a real instruction, so the marker + instruction together need the
|
|
||||||
--- separator on the LEFT to land between two real expressions).
|
|
||||||
--- @param tok string
|
|
||||||
--- @return boolean -- true if the token needs NO leading `,` separator.
|
|
||||||
local function token_skips_leading_comma(tok)
|
|
||||||
return is_pure_delay_marker_token(tok)
|
|
||||||
end
|
|
||||||
|
|
||||||
--- Emit the `#define mac_X(sig) \<newline>\t<tok1> \<newline>,\t<tok2> ...` block.
|
--- Emit the `#define mac_X(sig) \<newline>\t<tok1> \<newline>,\t<tok2> ...` block.
|
||||||
--- Converts `//` line comments to `/* */` block comments in each token so they don't break the C macro `\` line continuations.
|
--- Converts `//` line comments to `/* */` block comments in each token so they don't break the C macro `\` line continuations.
|
||||||
---
|
|
||||||
--- Pure delay-marker tokens (`GteDelay_` / `LdSlot_` / `BdSlot_` / `DmaSlot_` with only a trailing block comment, no real instruction) are emitted WITHOUT a leading `,` separator; the annotation IS preserved in the generated header (so the comment + marker remain visible to anyone reading `gen/macs.h`), but the C preprocessor expands the marker to empty, so leaving the `,` separator out is what stops the `,,` syntax error. See `token_skips_leading_comma` for the contract.
|
|
||||||
local function emit_macro_body(lines, c, sig, tokens)
|
local function emit_macro_body(lines, c, sig, tokens)
|
||||||
for tok_idx = 1, #tokens do
|
for tok_idx = 1, #tokens do
|
||||||
tokens[tok_idx] = convert_line_comments_to_block(tokens[tok_idx])
|
tokens[tok_idx] = convert_line_comments_to_block(tokens[tok_idx])
|
||||||
end
|
end
|
||||||
if #tokens == 0 then return end
|
|
||||||
lines[#lines + 1] = "#define mac_" .. c.name .. "(" .. sig .. ") \\"
|
lines[#lines + 1] = "#define mac_" .. c.name .. "(" .. sig .. ") \\"
|
||||||
lines[#lines + 1] = "\t" .. tokens[1] .. " \\"
|
lines[#lines + 1] = "\t" .. tokens[1] .. " \\"
|
||||||
for tok_idx = 2, #tokens do
|
for tok_idx = 2, #tokens do
|
||||||
local sep = token_skips_leading_comma(tokens[tok_idx]) and "\t" or ",\t"
|
lines[#lines + 1] = ",\t" .. tokens[tok_idx] .. " \\"
|
||||||
lines[#lines + 1] = sep .. tokens[tok_idx] .. " \\"
|
|
||||||
end
|
end
|
||||||
strip_trailing_continuation(lines)
|
strip_trailing_continuation(lines)
|
||||||
end
|
end
|
||||||
@@ -674,7 +520,7 @@ local function build_component_lines(c, counts)
|
|||||||
|
|
||||||
-- Marker comment: emitted once for every skipped component.
|
-- Marker comment: emitted once for every skipped component.
|
||||||
-- The marker is scanner-owned (declared by `atom_dbg_skip` immediately before the declaration in the source);
|
-- The marker is scanner-owned (declared by `atom_dbg_skip` immediately before the declaration in the source);
|
||||||
-- This pass projects `c.debug_skip` and emits the marker as a generated comment.
|
-- the components pass projects `c.debug_skip` and emits the marker as a generated comment.
|
||||||
if c.debug_skip then
|
if c.debug_skip then
|
||||||
lines[#lines + 1] = "/* atom_dbg_skip */"
|
lines[#lines + 1] = "/* atom_dbg_skip */"
|
||||||
end
|
end
|
||||||
@@ -708,8 +554,8 @@ end
|
|||||||
|
|
||||||
--- Build the boilerplate header lines (the `#ifdef INTELLISENSE_DIRECTIVES` block,
|
--- Build the boilerplate header lines (the `#ifdef INTELLISENSE_DIRECTIVES` block,
|
||||||
--- the `// Auto-generated` comment, the `// Source:` line, and the self-contained `WORD_COUNT` macro definition).
|
--- the `// Auto-generated` comment, the `// Source:` line, and the self-contained `WORD_COUNT` macro definition).
|
||||||
--- @param dir string -- Absolute source directory
|
--- @param dir string -- the absolute source directory
|
||||||
--- @param sources SourceFile[] -- Sources contributing to this directory (for the header comment)
|
--- @param sources SourceFile[] -- sources contributing to this directory (for the header comment)
|
||||||
--- @return string[]
|
--- @return string[]
|
||||||
local function header_boilerplate(dir, sources)
|
local function header_boilerplate(dir, sources)
|
||||||
local source_lines = { "// Directory: " .. duffle.to_absolute_path(dir) .. "/" }
|
local source_lines = { "// Directory: " .. duffle.to_absolute_path(dir) .. "/" }
|
||||||
@@ -740,9 +586,9 @@ end
|
|||||||
--- Compute the per-directory output path for `.macs.h`.
|
--- Compute the per-directory output path for `.macs.h`.
|
||||||
--- e.g. any source in `code/duffle/` produces `code/duffle/gen/macs.h` regardless of source filename.
|
--- e.g. any source in `code/duffle/` produces `code/duffle/gen/macs.h` regardless of source filename.
|
||||||
--- The directory name is the namespace; the filename does not repeat it.
|
--- The directory name is the namespace; the filename does not repeat it.
|
||||||
--- @param dir string -- Absolute source directory
|
--- @param dir string -- the absolute source directory
|
||||||
--- @return string -- Output directory
|
--- @return string -- the output directory
|
||||||
--- @return string -- Full output path
|
--- @return string -- the full output path
|
||||||
local function compute_macs_h_path(dir)
|
local function compute_macs_h_path(dir)
|
||||||
local out_dir = dir .. "/" .. GEN_SUBDIR
|
local out_dir = dir .. "/" .. GEN_SUBDIR
|
||||||
local out_path = out_dir .. "/" .. MACS_FILENAME
|
local out_path = out_dir .. "/" .. MACS_FILENAME
|
||||||
@@ -752,11 +598,11 @@ end
|
|||||||
--- Emit a per-directory `.macs.h` header with the aggregated `mac_X` macros + `WORD_COUNT` entries.
|
--- Emit a per-directory `.macs.h` header with the aggregated `mac_X` macros + `WORD_COUNT` entries.
|
||||||
--- Writes in BINARY mode so LF line endings are preserved (the git blob is LF; Windows text-mode would emit CRLF and break the byte-identical diff).
|
--- Writes in BINARY mode so LF line endings are preserved (the git blob is LF; Windows text-mode would emit CRLF and break the byte-identical diff).
|
||||||
--- @param ctx PassCtx
|
--- @param ctx PassCtx
|
||||||
--- @param dir string -- Absolute source directory
|
--- @param dir string -- the absolute source directory
|
||||||
--- @param sources SourceFile[] -- Sources contributing to this directory (for the header comment)
|
--- @param sources SourceFile[] -- sources contributing to this directory (for the header comment)
|
||||||
--- @param components Component[] -- Aggregated components from all sources in this directory
|
--- @param components Component[] -- aggregated components from all sources in this directory
|
||||||
--- @param counts table<string, integer> -- Precomputed word counts (from count_all_components)
|
--- @param counts table<string, integer> -- precomputed word counts (from count_all_components)
|
||||||
--- @return string|nil -- Path to the written file (nil if no components)
|
--- @return string|nil -- path to the written file (nil if no components)
|
||||||
local function emit_component_macros_h(ctx, dir, sources, components, counts)
|
local function emit_component_macros_h(ctx, dir, sources, components, counts)
|
||||||
if #components == 0 then return nil end
|
if #components == 0 then return nil end
|
||||||
local out_dir, out_path = compute_macs_h_path(dir)
|
local out_dir, out_path = compute_macs_h_path(dir)
|
||||||
@@ -795,11 +641,11 @@ local function update_canonical_word_counts(corpus, components, counts)
|
|||||||
end
|
end
|
||||||
|
|
||||||
--- @class ComponentDef
|
--- @class ComponentDef
|
||||||
--- @field name string -- Bare name (without ac_/mac_ prefix)
|
--- @field name string -- bare name (without ac_/mac_ prefix)
|
||||||
--- @field line integer -- Definition source line (line of `MipsAtomComp_(ac_X)` / `MipsAtomComp_Proc_(ac_X, ...)`)
|
--- @field line integer -- definition source line (line of `MipsAtomComp_(ac_X)` / `MipsAtomComp_Proc_(ac_X, ...)`)
|
||||||
--- @field path string -- Absolute source path of the definition
|
--- @field path string -- absolute source path of the definition
|
||||||
--- @field kind string -- "comp_bare" | "comp_proc" (atom_proc is NOT a component)
|
--- @field kind string -- "comp_bare" | "comp_proc"
|
||||||
--- @field debug_skip boolean -- Mirror of the scanner-owned `a.debug_skip`; consumers read this directly
|
--- @field debug_skip boolean -- mirror of the scanner-owned `a.debug_skip`; consumers read this directly
|
||||||
|
|
||||||
--- (internal) Populate `corpus.components` with this source's components-by-name map.
|
--- (internal) Populate `corpus.components` with this source's components-by-name map.
|
||||||
--- First declaration wins; later declarations of the same bare name are dropped and recorded as a collision via `corpus.collisions` (kind = "component").
|
--- First declaration wins; later declarations of the same bare name are dropped and recorded as a collision via `corpus.collisions` (kind = "component").
|
||||||
@@ -865,7 +711,6 @@ local function update_canonical_component_body_index(corpus, src, components, sc
|
|||||||
source = src.path,
|
source = src.path,
|
||||||
declaration = c.line,
|
declaration = c.line,
|
||||||
kind = c.kind,
|
kind = c.kind,
|
||||||
arg_names = c.arg_names,
|
|
||||||
}
|
}
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|||||||
@@ -2,7 +2,7 @@
|
|||||||
---
|
---
|
||||||
--- Reads the post-link ELF directly (io.open; walks the ELF32 section header table to find
|
--- Reads the post-link ELF directly (io.open; walks the ELF32 section header table to find
|
||||||
--- `.debug_info` + `.debug_abbrev` + `.debug_str` + `.debug_line` + `.debug_aranges` + `.debug_rnglists`),
|
--- `.debug_info` + `.debug_abbrev` + `.debug_str` + `.debug_line` + `.debug_aranges` + `.debug_rnglists`),
|
||||||
--- APPENDS synthetic DWARF line-program sequences for every tape atom, EXTENDS the `.debug_aranges`
|
--- APPENDS synthetic DWARF line-program sequences for every `code_<name>` atom, EXTENDS the `.debug_aranges`
|
||||||
--- and main-CU range tables with the atom ranges, and INSERTS synthetic atom/component DIE children into the
|
--- and main-CU range tables with the atom ranges, and INSERTS synthetic atom/component DIE children into the
|
||||||
--- existing main compilation unit in `.debug_info` (no second compilation unit).
|
--- existing main compilation unit in `.debug_info` (no second compilation unit).
|
||||||
--- Per-atom `DW_TAG_subprogram` + per-register `DW_TAG_variable` entries make
|
--- Per-atom `DW_TAG_subprogram` + per-register `DW_TAG_variable` entries make
|
||||||
@@ -703,9 +703,9 @@ end
|
|||||||
--- `{comp_name, call_file, call_line, comp_file, comp_line, start_pos, end_pos, body_lines, debug_skip}`. `body_lines[k]`
|
--- `{comp_name, call_file, call_line, comp_file, comp_line, start_pos, end_pos, body_lines, debug_skip}`. `body_lines[k]`
|
||||||
--- is the k-th word's source line within the component body.
|
--- is the k-th word's source line within the component body.
|
||||||
---
|
---
|
||||||
--- @param corpus table -- From `ctx.shared.corpus`
|
--- @param corpus table -- the corpus from `ctx.shared.corpus`
|
||||||
--- @param addrs table -- ELF symbols keyed by atom name from `elf_dwarf.read_nm`
|
--- @param addrs table -- ELF symbols keyed by atom name from `elf_dwarf.read_nm`
|
||||||
--- @return table[] -- List of {name, addr, size_bytes, words, entries, invocations, debug_skip?}
|
--- @return table[] -- list of {name, addr, size_bytes, words, entries, invocations, debug_skip?}
|
||||||
local function build_atom_table(corpus, addrs)
|
local function build_atom_table(corpus, addrs)
|
||||||
-- Cross-ref: keep only atoms present in BOTH the nm symbol table AND `corpus.atoms_by_name`. Output is sorted by ascending addr.
|
-- Cross-ref: keep only atoms present in BOTH the nm symbol table AND `corpus.atoms_by_name`. Output is sorted by ascending addr.
|
||||||
local atoms_by_name = corpus.atoms_by_name or {}
|
local atoms_by_name = corpus.atoms_by_name or {}
|
||||||
@@ -834,10 +834,10 @@ end
|
|||||||
--- This is intentional: silently falling back to a hardcoded GPR would mask the missing opt-in.
|
--- This is intentional: silently falling back to a hardcoded GPR would mask the missing opt-in.
|
||||||
---
|
---
|
||||||
--- Pre-tokenized: `body_tokens` is the scan-source pass's pre-split list of top-level statements (each entry is a single `load_*` call or other statement).
|
--- Pre-tokenized: `body_tokens` is the scan-source pass's pre-split list of top-level statements (each entry is a single `load_*` call or other statement).
|
||||||
--- @param body_tokens table[] -- The atom's pre-tokenized body statements (from atom.body_tokens)
|
--- @param body_tokens table[] -- the atom's pre-tokenized body statements (from atom.body_tokens)
|
||||||
--- @param binds_name string -- Expected Binds_X name (skip pairs with mismatching binds)
|
--- @param binds_name string -- expected Binds_X name (skip pairs with mismatching binds)
|
||||||
--- @param registries table -- Merged registries from collect_per_source_registries
|
--- @param registries table -- merged registries from collect_per_source_registries
|
||||||
--- @return table[] -- List of {reg = <MIPS index>, field = <field name>}
|
--- @return table[] -- list of {reg = <MIPS index>, field = <field name>}
|
||||||
local function parse_body_load_pairs(body_tokens, binds_name, registries)
|
local function parse_body_load_pairs(body_tokens, binds_name, registries)
|
||||||
local pairs = {}
|
local pairs = {}
|
||||||
local reg_index_by_name = (registries and registries.register_alias_registry) or {}
|
local reg_index_by_name = (registries and registries.register_alias_registry) or {}
|
||||||
@@ -880,9 +880,9 @@ end
|
|||||||
--- The piece chain uses (DW_OP_regN, DW_OP_piece, ULEB128(field_size)).
|
--- The piece chain uses (DW_OP_regN, DW_OP_piece, ULEB128(field_size)).
|
||||||
---
|
---
|
||||||
--- Binds fields come from `scan.binds`; the per-source `scan.binds[i].fields` already carries the typed-field record after the scan-source generalization.
|
--- Binds fields come from `scan.binds`; the per-source `scan.binds[i].fields` already carries the typed-field record after the scan-source generalization.
|
||||||
--- @param corpus table -- From `ctx.shared.corpus`
|
--- @param corpus table -- the corpus from `ctx.shared.corpus`
|
||||||
--- @param atom_table table[] -- Cross-ref'd atom table from build_atom_table
|
--- @param atom_table table[] -- the cross-ref'd atom table from build_atom_table
|
||||||
--- @param registries table -- Merged registries from collect_per_source_registries
|
--- @param registries table -- merged registries from collect_per_source_registries
|
||||||
--- @return table, table -- (rbind_atoms, rbind_structs)
|
--- @return table, table -- (rbind_atoms, rbind_structs)
|
||||||
local function parse_rbind_atoms(corpus, atom_table, registries)
|
local function parse_rbind_atoms(corpus, atom_table, registries)
|
||||||
registries = registries or {}
|
registries = registries or {}
|
||||||
@@ -944,7 +944,7 @@ local function parse_rbind_atoms(corpus, atom_table, registries)
|
|||||||
binds = ai.binds,
|
binds = ai.binds,
|
||||||
fields = struct.fields, -- {name, offset} from scan.binds
|
fields = struct.fields, -- {name, offset} from scan.binds
|
||||||
bytes = struct.bytes,
|
bytes = struct.bytes,
|
||||||
regs = pairs, -- Ordered list of {reg, field}
|
regs = pairs, -- ordered list of {reg, field}
|
||||||
info_line = ai.info_line,
|
info_line = ai.info_line,
|
||||||
}
|
}
|
||||||
table.insert(struct.atom_names, atom_name)
|
table.insert(struct.atom_names, atom_name)
|
||||||
@@ -992,7 +992,7 @@ local function build_dwarf_line_section(existing, atom_table)
|
|||||||
while unit_pos < #existing do
|
while unit_pos < #existing do
|
||||||
if unit_pos + 4 > #existing then return existing end
|
if unit_pos + 4 > #existing then return existing end
|
||||||
local unit_length = elf_dwarf.read_u32_le(existing, unit_pos)
|
local unit_length = elf_dwarf.read_u32_le(existing, unit_pos)
|
||||||
if unit_length == elf_dwarf.dw_dwarf32_terminator then return existing end
|
if unit_length == elf_dwarf.ELF32.dw_dwarf32_terminator then return existing end
|
||||||
local unit_end_excl = unit_pos + 4 + unit_length
|
local unit_end_excl = unit_pos + 4 + unit_length
|
||||||
if unit_end_excl > #existing then return existing end
|
if unit_end_excl > #existing then return existing end
|
||||||
last_pos, last_length, last_end = unit_pos, unit_length, unit_end_excl
|
last_pos, last_length, last_end = unit_pos, unit_length, unit_end_excl
|
||||||
@@ -1036,13 +1036,13 @@ local function build_dwarf_aranges_section(existing, atom_table)
|
|||||||
-- We bump the unit's length field accordingly.
|
-- We bump the unit's length field accordingly.
|
||||||
--
|
--
|
||||||
-- Unit structure (DWARF4 §7.21):
|
-- Unit structure (DWARF4 §7.21):
|
||||||
-- unit_length (4)
|
-- unit_length (4)
|
||||||
-- version (2)
|
-- version (2)
|
||||||
-- debug_info_offset (4) -- CU DIE offset in .debug_info
|
-- debug_info_offset (4) -- CU DIE offset in .debug_info
|
||||||
-- address_size (1)
|
-- address_size (1)
|
||||||
-- segment_size (1)
|
-- segment_size (1)
|
||||||
-- entries... (4-byte addr + 4-byte length)
|
-- entries... (4-byte addr + 4-byte length)
|
||||||
-- terminator (8 bytes: addr=0, length=0)
|
-- terminator (8 bytes: addr=0, length=0)
|
||||||
|
|
||||||
-- Walk all units and emit each one (preserving existing structure).
|
-- Walk all units and emit each one (preserving existing structure).
|
||||||
-- For the LAST unit, replace the terminator with my entries + new term.
|
-- For the LAST unit, replace the terminator with my entries + new term.
|
||||||
@@ -1053,7 +1053,7 @@ local function build_dwarf_aranges_section(existing, atom_table)
|
|||||||
while i < #existing do
|
while i < #existing do
|
||||||
-- Read this unit's length.
|
-- Read this unit's length.
|
||||||
local ul = elf_dwarf.read_u32_le(existing, i)
|
local ul = elf_dwarf.read_u32_le(existing, i)
|
||||||
if ul == elf_dwarf.dw_dwarf32_terminator then
|
if ul == elf_dwarf.ELF32.dw_dwarf32_terminator then
|
||||||
-- DWARF64 marker - not supported.
|
-- DWARF64 marker - not supported.
|
||||||
io.stderr:write("[dwarf_injection] WARN: .debug_aranges contains a DWARF64 marker (0xFFFFFFFF); the 64-bit extension is not supported by this metaprogram; passing through unchanged\n")
|
io.stderr:write("[dwarf_injection] WARN: .debug_aranges contains a DWARF64 marker (0xFFFFFFFF); the 64-bit extension is not supported by this metaprogram; passing through unchanged\n")
|
||||||
return existing
|
return existing
|
||||||
@@ -1344,7 +1344,7 @@ local function build_new_abbrev()
|
|||||||
attr( DW_AT_name, DW_FORM_string)
|
attr( DW_AT_name, DW_FORM_string)
|
||||||
.. attr(DW_AT_low_pc, DW_FORM_addr)
|
.. attr(DW_AT_low_pc, DW_FORM_addr)
|
||||||
.. attr(DW_AT_high_pc, DW_FORM_addr)
|
.. attr(DW_AT_high_pc, DW_FORM_addr)
|
||||||
.. attr(DW_AT_linkage_name, DW_FORM_string)) -- equals DW_AT_name; gdb resolves the subprogram, not the gcc global array
|
.. attr(DW_AT_linkage_name, DW_FORM_string)) -- equals DW_AT_name; lets gdb's symbol-table lookup resolve to our subprogram (not the gcc global `code_<name>` const U4 array)
|
||||||
|
|
||||||
local abbrev_variable = abbrev(ABBREV_VARIABLE, DW_TAG_variable, false, -- DW_CHILDREN_no
|
local abbrev_variable = abbrev(ABBREV_VARIABLE, DW_TAG_variable, false, -- DW_CHILDREN_no
|
||||||
attr( DW_AT_name, DW_FORM_string)
|
attr( DW_AT_name, DW_FORM_string)
|
||||||
@@ -1780,8 +1780,7 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
|
|||||||
emit(uleb128(ABBREV_TYPED_VIEW_POINTER)) -- DW_TAG_pointer_type (abbrev 110; NOT 9; void chain target)
|
emit(uleb128(ABBREV_TYPED_VIEW_POINTER)) -- DW_TAG_pointer_type (abbrev 110; NOT 9; void chain target)
|
||||||
emit(elf_dwarf.write_u32_le(ref4_of(void_chain_offset))) -- 4-byte ref4: points at the void base_type's tag byte
|
emit(elf_dwarf.write_u32_le(ref4_of(void_chain_offset))) -- 4-byte ref4: points at the void base_type's tag byte
|
||||||
-- type_chain_offsets["void|1"] is what step (f) of the per-RR_<R_Name> chain looks up.
|
-- type_chain_offsets["void|1"] is what step (f) of the per-RR_<R_Name> chain looks up.
|
||||||
type_chain_offsets["void|1"] = void_chain_offset -- both the base_type offset and the pointer_type are emitted consecutively; the OUTERMOST is the pointer_type.
|
type_chain_offsets["void|1"] = void_chain_offset -- both the base_type offset and the pointer_type are emitted consecutively; the OUTERMOST is the pointer_type. The variable's DW_AT_type must reference the pointer_type, not the base_type. Patch below.
|
||||||
-- The variable's DW_AT_type must reference the pointer_type, not the base_type. Patch below.
|
|
||||||
-- Capture the pointer_type's offset (the last-thing-emitted DIE start) and overwrite the lookup.
|
-- Capture the pointer_type's offset (the last-thing-emitted DIE start) and overwrite the lookup.
|
||||||
-- The pointer_type was emitted as: uleb(9) (1 byte) + 4-byte ref4 = 5 bytes. Its tag byte is at void_chain_offset + 8 (the base_type's 8 bytes: 1 tag + 5 name + 1 byte_size + 1 encoding).
|
-- The pointer_type was emitted as: uleb(9) (1 byte) + 4-byte ref4 = 5 bytes. Its tag byte is at void_chain_offset + 8 (the base_type's 8 bytes: 1 tag + 5 name + 1 byte_size + 1 encoding).
|
||||||
local ptr_void_offset = void_chain_offset + 8
|
local ptr_void_offset = void_chain_offset + 8
|
||||||
@@ -1857,7 +1856,7 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
|
|||||||
end
|
end
|
||||||
|
|
||||||
-- 4) Emit per-atom DW_TAG_subprograms (children of main CU).
|
-- 4) Emit per-atom DW_TAG_subprograms (children of main CU).
|
||||||
-- Subprogram names match the written C ident (the ELF symbol).
|
-- Subprogram names match nm symbols without a `code_` prefix.
|
||||||
-- The gcc global `<name>[]` is a DW_TAG_variable without children; our subprogram has the wave-context var children.
|
-- The gcc global `<name>[]` is a DW_TAG_variable without children; our subprogram has the wave-context var children.
|
||||||
-- gdb's symbol resolution picks our subprogram (it has low_pc/high_pc + children) over the gcc global for function-context lookups.
|
-- gdb's symbol resolution picks our subprogram (it has low_pc/high_pc + children) over the gcc global for function-context lookups.
|
||||||
for _, atom in ipairs(atom_table) do
|
for _, atom in ipairs(atom_table) do
|
||||||
@@ -2313,6 +2312,5 @@ end
|
|||||||
M.compute_loclists_offsets_for_test = compute_loclists_offsets
|
M.compute_loclists_offsets_for_test = compute_loclists_offsets
|
||||||
M.build_debug_loclists_section_for_test = build_debug_loclists_section
|
M.build_debug_loclists_section_for_test = build_debug_loclists_section
|
||||||
M.tape_piece_size_for_test = tape_piece_size
|
M.tape_piece_size_for_test = tape_piece_size
|
||||||
M.build_atom_table_for_test = build_atom_table
|
|
||||||
|
|
||||||
return M
|
return M
|
||||||
|
|||||||
@@ -155,28 +155,8 @@ local function project_atom(atom_record, src, corpus)
|
|||||||
local body = atom_record.body or ""
|
local body = atom_record.body or ""
|
||||||
local wc = corpus.word_counts or {}
|
local wc = corpus.word_counts or {}
|
||||||
local cbi = corpus.component_body_index or {}
|
local cbi = corpus.component_body_index or {}
|
||||||
local schema = nil
|
|
||||||
if atom_record.reg_use_schema_name then
|
|
||||||
schema = corpus.reg_use_schemas and corpus.reg_use_schemas[atom_record.reg_use_schema_name]
|
|
||||||
end
|
|
||||||
-- That construction site stamps `invocation.debug_skip` while appending each record to `proj.invocations`.
|
-- That construction site stamps `invocation.debug_skip` while appending each record to `proj.invocations`.
|
||||||
local proj = duffle.project_emission(body, cbi, wc, corpus.components, {
|
local proj = duffle.project_emission(body, cbi, wc, corpus.components)
|
||||||
reg_use_schema = schema,
|
|
||||||
reg_use_param = atom_record.reg_use_param_name,
|
|
||||||
atom_name = atom_record.name,
|
|
||||||
schema_name = atom_record.reg_use_schema_name,
|
|
||||||
})
|
|
||||||
if atom_record.reg_use_schema_name and not schema then
|
|
||||||
proj.errors[#proj.errors + 1] = {
|
|
||||||
kind = "reguse_missing_schema",
|
|
||||||
msg = string.format("RegUse schema %q is missing", atom_record.reg_use_schema_name),
|
|
||||||
}
|
|
||||||
end
|
|
||||||
for _, err in ipairs(corpus.reg_use_errors or {}) do
|
|
||||||
if err.schema_name == atom_record.reg_use_schema_name then
|
|
||||||
proj.errors[#proj.errors + 1] = err
|
|
||||||
end
|
|
||||||
end
|
|
||||||
local paths = {
|
local paths = {
|
||||||
tokens = atom_record.body_tokens or {},
|
tokens = atom_record.body_tokens or {},
|
||||||
line_in_body = duffle.build_body_line_index(body),
|
line_in_body = duffle.build_body_line_index(body),
|
||||||
@@ -208,11 +188,11 @@ function M.run(ctx)
|
|||||||
if type(corpus.source_order) ~= "table" then error("emission_model: ctx.shared.corpus.source_order is required", 0) end
|
if type(corpus.source_order) ~= "table" then error("emission_model: ctx.shared.corpus.source_order is required", 0) end
|
||||||
|
|
||||||
-- Project once, collect errors + warnings for one atom.
|
-- Project once, collect errors + warnings for one atom.
|
||||||
-- Kind must be one of: atom | atom_proc | raw_atom | comp_bare | comp_proc.
|
-- Kind must be one of: atom | raw_atom | comp_bare | comp_proc.
|
||||||
local function process_atom(atom, src)
|
local function process_atom(atom, src)
|
||||||
if not (atom and atom.body) then return end
|
if not (atom and atom.body) then return end
|
||||||
local kind = atom.kind
|
local kind = atom.kind
|
||||||
if kind ~= "atom" and kind ~= "atom_proc" and kind ~= "raw_atom" and kind ~= "comp_bare" and kind ~= "comp_proc" then
|
if kind ~= "atom" and kind ~= "raw_atom" and kind ~= "comp_bare" and kind ~= "comp_proc" then
|
||||||
return
|
return
|
||||||
end
|
end
|
||||||
local proj = project_atom(atom, src, corpus)
|
local proj = project_atom(atom, src, corpus)
|
||||||
@@ -235,7 +215,7 @@ function M.run(ctx)
|
|||||||
end
|
end
|
||||||
|
|
||||||
-- Walk `corpus.source_order`; within each source, visit atoms followed by raw_atoms.
|
-- Walk `corpus.source_order`; within each source, visit atoms followed by raw_atoms.
|
||||||
-- Recognized kinds (atom | atom_proc | raw_atom | comp_bare | comp_proc) each receive the atom.paths projection via duffle.project_emission.
|
-- Recognized kinds (atom | raw_atom | comp_bare | comp_proc) each receive the atom.paths projection via duffle.project_emission.
|
||||||
-- Components are macros inlined into atom bodies; focused tests and isolated component analyses consume atom.paths directly.
|
-- Components are macros inlined into atom bodies; focused tests and isolated component analyses consume atom.paths directly.
|
||||||
for _, src in ipairs(corpus.source_order) do
|
for _, src in ipairs(corpus.source_order) do
|
||||||
local scan = src.scan or {}
|
local scan = src.scan or {}
|
||||||
|
|||||||
@@ -1,21 +1,12 @@
|
|||||||
--- passes/offsets.lua — Branch-offset generator.
|
--- passes/offsets.lua — Branch-offset generator.
|
||||||
---
|
---
|
||||||
--- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`)
|
--- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`)
|
||||||
--- for `MipsAtom_(name)` and leftover `MipsCode code_*` declarations, computes the word offset
|
--- for `MipsAtom_(name)` and `MipsCode code_<name>` declarations, computes the word offset
|
||||||
--- (ELF symbol is the C ident; raw `code_*` is leftover, not the atom rule)
|
|
||||||
--- from each `atom_offset(F, T)` marker to its target `atom_label(T)` declaration, and emits
|
--- from each `atom_offset(F, T)` marker to its target `atom_label(T)` declaration, and emits
|
||||||
--- `gen/offsets.h` with one `#define _atom_offset_F_T = N` per branch.
|
--- `gen/offsets.h` with one `#define _atom_offset_F_T = N` per branch.
|
||||||
---
|
|
||||||
--- Per-directory aggregation: every source in the same directory contributes to the same `gen/offsets.h`.
|
--- Per-directory aggregation: every source in the same directory contributes to the same `gen/offsets.h`.
|
||||||
--- The directory itself is the namespace; the filename does not repeat the module name.
|
--- The directory itself is the namespace; the filename does not repeat the module name.
|
||||||
---
|
---
|
||||||
--- (Task 12.16 note: atom-namespaced enum names — e.g., `atom_offset__normalize_v3s4__srav_path__aligned_done` —
|
|
||||||
--- were considered to prevent cross-atom label collisions, but the C-side `atom_offset(F, T)` macro in
|
|
||||||
--- `code/duffle/dsl.atom.h` doesn't know the current atom_name at expansion time, so any namespacing
|
|
||||||
--- on the metaprogram side breaks the C build. Reverted. The C-side would need a per-atom
|
|
||||||
--- `CURRENT_ATOM` #define (set by `MipsAtom_`/`MipsAtom_Proc_` macros) plus an updated `atom_offset`
|
|
||||||
--- macro that uses it. That's a coordinated refactor — deferred to a future track.)
|
|
||||||
---
|
|
||||||
--- The offset is `target_word - branch_word - 1` (the standard MIPS branch-immediate encoding: branch_offset = relative_pc_in_words - 1).
|
--- The offset is `target_word - branch_word - 1` (the standard MIPS branch-immediate encoding: branch_offset = relative_pc_in_words - 1).
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|||||||
+225
-576
@@ -4,8 +4,8 @@
|
|||||||
--- - `build/gen/<dir_basename>.annotations.txt` — one per source-directory containing atoms; aggregates across all sources in the directory.
|
--- - `build/gen/<dir_basename>.annotations.txt` — one per source-directory containing atoms; aggregates across all sources in the directory.
|
||||||
--- - `build/gen/annotation_validation.txt` — the project summary.
|
--- - `build/gen/annotation_validation.txt` — the project summary.
|
||||||
---
|
---
|
||||||
--- The canonical `corpus.sources_by_dir` projection groups sources by directory.
|
--- The annotation pass emits `errors.h` files per module and the canonical `corpus.sources_by_dir` projection groups sources by directory.
|
||||||
--- This pass builds one ModuleView per directory and walks SECTION_RENDERERS.
|
--- This pass iterates the dir projection directly and re-validates each source via `annotation.validate()` to get the detailed per-source results.
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- Module-scope requires + package.path setup
|
-- Module-scope requires + package.path setup
|
||||||
@@ -20,6 +20,11 @@
|
|||||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||||
|
|
||||||
|
-- Load the annotation pass so we can re-validate each source against the canonical corpus projection.
|
||||||
|
-- The annotation pass exposes `M.validate`, which returns the per-source AnnotationResult (atoms / annots / macros / binds / errors / warnings)
|
||||||
|
-- that the report pass renders into the per-module `<dir_basename>.annotations.txt` output.
|
||||||
|
local annotation = dofile(_bootstrap_dir .. "annotation.lua")
|
||||||
|
|
||||||
-- Load atoms_source_map for the `render_source_map` / `render_provenance` module functions (used by `render_module_atoms_md` to produce `<module>.atoms.md` without re-walking source tokens).
|
-- Load atoms_source_map for the `render_source_map` / `render_provenance` module functions (used by `render_module_atoms_md` to produce `<module>.atoms.md` without re-walking source tokens).
|
||||||
-- The pass itself emits no per-source files anymore; we only consume the two pure renderers here.
|
-- The pass itself emits no per-source files anymore; we only consume the two pure renderers here.
|
||||||
-- Defined BEFORE the renderer functions below so their upvalues resolve to this local (not the global `atoms_source_map`, which is nil).
|
-- Defined BEFORE the renderer functions below so their upvalues resolve to this local (not the global `atoms_source_map`, which is nil).
|
||||||
@@ -220,7 +225,7 @@ local function render_module_atoms_md(dir, dir_sources, wc)
|
|||||||
for _, atom in ipairs(atoms_list) do
|
for _, atom in ipairs(atoms_list) do
|
||||||
lines[#lines + 1] = string.format(
|
lines[#lines + 1] = string.format(
|
||||||
"### atom: %s (line %d, %d words)",
|
"### atom: %s (line %d, %d words)",
|
||||||
atom.name, atom.line or 0, #((atom.paths or {}).word_events or {}))
|
atom.name, atom.line or 0, #(atom.paths.items or {}))
|
||||||
lines[#lines + 1] = ""
|
lines[#lines + 1] = ""
|
||||||
lines[#lines + 1] = "**Sourcemap** — per-word call site:"
|
lines[#lines + 1] = "**Sourcemap** — per-word call site:"
|
||||||
lines[#lines + 1] = "```"
|
lines[#lines + 1] = "```"
|
||||||
@@ -241,544 +246,17 @@ local function render_module_atoms_md(dir, dir_sources, wc)
|
|||||||
return table.concat(lines, "\n") .. "\n"
|
return table.concat(lines, "\n") .. "\n"
|
||||||
end
|
end
|
||||||
|
|
||||||
local function decl_words(atom)
|
|
||||||
local p = atom.paths or {}
|
|
||||||
return #(p.word_events or {})
|
|
||||||
end
|
|
||||||
|
|
||||||
local function count_kinds(decls)
|
|
||||||
local n = { atom = 0, atom_proc = 0, comp_bare = 0, comp_proc = 0 }
|
|
||||||
for _, a in ipairs(decls or {}) do
|
|
||||||
if n[a.kind] ~= nil then n[a.kind] = n[a.kind] + 1 end
|
|
||||||
end
|
|
||||||
return n
|
|
||||||
end
|
|
||||||
|
|
||||||
local function slot_suffix(key)
|
|
||||||
if type(key) ~= "string" or key:sub(1, 7) ~= "reguse:" then return nil end
|
|
||||||
return key:match("([^:]+)$")
|
|
||||||
end
|
|
||||||
|
|
||||||
local function decl_names(view)
|
|
||||||
local names = {}
|
|
||||||
for _, a in ipairs(view.decls or {}) do
|
|
||||||
if a.name then names[a.name] = true end
|
|
||||||
end
|
|
||||||
return names
|
|
||||||
end
|
|
||||||
|
|
||||||
local function path_in_module(path, view)
|
|
||||||
if type(path) ~= "string" or path == "" then return false end
|
|
||||||
local norm = path:gsub("\\", "/")
|
|
||||||
local dir = (view.dir or ""):gsub("\\", "/")
|
|
||||||
if dir ~= "" and (norm == dir or norm:sub(1, #dir + 1) == dir .. "/") then
|
|
||||||
return true
|
|
||||||
end
|
|
||||||
for _, src in ipairs(view.sources or {}) do
|
|
||||||
if (src.path or ""):gsub("\\", "/") == norm then return true end
|
|
||||||
end
|
|
||||||
return false
|
|
||||||
end
|
|
||||||
|
|
||||||
local function build_module_view(dir, dir_sources, corpus)
|
|
||||||
local decls = {}
|
|
||||||
for _, src in ipairs(dir_sources or {}) do
|
|
||||||
for _, a in ipairs((src.scan and src.scan.atoms) or {}) do
|
|
||||||
if not a.source_path then a.source_path = src.path end
|
|
||||||
decls[#decls + 1] = a
|
|
||||||
end
|
|
||||||
end
|
|
||||||
local dir_basename = source_basename(dir)
|
|
||||||
local sa = (corpus.static_analysis_results or {})[dir_basename] or {}
|
|
||||||
local schemas = {}
|
|
||||||
for name, schema in pairs(corpus.reg_use_schemas or {}) do
|
|
||||||
for _, a in ipairs(decls) do
|
|
||||||
if a.reg_use_schema_name == name then
|
|
||||||
schemas[#schemas + 1] = schema
|
|
||||||
break
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
return {
|
|
||||||
dir = dir,
|
|
||||||
sources = dir_sources or {},
|
|
||||||
decls = decls,
|
|
||||||
schemas = schemas,
|
|
||||||
findings = sa.findings or {},
|
|
||||||
sa = sa,
|
|
||||||
corpus = corpus,
|
|
||||||
}
|
|
||||||
end
|
|
||||||
|
|
||||||
local function render_section_declarations(add, view)
|
|
||||||
if #view.decls == 0 then add("_(none)_"); add(""); return end
|
|
||||||
add("| kind | name | source | line | words | min | max | branches | paths |")
|
|
||||||
add("|------|------|--------|------|-------|-----|-----|----------|-------|")
|
|
||||||
for _, a in ipairs(view.decls) do
|
|
||||||
local p = a.paths or {}
|
|
||||||
add(string.format("| %s | %s | %s | %d | %d | %s | %s | %s | %s |",
|
|
||||||
a.kind or "?",
|
|
||||||
a.name or "?",
|
|
||||||
source_basename(a.source_path or ""),
|
|
||||||
a.line or 0,
|
|
||||||
decl_words(a),
|
|
||||||
tostring(p.cycles_min or "—"),
|
|
||||||
tostring(p.cycles_max or "—"),
|
|
||||||
tostring(p.branches or "—"),
|
|
||||||
tostring(p.paths or "—")))
|
|
||||||
end
|
|
||||||
add("")
|
|
||||||
end
|
|
||||||
|
|
||||||
local function render_section_components(add, view)
|
|
||||||
local rows = {}
|
|
||||||
local index = (view.corpus and view.corpus.component_body_index) or {}
|
|
||||||
for _, a in ipairs(view.decls) do
|
|
||||||
if a.kind == "comp_bare" or a.kind == "comp_proc" then
|
|
||||||
local idx = index[a.name] or {}
|
|
||||||
local args = idx.arg_names or {}
|
|
||||||
rows[#rows + 1] = {
|
|
||||||
name = a.name,
|
|
||||||
kind = a.kind,
|
|
||||||
args = table.concat(args, ", "),
|
|
||||||
words = decl_words(a),
|
|
||||||
map = a.map_command or "—",
|
|
||||||
}
|
|
||||||
end
|
|
||||||
end
|
|
||||||
if #rows == 0 then add("_(none)_"); add(""); return end
|
|
||||||
add("| name | kind | arg_names | words | map |")
|
|
||||||
add("|------|------|-----------|-------|-----|")
|
|
||||||
for _, r in ipairs(rows) do
|
|
||||||
add(string.format("| %s | %s | %s | %d | %s |",
|
|
||||||
r.name, r.kind, r.args ~= "" and r.args or "—", r.words, r.map))
|
|
||||||
end
|
|
||||||
add("")
|
|
||||||
end
|
|
||||||
|
|
||||||
local function render_section_reguse(add, view)
|
|
||||||
local wrote = false
|
|
||||||
for _, schema in ipairs(view.schemas or {}) do
|
|
||||||
wrote = true
|
|
||||||
add(string.format("### %s", schema.name or "?"))
|
|
||||||
for _, slot in ipairs(schema.slots or {}) do
|
|
||||||
local aliases = table.concat(slot.aliases or { slot.name }, ", ")
|
|
||||||
local ro = slot.readonly and " readonly" or ""
|
|
||||||
add(string.format("- slot `%s` aliases %s%s", slot.name, aliases, ro))
|
|
||||||
end
|
|
||||||
for _, a in ipairs(view.decls) do
|
|
||||||
if a.reg_use_schema_name == schema.name then
|
|
||||||
add(string.format("- bound `%s` param `%s`", a.name, a.reg_use_param_name or "?"))
|
|
||||||
end
|
|
||||||
end
|
|
||||||
add("")
|
|
||||||
end
|
|
||||||
local bound = {}
|
|
||||||
for _, schema in ipairs(view.schemas or {}) do
|
|
||||||
if schema.name then bound[schema.name] = true end
|
|
||||||
end
|
|
||||||
local errors = {}
|
|
||||||
for _, err in ipairs((view.corpus and view.corpus.reg_use_errors) or {}) do
|
|
||||||
if bound[err.schema_name] or path_in_module(err.source_file, view) then
|
|
||||||
errors[#errors + 1] = err
|
|
||||||
end
|
|
||||||
end
|
|
||||||
if #errors > 0 then
|
|
||||||
wrote = true
|
|
||||||
add("### parse errors")
|
|
||||||
for _, err in ipairs(errors) do
|
|
||||||
add(string.format("- `%s` %s", err.kind or "?", err.schema_name or ""))
|
|
||||||
end
|
|
||||||
add("")
|
|
||||||
end
|
|
||||||
if not wrote then add("_(none)_"); add("") end
|
|
||||||
end
|
|
||||||
|
|
||||||
local function render_section_annotations(add, view)
|
|
||||||
local rows = {}
|
|
||||||
for _, src in ipairs(view.sources) do
|
|
||||||
for _, info in ipairs((src.scan and src.scan.atom_infos) or {}) do
|
|
||||||
rows[#rows + 1] = {
|
|
||||||
source = source_basename(src.path),
|
|
||||||
line = info.info_line or 0,
|
|
||||||
name = info.atom_name or "?",
|
|
||||||
binds = info.binds or "—",
|
|
||||||
reads = (#(info.reads or {}) > 0 and table.concat(info.reads, ",")) or "—",
|
|
||||||
writes = (#(info.writes or {}) > 0 and table.concat(info.writes, ",")) or "—",
|
|
||||||
phase = info.phase or "—",
|
|
||||||
}
|
|
||||||
end
|
|
||||||
end
|
|
||||||
if #rows == 0 then add("_(none)_"); add(""); return end
|
|
||||||
add("| source | line | name | binds | reads | writes | phase |")
|
|
||||||
add("|--------|------|------|-------|-------|--------|-------|")
|
|
||||||
for _, r in ipairs(rows) do
|
|
||||||
add(string.format("| %s | %d | %s | %s | %s | %s | %s |",
|
|
||||||
r.source, r.line, r.name, r.binds, r.reads, r.writes, r.phase))
|
|
||||||
end
|
|
||||||
add("")
|
|
||||||
end
|
|
||||||
|
|
||||||
local function render_section_component_annotations(add, view)
|
|
||||||
local rows = {}
|
|
||||||
for _, src in ipairs(view.sources) do
|
|
||||||
for _, info in ipairs((src.scan and src.scan.component_atom_infos) or {}) do
|
|
||||||
rows[#rows + 1] = {
|
|
||||||
source = source_basename(src.path),
|
|
||||||
line = info.info_line or 0,
|
|
||||||
name = info.atom_name or "?",
|
|
||||||
reads = (#(info.reads or {}) > 0 and table.concat(info.reads, ",")) or "—",
|
|
||||||
writes = (#(info.writes or {}) > 0 and table.concat(info.writes, ",")) or "—",
|
|
||||||
}
|
|
||||||
end
|
|
||||||
end
|
|
||||||
if #rows == 0 then add("_(none)_"); add(""); return end
|
|
||||||
add("| source | line | name | reads | writes |")
|
|
||||||
add("|--------|------|------|-------|--------|")
|
|
||||||
for _, r in ipairs(rows) do
|
|
||||||
add(string.format("| %s | %d | %s | %s | %s |",
|
|
||||||
r.source, r.line, r.name, r.reads, r.writes))
|
|
||||||
end
|
|
||||||
add("")
|
|
||||||
end
|
|
||||||
|
|
||||||
local function render_section_binds(add, view)
|
|
||||||
local wrote = false
|
|
||||||
for _, src in ipairs(view.sources) do
|
|
||||||
for _, b in ipairs((src.scan and src.scan.binds) or {}) do
|
|
||||||
wrote = true
|
|
||||||
add(string.format("### %s (%s:%s, %s bytes)",
|
|
||||||
b.name, source_basename(src.path), tostring(b.line or 0), tostring(b.bytes or "—")))
|
|
||||||
for _, f in ipairs(b.fields or {}) do
|
|
||||||
add(string.format("- `+%s %s`", tostring(f.offset or "?"), f.name or "?"))
|
|
||||||
end
|
|
||||||
add("")
|
|
||||||
end
|
|
||||||
end
|
|
||||||
if not wrote then add("_(none)_"); add("") end
|
|
||||||
end
|
|
||||||
|
|
||||||
local function render_section_phases(add, view)
|
|
||||||
local corpus = view.corpus or {}
|
|
||||||
local names = decl_names(view)
|
|
||||||
local wrote = false
|
|
||||||
for phase, entry in pairs(corpus.atom_phases or {}) do
|
|
||||||
local here = {}
|
|
||||||
for _, atom_name in ipairs(entry.atoms or {}) do
|
|
||||||
if names[atom_name] then here[#here + 1] = atom_name end
|
|
||||||
end
|
|
||||||
if #here > 0 then
|
|
||||||
wrote = true
|
|
||||||
add(string.format("- phase `%s`: %s", phase, table.concat(here, ", ")))
|
|
||||||
end
|
|
||||||
end
|
|
||||||
for name, entry in pairs(corpus.atom_views or {}) do
|
|
||||||
if names[name] then
|
|
||||||
wrote = true
|
|
||||||
add(string.format("- view `%s` binds `%s`", name, entry.binds_name or "—"))
|
|
||||||
end
|
|
||||||
end
|
|
||||||
for name, entry in pairs(corpus.atom_ctxs or {}) do
|
|
||||||
if names[name] then
|
|
||||||
wrote = true
|
|
||||||
add(string.format("- ctx `%s` rbind `%s`", name, entry.rbind_atom or "—"))
|
|
||||||
end
|
|
||||||
end
|
|
||||||
if not wrote then add("_(none)_") end
|
|
||||||
add("")
|
|
||||||
end
|
|
||||||
|
|
||||||
local function render_section_aliases(add, view)
|
|
||||||
local names = {}
|
|
||||||
local seen = {}
|
|
||||||
for _, src in ipairs(view.sources or {}) do
|
|
||||||
for name, entry in pairs((src.scan and src.scan.register_alias_registry) or {}) do
|
|
||||||
if not seen[name] then
|
|
||||||
seen[name] = entry
|
|
||||||
names[#names + 1] = name
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
table.sort(names)
|
|
||||||
if #names == 0 then add("_(none)_"); add(""); return end
|
|
||||||
add("| alias | type |")
|
|
||||||
add("|-------|------|")
|
|
||||||
for _, name in ipairs(names) do
|
|
||||||
local e = seen[name]
|
|
||||||
add(string.format("| %s | %s |", name, (e and e.default_type) or "—"))
|
|
||||||
end
|
|
||||||
add("")
|
|
||||||
end
|
|
||||||
|
|
||||||
local function render_section_autoreg(add, view)
|
|
||||||
local allowed = decl_names(view)
|
|
||||||
for phase, entry in pairs((view.corpus and view.corpus.atom_phases) or {}) do
|
|
||||||
for _, atom_name in ipairs(entry.atoms or {}) do
|
|
||||||
if allowed[atom_name] then allowed[phase] = true end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
local wrote = false
|
|
||||||
local seen = {}
|
|
||||||
local function dump(label, table_map)
|
|
||||||
local scopes = {}
|
|
||||||
for scope in pairs(table_map or {}) do
|
|
||||||
if allowed[scope] and not seen[label .. "\0" .. scope] then
|
|
||||||
scopes[#scopes + 1] = scope
|
|
||||||
end
|
|
||||||
end
|
|
||||||
table.sort(scopes)
|
|
||||||
for _, scope in ipairs(scopes) do
|
|
||||||
seen[label .. "\0" .. scope] = true
|
|
||||||
wrote = true
|
|
||||||
local syms = {}
|
|
||||||
for sym, gpr in pairs(table_map[scope] or {}) do
|
|
||||||
if type(gpr) == "string" and gpr ~= sym then
|
|
||||||
syms[#syms + 1] = string.format("%s → %s", sym, gpr)
|
|
||||||
else
|
|
||||||
syms[#syms + 1] = tostring(sym)
|
|
||||||
end
|
|
||||||
end
|
|
||||||
table.sort(syms)
|
|
||||||
add(string.format("- %s `%s`: %s", label, scope, table.concat(syms, ", ")))
|
|
||||||
end
|
|
||||||
end
|
|
||||||
local corpus = view.corpus or {}
|
|
||||||
dump("atom", corpus.atom_auto_regs)
|
|
||||||
dump("phase", corpus.phase_auto_regs)
|
|
||||||
for _, src in ipairs(view.sources or {}) do
|
|
||||||
dump("atom", src.scan and src.scan.atom_auto_regs)
|
|
||||||
dump("phase", src.scan and src.scan.phase_auto_regs)
|
|
||||||
end
|
|
||||||
if not wrote then add("_(none)_") end
|
|
||||||
add("")
|
|
||||||
end
|
|
||||||
|
|
||||||
local function render_section_collisions(add, view)
|
|
||||||
local rows = {}
|
|
||||||
for _, c in ipairs((view.corpus and view.corpus.collisions) or {}) do
|
|
||||||
local first = c.first_site or {}
|
|
||||||
local other = c.conflicting_site or {}
|
|
||||||
if path_in_module(first.path, view) or path_in_module(other.path, view) then
|
|
||||||
rows[#rows + 1] = c
|
|
||||||
end
|
|
||||||
end
|
|
||||||
if #rows == 0 then add("_(none)_"); add(""); return end
|
|
||||||
for _, c in ipairs(rows) do
|
|
||||||
local first = c.first_site or {}
|
|
||||||
local other = c.conflicting_site or {}
|
|
||||||
add(string.format("- `%s` `%s` first %s:%s conflict %s:%s",
|
|
||||||
c.kind or "?", c.name or "?",
|
|
||||||
tostring(first.path or "?"), tostring(first.line or "?"),
|
|
||||||
tostring(other.path or "?"), tostring(other.line or "?")))
|
|
||||||
end
|
|
||||||
add("")
|
|
||||||
end
|
|
||||||
|
|
||||||
local function render_section_findings(add, view)
|
|
||||||
local by_atom = {}
|
|
||||||
for _, f in ipairs(view.findings or {}) do
|
|
||||||
local key = f.atom or "?"
|
|
||||||
by_atom[key] = by_atom[key] or {}
|
|
||||||
by_atom[key][#by_atom[key] + 1] = f
|
|
||||||
end
|
|
||||||
if next(by_atom) == nil then add("_(none)_"); add(""); return end
|
|
||||||
local seen = {}
|
|
||||||
local function emit(name, fs)
|
|
||||||
add("### " .. name)
|
|
||||||
for _, f in ipairs(fs) do
|
|
||||||
local msg = f.msg or ""
|
|
||||||
local slot = slot_suffix(f.gpr_key or f.producer_destination)
|
|
||||||
if slot and not msg:find("(slot ", 1, true) then
|
|
||||||
msg = msg .. " (slot " .. slot .. ")"
|
|
||||||
end
|
|
||||||
add(string.format("- `[%s/%s] %s`", f.kind or "info", f.check or "?", msg))
|
|
||||||
end
|
|
||||||
add("")
|
|
||||||
end
|
|
||||||
for _, a in ipairs(view.decls) do
|
|
||||||
if by_atom[a.name] then
|
|
||||||
seen[a.name] = true
|
|
||||||
emit(a.name, by_atom[a.name])
|
|
||||||
end
|
|
||||||
end
|
|
||||||
local leftovers = {}
|
|
||||||
for name in pairs(by_atom) do
|
|
||||||
if not seen[name] then leftovers[#leftovers + 1] = name end
|
|
||||||
end
|
|
||||||
table.sort(leftovers)
|
|
||||||
for _, name in ipairs(leftovers) do emit(name, by_atom[name]) end
|
|
||||||
end
|
|
||||||
|
|
||||||
local function render_section_relations(add, view)
|
|
||||||
local wrote = false
|
|
||||||
for _, a in ipairs(view.decls) do
|
|
||||||
local rels = (a.paths and a.paths.relations) or {}
|
|
||||||
if #rels > 0 then
|
|
||||||
wrote = true
|
|
||||||
add("### " .. a.name)
|
|
||||||
for _, rel in ipairs(rels) do
|
|
||||||
local dest = rel.destination or rel.producer_destination or "—"
|
|
||||||
local slot = slot_suffix(dest)
|
|
||||||
local dest_s = tostring(dest)
|
|
||||||
if slot then dest_s = dest_s .. " (slot " .. slot .. ")" end
|
|
||||||
add(string.format("- `%s` words %s → %s dest %s",
|
|
||||||
rel.semantic or "?",
|
|
||||||
tostring(rel.producer_word or "?"),
|
|
||||||
tostring(rel.consumer_word or "?"),
|
|
||||||
dest_s))
|
|
||||||
end
|
|
||||||
add("")
|
|
||||||
end
|
|
||||||
end
|
|
||||||
if not wrote then add("_(none)_"); add("") end
|
|
||||||
end
|
|
||||||
|
|
||||||
local HIDDEN_UNLESS_WRITTEN = {
|
|
||||||
R_AT = true, R_TapePtr = true, R_AtomJmp = true,
|
|
||||||
}
|
|
||||||
|
|
||||||
local PHYSICAL_GPR = {
|
|
||||||
R_T0 = true, R_T1 = true, R_T2 = true, R_T3 = true,
|
|
||||||
R_T4 = true, R_T5 = true, R_T6 = true, R_T7 = true,
|
|
||||||
R_V0 = true, R_V1 = true,
|
|
||||||
}
|
|
||||||
|
|
||||||
local function encoder_wrote_key(atom, key)
|
|
||||||
for _, ev in ipairs((atom.paths and atom.paths.word_events) or {}) do
|
|
||||||
for _, dest in pairs(ev.gpr_keys or {}) do
|
|
||||||
if dest == key then return true end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
return false
|
|
||||||
end
|
|
||||||
|
|
||||||
local function written_name_for(key, atom)
|
|
||||||
local slot = key:match("^reguse:.+:(.+)$")
|
|
||||||
if slot then
|
|
||||||
local param = atom.reg_use_param_name
|
|
||||||
if param and param ~= "" then return param .. "." .. slot end
|
|
||||||
return slot
|
|
||||||
end
|
|
||||||
return key
|
|
||||||
end
|
|
||||||
|
|
||||||
local function aliases_for_key(key, atom, view)
|
|
||||||
local slot = key:match("^reguse:.+:(.+)$")
|
|
||||||
if not slot then return "—" end
|
|
||||||
local schema_name = atom.reg_use_schema_name
|
|
||||||
local schema = view.corpus and view.corpus.reg_use_schemas and view.corpus.reg_use_schemas[schema_name]
|
|
||||||
if not schema then return "—" end
|
|
||||||
for _, s in ipairs(schema.slots or {}) do
|
|
||||||
if s.name == slot then
|
|
||||||
local names = {}
|
|
||||||
for _, alias in ipairs(s.aliases or {}) do
|
|
||||||
if alias ~= slot then names[#names + 1] = alias end
|
|
||||||
end
|
|
||||||
if #names == 0 then
|
|
||||||
if s.aliases and #s.aliases > 0 then return table.concat(s.aliases, ", ") end
|
|
||||||
return "—"
|
|
||||||
end
|
|
||||||
return table.concat(names, ", ")
|
|
||||||
end
|
|
||||||
end
|
|
||||||
return "—"
|
|
||||||
end
|
|
||||||
|
|
||||||
local function physical_for_key(key, atom, view)
|
|
||||||
if PHYSICAL_GPR[key] then return key end
|
|
||||||
local corpus = view.corpus or {}
|
|
||||||
local alias = (corpus.register_alias_registry or {})[key]
|
|
||||||
if type(alias) == "table" then
|
|
||||||
local phys = alias.physical or alias.gpr or alias.code_name
|
|
||||||
if type(phys) == "string" and PHYSICAL_GPR[phys] then return phys end
|
|
||||||
if type(alias.name) == "string" and PHYSICAL_GPR[alias.name] then return alias.name end
|
|
||||||
elseif type(alias) == "string" and PHYSICAL_GPR[alias] then
|
|
||||||
return alias
|
|
||||||
end
|
|
||||||
local atom_map = (corpus.atom_auto_regs or {})[atom.name]
|
|
||||||
if type(atom_map) == "table" then
|
|
||||||
local slot = key:match("^reguse:.+:(.+)$") or key
|
|
||||||
local bound = atom_map[slot] or atom_map["R_" .. slot]
|
|
||||||
if type(bound) == "string" and PHYSICAL_GPR[bound] then return bound end
|
|
||||||
end
|
|
||||||
return "—"
|
|
||||||
end
|
|
||||||
|
|
||||||
local function last_relation_for(key, atom)
|
|
||||||
local last = nil
|
|
||||||
for _, rel in ipairs((atom.paths and atom.paths.relations) or {}) do
|
|
||||||
local dest = rel.destination or rel.producer_destination
|
|
||||||
if dest == key then last = rel end
|
|
||||||
end
|
|
||||||
if not last then return "—" end
|
|
||||||
local sem = last.semantic or "?"
|
|
||||||
local a = last.producer_word
|
|
||||||
local b = last.consumer_word
|
|
||||||
if a and b then return string.format("%s w%s→%s", sem, tostring(a), tostring(b)) end
|
|
||||||
return sem
|
|
||||||
end
|
|
||||||
|
|
||||||
local function render_section_forward(add, view)
|
|
||||||
local wrote = false
|
|
||||||
for _, a in ipairs(view.decls) do
|
|
||||||
local gpr = a.paths and a.paths.forward_state and a.paths.forward_state.gpr_values
|
|
||||||
local keys = {}
|
|
||||||
for k in pairs(gpr or {}) do
|
|
||||||
if k == "R_0" then
|
|
||||||
-- hidden
|
|
||||||
elseif HIDDEN_UNLESS_WRITTEN[k] and not encoder_wrote_key(a, k) then
|
|
||||||
-- hidden
|
|
||||||
else
|
|
||||||
keys[#keys + 1] = k
|
|
||||||
end
|
|
||||||
end
|
|
||||||
if #keys > 0 then
|
|
||||||
wrote = true
|
|
||||||
add("### " .. a.name)
|
|
||||||
add("| written | aliases | physical | lattice | last relation |")
|
|
||||||
add("|---|---|---|---|---|")
|
|
||||||
table.sort(keys)
|
|
||||||
for _, k in ipairs(keys) do
|
|
||||||
local slot = gpr[k]
|
|
||||||
local lattice = "—"
|
|
||||||
if slot and slot.kind == "constant" then
|
|
||||||
lattice = tostring(slot.value)
|
|
||||||
end
|
|
||||||
add(string.format("| `%s` | %s | %s | %s | %s |",
|
|
||||||
written_name_for(k, a),
|
|
||||||
aliases_for_key(k, a, view),
|
|
||||||
physical_for_key(k, a, view),
|
|
||||||
lattice,
|
|
||||||
last_relation_for(k, a)))
|
|
||||||
end
|
|
||||||
add("")
|
|
||||||
end
|
|
||||||
end
|
|
||||||
if not wrote then add("_(none)_"); add("") end
|
|
||||||
end
|
|
||||||
|
|
||||||
local SECTION_RENDERERS = {
|
|
||||||
{ header = "## Declarations", render = render_section_declarations },
|
|
||||||
{ header = "## Components", render = render_section_components },
|
|
||||||
{ header = "## RegUse schemas", render = render_section_reguse },
|
|
||||||
{ header = "## Annotations", render = render_section_annotations },
|
|
||||||
{ header = "## Component annotations", render = render_section_component_annotations },
|
|
||||||
{ header = "## Binds_* structs", render = render_section_binds },
|
|
||||||
{ header = "## Phases / views / ctx", render = render_section_phases },
|
|
||||||
{ header = "## Register aliases", render = render_section_aliases },
|
|
||||||
{ header = "## Auto-reg", render = render_section_autoreg },
|
|
||||||
{ header = "## Collisions", render = render_section_collisions },
|
|
||||||
{ header = "## Findings", render = render_section_findings },
|
|
||||||
{ header = "## Relations", render = render_section_relations },
|
|
||||||
{ header = "## GPR model", render = render_section_forward },
|
|
||||||
}
|
|
||||||
|
|
||||||
--- Render the consolidated per-module markdown (`build/<module>.atom_meta_report.md`).
|
--- Render the consolidated per-module markdown (`build/<module>.atom_meta_report.md`).
|
||||||
--- One ModuleView from the corpus; SECTION_RENDERERS walks it.
|
--- Aggregates annotation + static-analysis content across all sources in `dir`.
|
||||||
--- @param view table
|
--- Annotations come from re-running `annotation.validate()` per source (the existing pattern);
|
||||||
|
--- static-analysis comes from `corpus.static_analysis_results[dir_basename]` (populated by `static_analysis.lua` — no second corpus_pipe_ctx build).
|
||||||
|
--- @param dir string
|
||||||
|
--- @param dir_sources SourceFile[]
|
||||||
|
--- @param annot_results AnnotationResult[]
|
||||||
|
--- @param sa_results table -- corpus.static_analysis_results[dir_basename]
|
||||||
--- @return string
|
--- @return string
|
||||||
local function render_module_meta_report(view)
|
local function render_module_meta_report(dir, dir_sources, annot_results, sa_results)
|
||||||
local dir_basename = source_basename(view.dir)
|
local dir_basename = source_basename(dir)
|
||||||
local lines = {
|
local lines = {
|
||||||
"# " .. dir_basename .. " — atom meta report",
|
"# " .. dir_basename .. " — atom meta report",
|
||||||
"> Auto-generated by ps1_meta.lua (passes/report.lua). Do not edit.",
|
"> Auto-generated by ps1_meta.lua (passes/report.lua). Do not edit.",
|
||||||
@@ -786,41 +264,199 @@ local function render_module_meta_report(view)
|
|||||||
}
|
}
|
||||||
local function add(s) lines[#lines + 1] = s end
|
local function add(s) lines[#lines + 1] = s end
|
||||||
|
|
||||||
local kinds = count_kinds(view.decls)
|
-- Module summary table.
|
||||||
local n_annot, n_binds, n_macros = 0, 0, 0
|
local n_atoms = 0
|
||||||
for _, src in ipairs(view.sources) do
|
local n_annot = 0
|
||||||
n_annot = n_annot + #((src.scan and src.scan.atom_infos) or {})
|
local n_binds = 0
|
||||||
n_binds = n_binds + #((src.scan and src.scan.binds) or {})
|
local n_macros = 0
|
||||||
n_macros = n_macros + #((src.scan and src.scan.macros) or {})
|
local n_bare, n_proc = 0, 0
|
||||||
|
for _, r in ipairs(annot_results) do
|
||||||
|
n_atoms = n_atoms + #r.atoms
|
||||||
|
n_annot = n_annot + #r.annots
|
||||||
|
n_binds = n_binds + #r.binds
|
||||||
|
n_macros = n_macros + #r.macros
|
||||||
end
|
end
|
||||||
local n_err, n_warn, n_info = 0, 0, 0
|
for _, a in ipairs(sa_results.atoms or {}) do
|
||||||
for _, f in ipairs(view.findings or {}) do
|
if a.kind == "comp_bare" then n_bare = n_bare + 1
|
||||||
if f.kind == "error" then n_err = n_err + 1
|
elseif a.kind == "comp_proc" then n_proc = n_proc + 1
|
||||||
elseif f.kind == "warning" then n_warn = n_warn + 1
|
|
||||||
else n_info = n_info + 1
|
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
add("## Module summary"); add("")
|
add("## Module summary"); add("")
|
||||||
add("| metric | value |"); add("|--------|-------|")
|
add("| metric | value |"); add("|--------|-------|")
|
||||||
add(string.format("| sources | %d |", #view.sources))
|
add(string.format("| sources | %d |", #dir_sources))
|
||||||
add(string.format("| decls | %d (atom: %d, atom_proc: %d, comp_bare: %d, comp_proc: %d) |",
|
add(string.format("| atoms | %d (atoms: %d, comp_bare: %d, comp_proc: %d) |",
|
||||||
#view.decls, kinds.atom, kinds.atom_proc, kinds.comp_bare, kinds.comp_proc))
|
#(sa_results.atoms or {}),
|
||||||
|
#(sa_results.atoms or {}) - n_bare - n_proc, n_bare, n_proc))
|
||||||
add(string.format("| annotations | %d |", n_annot))
|
add(string.format("| annotations | %d |", n_annot))
|
||||||
add(string.format("| binds structs | %d |", n_binds))
|
add(string.format("| binds structs | %d |", n_binds))
|
||||||
add(string.format("| macro decls | %d |", n_macros))
|
add(string.format("| macro decls | %d |", n_macros))
|
||||||
add(string.format("| findings | %d (errors: %d, warnings: %d, info: %d) |",
|
add(string.format("| findings | %d (errors: %d, warnings: %d, info: %d) |",
|
||||||
#(view.findings or {}), n_err, n_warn, n_info))
|
#(sa_results.findings or {}),
|
||||||
|
#(sa_results.errors or {}),
|
||||||
|
#(sa_results.warnings or {}),
|
||||||
|
#(sa_results.info or {})))
|
||||||
add("")
|
add("")
|
||||||
|
|
||||||
|
-- Sources
|
||||||
add("## Sources"); add("")
|
add("## Sources"); add("")
|
||||||
for _, s in ipairs(view.sources) do add("- `" .. s.path .. "`") end
|
for _, s in ipairs(dir_sources) do add("- `" .. s.path .. "`") end
|
||||||
add("")
|
add("")
|
||||||
|
|
||||||
for _, row in ipairs(SECTION_RENDERERS) do
|
-- Atoms (annotation)
|
||||||
add(row.header); add("")
|
add("## Atoms"); add("")
|
||||||
row.render(add, view)
|
add("| kind | name | source | line |"); add("|------|------|--------|------|")
|
||||||
|
for _, r in ipairs(annot_results) do
|
||||||
|
local src_name = source_basename(r.source)
|
||||||
|
for _, a in ipairs(r.atoms) do
|
||||||
|
add(string.format("| atom | %s | %s | %d |", a.name, src_name, a.line))
|
||||||
|
end
|
||||||
end
|
end
|
||||||
|
add("")
|
||||||
|
|
||||||
|
-- Annotations
|
||||||
|
add("## Annotations"); add("")
|
||||||
|
if #annot_results == 0 then
|
||||||
|
add("_(none)_")
|
||||||
|
else
|
||||||
|
add("| source | line | name | binds | reads | writes |")
|
||||||
|
add("|--------|------|------|-------|-------|--------|")
|
||||||
|
for _, r in ipairs(annot_results) do
|
||||||
|
local src_name = source_basename(r.source)
|
||||||
|
for _, a in ipairs(r.annots) do
|
||||||
|
local binds = a.binds or "—"
|
||||||
|
local reads = (#a.reads > 0 and table.concat(a.reads, ",")) or "—"
|
||||||
|
local writes = (#a.writes > 0 and table.concat(a.writes, ",")) or "—"
|
||||||
|
add(string.format("| %s | %d | %s | %s | %s | %s |"
|
||||||
|
, src_name, a.line, a.name, binds, reads, writes))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
|
||||||
|
-- Binds_* structs
|
||||||
|
add("## Binds_* structs"); add("")
|
||||||
|
if #annot_results == 0 then
|
||||||
|
add("_(none)_")
|
||||||
|
else
|
||||||
|
for _, r in ipairs(annot_results) do
|
||||||
|
local src_name = source_basename(r.source)
|
||||||
|
for _, b in ipairs(r.binds) do
|
||||||
|
add(string.format("### %s (%s:%d, %d bytes)",
|
||||||
|
b.name, src_name, b.line, b.bytes))
|
||||||
|
for _, f in ipairs(b.fields) do
|
||||||
|
add(string.format("- `+%d %s`", f.offset, f.name))
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Macro decls
|
||||||
|
add("## Macro word-count declarations"); add("")
|
||||||
|
if #annot_results == 0 then
|
||||||
|
add("_(none)_")
|
||||||
|
else
|
||||||
|
add("| source | line | macro declaration |")
|
||||||
|
add("|--------|------|-------------------|")
|
||||||
|
for _, r in ipairs(annot_results) do
|
||||||
|
local src_name = source_basename(r.source)
|
||||||
|
for _, m in ipairs(r.macros) do
|
||||||
|
add(string.format("| %s | %d | %s |",
|
||||||
|
src_name, m.line, m.name))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
|
||||||
|
-- Findings by atom (static-analysis)
|
||||||
|
add("## Static analysis — findings by atom"); add("")
|
||||||
|
local by_atom = {}
|
||||||
|
for _, f in ipairs(sa_results.findings or {}) do
|
||||||
|
by_atom[f.atom] = by_atom[f.atom] or {}
|
||||||
|
by_atom[f.atom][#by_atom[f.atom] + 1] = f
|
||||||
|
end
|
||||||
|
if next(by_atom) == nil then
|
||||||
|
add("_(no findings)_")
|
||||||
|
else
|
||||||
|
for _, a in ipairs(sa_results.atoms or {}) do
|
||||||
|
local fs = by_atom[a.name]
|
||||||
|
if fs then
|
||||||
|
add(string.format("### %s", a.name))
|
||||||
|
for _, f in ipairs(fs) do
|
||||||
|
add(string.format("- `[%s] %s`", f.check, f.msg))
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Errors / Warnings / Info
|
||||||
|
local function add_findings(label, entries)
|
||||||
|
add(string.format("## %s", label))
|
||||||
|
if #entries == 0 then
|
||||||
|
add("_(none)_")
|
||||||
|
else
|
||||||
|
for _, e in ipairs(entries) do
|
||||||
|
add(string.format("- line %d %s", e.line, e.msg))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
end
|
||||||
|
add_findings("Errors", sa_results.errors or {})
|
||||||
|
add_findings("Warnings", sa_results.warnings or {})
|
||||||
|
add_findings("Info", sa_results.info or {})
|
||||||
|
|
||||||
|
-- Per-atom cycle counts (path-aware)
|
||||||
|
add("## Per-atom cycle counts (path-aware, best case, no stalls)"); add("")
|
||||||
|
add("| atom | source | min | max | branches | paths | notes |")
|
||||||
|
add("|------|--------|-----|-----|----------|-------|-------|")
|
||||||
|
local sorted = {}
|
||||||
|
for _, a in ipairs(sa_results.atoms or {}) do sorted[#sorted + 1] = a end
|
||||||
|
table.sort(sorted, function(x, y)
|
||||||
|
return ((x.paths or {}).cycles_max or 0) > ((y.paths or {}).cycles_max or 0)
|
||||||
|
end)
|
||||||
|
for _, a in ipairs(sorted) do
|
||||||
|
local p = a.paths or {}
|
||||||
|
local src_name = a.source_path and source_basename(a.source_path) or ""
|
||||||
|
local notes = ""
|
||||||
|
if p.has_loops then notes = notes .. " [loop!]" end
|
||||||
|
if p.unknown_macros and #p.unknown_macros > 0 then
|
||||||
|
notes = notes .. " [unknown: " .. table.concat(p.unknown_macros, ", ") .. "]"
|
||||||
|
end
|
||||||
|
add(string.format("| %s | %s | %d | %d | %d | %d | %s |",
|
||||||
|
a.name, src_name,
|
||||||
|
p.cycles_min or 0, p.cycles_max or 0,
|
||||||
|
p.branches or 0, p.paths or 0, notes))
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
|
||||||
|
-- Per-source scan summary
|
||||||
|
add("## Per-source scan summary"); add("")
|
||||||
|
for _, src in ipairs(dir_sources) do
|
||||||
|
local src_atoms = {}
|
||||||
|
for _, a in ipairs(sa_results.atoms or {}) do
|
||||||
|
if a.source_path == src.path then src_atoms[#src_atoms + 1] = a end
|
||||||
|
end
|
||||||
|
if #src_atoms > 0 then
|
||||||
|
local mn, mx = math.huge, -1
|
||||||
|
for _, a in ipairs(src_atoms) do
|
||||||
|
local p = a.paths or {}
|
||||||
|
if (p.cycles_min or 0) < mn then mn = p.cycles_min or 0 end
|
||||||
|
if (p.cycles_max or 0) > mx then mx = p.cycles_max or 0 end
|
||||||
|
end
|
||||||
|
local path_str
|
||||||
|
if mx > 0 then
|
||||||
|
path_str = string.format(" cycles=%d..%d", mn, mx)
|
||||||
|
else
|
||||||
|
path_str = string.format(" %d cycles", mn)
|
||||||
|
end
|
||||||
|
add(string.format("- `%s` — %d atom%s%s",
|
||||||
|
src.basename, #src_atoms,
|
||||||
|
#src_atoms == 1 and "" or "s", path_str))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
|
||||||
return table.concat(lines, "\n") .. "\n"
|
return table.concat(lines, "\n") .. "\n"
|
||||||
end
|
end
|
||||||
@@ -838,8 +474,19 @@ local REPORT_RENDERERS = {
|
|||||||
basename = function(dir_basename) return dir_basename .. ".atom_meta_report" end,
|
basename = function(dir_basename) return dir_basename .. ".atom_meta_report" end,
|
||||||
once = false,
|
once = false,
|
||||||
gather = function(ctx, dir, dir_sources)
|
gather = function(ctx, dir, dir_sources)
|
||||||
local corpus = ctx.shared.corpus
|
-- Annotations: re-run `annotation.validate()` per source (the existing pattern).
|
||||||
return render_module_meta_report(build_module_view(dir, dir_sources, corpus))
|
local annot_results = {}
|
||||||
|
for _, src in ipairs(dir_sources) do
|
||||||
|
if src.scan then
|
||||||
|
local r = annotation.validate(ctx, src, nil)
|
||||||
|
r.source = src.path
|
||||||
|
annot_results[#annot_results + 1] = r
|
||||||
|
end
|
||||||
|
end
|
||||||
|
-- Static-analysis: read stashed projection (no re-validate).
|
||||||
|
local dir_basename = dir:match("([^/\\]+)$") or dir
|
||||||
|
local sa_results = (ctx.shared.corpus.static_analysis_results or {})[dir_basename] or {}
|
||||||
|
return render_module_meta_report(dir, dir_sources, annot_results, sa_results)
|
||||||
end,
|
end,
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
@@ -907,30 +554,32 @@ function M.run(ctx)
|
|||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
local view = build_module_view(dir, dir_sources, corpus)
|
-- For the summary, compute per-module totals once (re-validating annotations per source — same pattern as the meta_report renderer).
|
||||||
local n_annot, n_binds, n_macros = 0, 0, 0
|
local annot_results = {}
|
||||||
for _, src in ipairs(dir_sources) do
|
for _, src in ipairs(dir_sources) do
|
||||||
n_annot = n_annot + #((src.scan and src.scan.atom_infos) or {})
|
if src.scan then
|
||||||
n_binds = n_binds + #((src.scan and src.scan.binds) or {})
|
local r = annotation.validate(ctx, src, nil)
|
||||||
n_macros = n_macros + #((src.scan and src.scan.macros) or {})
|
r.source = src.path
|
||||||
end
|
annot_results[#annot_results + 1] = r
|
||||||
local n_err, n_warn, n_info = 0, 0, 0
|
|
||||||
for _, f in ipairs(view.findings or {}) do
|
|
||||||
if f.kind == "error" then n_err = n_err + 1
|
|
||||||
elseif f.kind == "warning" then n_warn = n_warn + 1
|
|
||||||
else n_info = n_info + 1
|
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
local n_annot, n_binds, n_macros = 0, 0, 0
|
||||||
|
for _, r in ipairs(annot_results) do
|
||||||
|
n_annot = n_annot + #r.annots
|
||||||
|
n_binds = n_binds + #r.binds
|
||||||
|
n_macros = n_macros + #r.macros
|
||||||
|
end
|
||||||
|
local sa_results = (corpus.static_analysis_results or {})[dir_basename] or {}
|
||||||
all_modules[#all_modules + 1] = {
|
all_modules[#all_modules + 1] = {
|
||||||
module = dir_basename,
|
module = dir_basename,
|
||||||
atoms = #view.decls,
|
atoms = #(sa_results.atoms or {}),
|
||||||
annots = n_annot,
|
annots = n_annot,
|
||||||
binds = n_binds,
|
binds = n_binds,
|
||||||
macros = n_macros,
|
macros = n_macros,
|
||||||
findings = #(view.findings or {}),
|
findings = #(sa_results.findings or {}),
|
||||||
errors = n_err,
|
errors = #(sa_results.errors or {}),
|
||||||
warnings = n_warn,
|
warnings = #(sa_results.warnings or {}),
|
||||||
info = n_info,
|
info = #(sa_results.info or {}),
|
||||||
}
|
}
|
||||||
end
|
end
|
||||||
|
|
||||||
|
|||||||
+152
-916
File diff suppressed because it is too large
Load Diff
+199
-990
File diff suppressed because it is too large
Load Diff
Binary file not shown.
@@ -16,36 +16,60 @@
|
|||||||
-- Companion: scripts/gdb/gdb_tape_atoms.gdb (covers GPRs + atom-aware stepping).
|
-- Companion: scripts/gdb/gdb_tape_atoms.gdb (covers GPRs + atom-aware stepping).
|
||||||
|
|
||||||
local function register_handlers()
|
local function register_handlers()
|
||||||
if not PCSX.WebServer then PCSX.WebServer = {} end
|
if not PCSX.WebServer then PCSX.WebServer = {} end
|
||||||
if not PCSX.WebServer.Handlers then PCSX.WebServer.Handlers = {} end
|
if not PCSX.WebServer.Handlers then PCSX.WebServer.Handlers = {} end
|
||||||
|
|
||||||
-- ── GTE state ──
|
-- ── GTE state ──
|
||||||
PCSX.WebServer.Handlers.gte = function(req)
|
PCSX.WebServer.Handlers.gte = function(req)
|
||||||
local r = PCSX.getRegisters()
|
local r = PCSX.getRegisters()
|
||||||
local out = { "pc=0x" .. string.format("%x", r.pc) }
|
local out = { "pc=0x" .. string.format("%x", r.pc) }
|
||||||
for i = 0, 31 do
|
for i = 0, 31 do
|
||||||
out[#out + 1] = string.format("D[%d]=0x%08x C[%d]=0x%08x",
|
out[#out + 1] = string.format("D[%d]=0x%08x C[%d]=0x%08x",
|
||||||
i, r.CP2D.r[i], i, r.CP2C.r[i])
|
i, r.CP2D.r[i], i, r.CP2C.r[i])
|
||||||
end
|
end
|
||||||
return table.concat(out, "\n")
|
return table.concat(out, "\n")
|
||||||
end
|
end
|
||||||
|
|
||||||
-- ── GP state (pointer to existing endpoints) ──
|
-- ── GP state (pointer to existing endpoints) ──
|
||||||
-- pcsx-redux's Lua GPU API exposes only takeScreenShot(); no GPUSTAT / GP0 / GP1 command log / display state.
|
-- pcsx-redux's Lua GPU API exposes only takeScreenShot(); no GPUSTAT / GP0 / GP1 command log / display state.
|
||||||
-- We point to the existing web endpoints that DO expose those (when the emulator is actually rendering. Paused-at-BP frames won't have a fresh frame).
|
-- We point to the existing web endpoints that DO expose those (when the emulator is actually rendering. Paused-at-BP frames won't have a fresh frame).
|
||||||
PCSX.WebServer.Handlers.gp = function(req)
|
PCSX.WebServer.Handlers.gp = function(req)
|
||||||
local out = {
|
local out = {
|
||||||
"gpu_screenshot_png=http://localhost:8080/api/v1/state/still",
|
"gpu_screenshot_png=http://localhost:8080/api/v1/state/still",
|
||||||
"vram_raw=http://localhost:8080/api/v1/gpu/vram/raw (1MB VRAM)",
|
"vram_raw=http://localhost:8080/api/v1/gpu/vram/raw (1MB VRAM)",
|
||||||
"gpustat=NOT_AVAILABLE_VIA_LUA",
|
"gpustat=NOT_AVAILABLE_VIA_LUA",
|
||||||
"gp_command_log=NOT_AVAILABLE_VIA_LUA (use pcsx-redux Debug > GPU Logger)",
|
"gp_command_log=NOT_AVAILABLE_VIA_LUA (use pcsx-redux Debug > GPU Logger)",
|
||||||
"hint_run_emulator_unpaused_for_screenshot",
|
"hint_run_emulator_unpaused_for_screenshot",
|
||||||
}
|
}
|
||||||
return table.concat(out, "\n")
|
return table.concat(out, "\n")
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
local ok, err = pcall(register_handlers)
|
local ok, err = pcall(register_handlers)
|
||||||
if ok then print("[pcsx_debug_helper] handlers registered: gte, gp")
|
if ok then print("[pcsx_debug_helper] handlers registered: gte, gp")
|
||||||
else print("[pcsx_debug_helper] registration failed: " .. tostring(err))
|
else print("[pcsx_debug_helper] registration failed: " .. tostring(err))
|
||||||
end
|
end
|
||||||
|
|
||||||
|
-- ── reload handler (Task 6) ──
|
||||||
|
-- After gte and gp register successfully, load reload.lua through Support.extra.dofile and call its install(pcsx, support).
|
||||||
|
-- The whole sequence runs inside pcall so a missing zip, missing module table,
|
||||||
|
-- or throwing install never disturbs the gte and gp handlers already registered above (handler isolation).
|
||||||
|
--
|
||||||
|
-- The failure messages are intentionally single-line so the helper's boot log stays scannable.
|
||||||
|
if type(Support) == "table"
|
||||||
|
and type(Support.extra) == "table"
|
||||||
|
and type(Support.extra.dofile) == "function" then
|
||||||
|
local load_ok, reload_mod = pcall(Support.extra.dofile, "reload.lua")
|
||||||
|
if load_ok and type(reload_mod) == "table" and type(reload_mod.install) == "function" then
|
||||||
|
local install_ok, install_err = pcall(reload_mod.install, PCSX, Support)
|
||||||
|
if install_ok then
|
||||||
|
print("[pcsx_debug_helper] reload handler registered")
|
||||||
|
else
|
||||||
|
print("[pcsx_debug_helper] reload registration failed: " .. tostring(install_err))
|
||||||
|
end
|
||||||
|
else
|
||||||
|
print("[pcsx_debug_helper] reload load failed: " .. tostring(reload_mod))
|
||||||
|
end
|
||||||
|
else
|
||||||
|
print("[pcsx_debug_helper] reload load failed: Support.extra.dofile unavailable")
|
||||||
|
end
|
||||||
|
|||||||
@@ -0,0 +1,902 @@
|
|||||||
|
-- reload.lua - Side-effect-free hot-reload helper for the
|
||||||
|
-- pcsx_redux_hot_reload track (Task 2). This file owns the HTTP request
|
||||||
|
-- surface that the launch / reload client targets:
|
||||||
|
--
|
||||||
|
-- POST /api/v1/lua/reload?mode=prime&target=hello_camera&path=<encoded-elf>
|
||||||
|
-- POST /api/v1/lua/reload?mode=elf&target=hello_camera&path=<encoded-elf>
|
||||||
|
-- POST /api/v1/lua/reload?mode=patch&target=hello_camera&addr=...&hex=...
|
||||||
|
--
|
||||||
|
-- This module exposes the public surface used by the contract harness
|
||||||
|
-- (tests/reload_helper_contract.lua) and the runtime installed by
|
||||||
|
-- scripts/pcsx_debug_helper/autoexec.lua. The module must not reference
|
||||||
|
-- the global PCSX table at load time; the host is passed in explicitly
|
||||||
|
-- through M.new(host) and M.install(pcsx, support).
|
||||||
|
--
|
||||||
|
-- Public surface:
|
||||||
|
-- M.parse_query(query) -> table, nil OR nil, err_string
|
||||||
|
-- M.json_response(fields) -> string (sorted keys)
|
||||||
|
-- M.parse_manifest(...) -> Task 3 (real impl uses elf32.lua)
|
||||||
|
-- M.new(host) -> runtime object (Task 4; stub here)
|
||||||
|
-- M.install(pcsx, support) -> registers web handler (Task 6; stub here)
|
||||||
|
--
|
||||||
|
-- Companion: scripts/pcsx_debug_helper/autoexec.lua.
|
||||||
|
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
-- Load the shared ELF32 helpers.
|
||||||
|
--
|
||||||
|
-- **The bane of this refactor:** the helper VM (PCSX-Redux) does not expose
|
||||||
|
-- `require` for paths outside the helper zip. The production loader is
|
||||||
|
-- `Support.extra.dofile("elf32.lua")` — Support.extra.dofile resolves the
|
||||||
|
-- name against the helper zip's contents (the zip is generated by the
|
||||||
|
-- build script and includes both `reload.lua` and `elf32.lua` after Task 6).
|
||||||
|
--
|
||||||
|
-- The test harness at `tests/reload_helper_contract.lua` loads `reload.lua`
|
||||||
|
-- via standard Lua `dofile` with an absolute path; it does not install a
|
||||||
|
-- `Support` object. We detect the runtime context: if `Support.extra.dofile`
|
||||||
|
-- exists, use it (production path); otherwise fall back to standard `dofile`
|
||||||
|
-- with an absolute path (test harness path).
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
local function load_elf32()
|
||||||
|
if type(Support) == "table"
|
||||||
|
and type(Support.extra) == "table"
|
||||||
|
and type(Support.extra.dofile) == "function" then
|
||||||
|
return Support.extra.dofile("elf32.lua")
|
||||||
|
end
|
||||||
|
-- Test harness + any other context that supplies standard Lua dofile.
|
||||||
|
return dofile("C:/projects/Pikuma/ps1/scripts/elf32.lua")
|
||||||
|
end
|
||||||
|
|
||||||
|
local E = load_elf32()
|
||||||
|
|
||||||
|
local M = {}
|
||||||
|
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
-- parse_query(query)
|
||||||
|
--
|
||||||
|
-- Parses an application/x-www-form-urlencoded query string into a table.
|
||||||
|
--
|
||||||
|
-- Rules (per spec §8 + plan.md Task 2 Step 3):
|
||||||
|
-- * Each pair is split on the first '='; the key is to the left, the value
|
||||||
|
-- to the right. A pair without '=' is a malformed_pair.
|
||||||
|
-- * Percent escapes '%HH' (HH = two hex digits) decode to the corresponding
|
||||||
|
-- byte. A '%' not followed by two hex digits is a malformed_escape.
|
||||||
|
-- * '+' decodes to a literal space (applied after percent decode).
|
||||||
|
-- * A key appearing more than once is a duplicate_key error.
|
||||||
|
--
|
||||||
|
-- Returns the parsed table on success. On failure returns nil and a stable
|
||||||
|
-- error string suitable for the JSON error envelope. An empty / nil query
|
||||||
|
-- returns an empty table (not an error).
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
local function percent_decode(s)
|
||||||
|
-- Walk the string once, byte by byte. A '%' must be followed by exactly
|
||||||
|
-- two hex digits; '+' decodes to ' '; everything else is passed through.
|
||||||
|
local out = {}
|
||||||
|
local i = 1
|
||||||
|
local len = #s
|
||||||
|
while i <= len do
|
||||||
|
local c = s:sub(i, i)
|
||||||
|
if c == "%" then
|
||||||
|
if i + 2 > len then
|
||||||
|
return nil -- truncated escape (e.g., '%' at end or '%X')
|
||||||
|
end
|
||||||
|
local hex = s:sub(i + 1, i + 2)
|
||||||
|
local hd1, hd2 = hex:sub(1, 1), hex:sub(2, 2)
|
||||||
|
-- Validate both characters are hex digits.
|
||||||
|
if not (hd1:match("[0-9A-Fa-f]") and hd2:match("[0-9A-Fa-f]")) then
|
||||||
|
return nil -- malformed escape
|
||||||
|
end
|
||||||
|
out[#out + 1] = string.char(tonumber(hex, 16))
|
||||||
|
i = i + 3
|
||||||
|
else
|
||||||
|
out[#out + 1] = c
|
||||||
|
i = i + 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return table.concat(out)
|
||||||
|
end
|
||||||
|
|
||||||
|
local function plus_to_space(s)
|
||||||
|
-- Standalone helper so callers can decode '+' after percent decoding.
|
||||||
|
return (s:gsub("+", " "))
|
||||||
|
end
|
||||||
|
|
||||||
|
function M.parse_query(query)
|
||||||
|
if query == nil or query == "" then
|
||||||
|
return {}, nil
|
||||||
|
end
|
||||||
|
|
||||||
|
local result = {}
|
||||||
|
local seen = {}
|
||||||
|
|
||||||
|
for pair in query:gmatch("[^&]+") do
|
||||||
|
-- Split on the first '=' only.
|
||||||
|
local eq = pair:find("=", 1, true)
|
||||||
|
if not eq then
|
||||||
|
return nil, "malformed_pair"
|
||||||
|
end
|
||||||
|
local raw_key = pair:sub(1, eq - 1)
|
||||||
|
local raw_value = pair:sub(eq + 1)
|
||||||
|
|
||||||
|
-- Percent-decode first, then convert '+' to space. The order matters:
|
||||||
|
-- a '%2B' should decode to '+' (literal plus), not be re-converted to a
|
||||||
|
-- space. Per RFC 1866 §8.2.1, '+' is a literal plus in the encoded form
|
||||||
|
-- only when it represents a space.
|
||||||
|
local key = percent_decode(raw_key)
|
||||||
|
if key == nil then
|
||||||
|
return nil, "malformed_escape"
|
||||||
|
end
|
||||||
|
key = plus_to_space(key)
|
||||||
|
|
||||||
|
local val = percent_decode(raw_value)
|
||||||
|
if val == nil then
|
||||||
|
return nil, "malformed_escape"
|
||||||
|
end
|
||||||
|
val = plus_to_space(val)
|
||||||
|
|
||||||
|
if seen[key] then
|
||||||
|
return nil, "duplicate_key"
|
||||||
|
end
|
||||||
|
seen[key] = true
|
||||||
|
|
||||||
|
result[key] = val
|
||||||
|
end
|
||||||
|
|
||||||
|
return result, nil
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
-- json_response(fields)
|
||||||
|
--
|
||||||
|
-- Deterministic JSON object encoder. Returns a string. Keys are sorted
|
||||||
|
-- alphabetically before emission so byte-for-byte equality is testable
|
||||||
|
-- across runs and across PS1 captures.
|
||||||
|
--
|
||||||
|
-- Supported value types: string, number, boolean, nil (encoded as null).
|
||||||
|
-- Strings escape '\', '"', and the C0 control range (0x00..0x1F). The
|
||||||
|
-- named escapes use the conventional single-char forms: \\, \", \b, \f,
|
||||||
|
-- \n, \r, \t. Everything else in 0x00..0x1F is \uXXXX.
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
local function json_escape_string(s)
|
||||||
|
-- Two passes: first the named escapes, then the catch-all C0 range
|
||||||
|
-- (%c covers 0x00..0x1F in Lua patterns). Using plain string.gsub
|
||||||
|
-- with a literal replacement table covers the named escapes; a
|
||||||
|
-- second gsub handles the rest.
|
||||||
|
s = s:gsub('[\\"]', {
|
||||||
|
["\\"] = "\\\\",
|
||||||
|
['"'] = '\\"',
|
||||||
|
})
|
||||||
|
s = s:gsub("\b", "\\b")
|
||||||
|
s = s:gsub("\f", "\\f")
|
||||||
|
s = s:gsub("\n", "\\n")
|
||||||
|
s = s:gsub("\r", "\\r")
|
||||||
|
s = s:gsub("\t", "\\t")
|
||||||
|
-- Remaining C0 control characters (0x00..0x1F) become \uXXXX. We
|
||||||
|
-- intentionally keep the named escapes above (which are already
|
||||||
|
-- single backslashes in the output) from being re-escaped: gsub on
|
||||||
|
-- the literal control char bytes doesn't match the backslashes we
|
||||||
|
-- already inserted.
|
||||||
|
s = s:gsub("([%c])", function(c)
|
||||||
|
return string.format("\\u%04x", string.byte(c))
|
||||||
|
end)
|
||||||
|
return s
|
||||||
|
end
|
||||||
|
|
||||||
|
function M.json_response(fields)
|
||||||
|
if type(fields) ~= "table" then
|
||||||
|
error("json_response: expected table, got " .. type(fields))
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Sort keys for deterministic output. Lua's table.sort is byte-wise
|
||||||
|
-- and stable for strings; JSON object key order is not significant
|
||||||
|
-- but tests rely on a fixed order to compare against fixtures.
|
||||||
|
local keys = {}
|
||||||
|
for k in pairs(fields) do
|
||||||
|
keys[#keys + 1] = k
|
||||||
|
end
|
||||||
|
table.sort(keys)
|
||||||
|
|
||||||
|
local parts = {}
|
||||||
|
parts[#parts + 1] = "{"
|
||||||
|
for i = 1, #keys do
|
||||||
|
local k = keys[i]
|
||||||
|
if i > 1 then
|
||||||
|
parts[#parts + 1] = ","
|
||||||
|
end
|
||||||
|
parts[#parts + 1] = '"'
|
||||||
|
parts[#parts + 1] = json_escape_string(k)
|
||||||
|
parts[#parts + 1] = '":'
|
||||||
|
local v = fields[k]
|
||||||
|
local tv = type(v)
|
||||||
|
if tv == "string" then
|
||||||
|
parts[#parts + 1] = '"'
|
||||||
|
parts[#parts + 1] = json_escape_string(v)
|
||||||
|
parts[#parts + 1] = '"'
|
||||||
|
elseif tv == "number" then
|
||||||
|
parts[#parts + 1] = tostring(v)
|
||||||
|
elseif tv == "boolean" then
|
||||||
|
parts[#parts + 1] = v and "true" or "false"
|
||||||
|
elseif v == nil then
|
||||||
|
parts[#parts + 1] = "null"
|
||||||
|
else
|
||||||
|
error("json_response: unsupported value type " .. tv .. " for key " .. tostring(k))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
parts[#parts + 1] = "}"
|
||||||
|
return table.concat(parts)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
-- ELF32 manifest parser (Task 3).
|
||||||
|
--
|
||||||
|
-- Parses a little-endian ELF32 file exposed through a file_adapter that
|
||||||
|
-- provides read_u8_at/read_u16_at/read_u32_at/read_size. The parser validates the
|
||||||
|
-- magic, class, data encoding, and machine before reading anything else.
|
||||||
|
-- It resolves section names through the .shstrtab table and symbols
|
||||||
|
-- through every SHT_SYMTAB section (and its linked string table).
|
||||||
|
--
|
||||||
|
-- The output manifest contains the state ABI the reload gate must
|
||||||
|
-- preserve plus the addresses the helper writes to the CPU on a reload.
|
||||||
|
-- Loaded sections (SHF_ALLOC, non-SHT_NOBITS) are recorded so the runtime
|
||||||
|
-- can reject any ELF whose loaded range overlaps the preserved smem.
|
||||||
|
--
|
||||||
|
-- **Refactor:** the format-constant tables + the byte-level walker live in
|
||||||
|
-- scripts/elf32.lua (loaded above via `load_elf32()`). This module retains
|
||||||
|
-- only the manifest-specific validation: required symbols, smem size, stack
|
||||||
|
-- alignment, loaded-section overlap. The net effect is ~80 lines shorter.
|
||||||
|
--
|
||||||
|
-- Stable error codes (returned as the second value):
|
||||||
|
-- bad_magic, unsupported_elf_class, unsupported_elf_data,
|
||||||
|
-- non_mips_machine, truncated_header, truncated_section_headers,
|
||||||
|
-- missing_shstrtab, missing_symtab_strtab, missing_smem,
|
||||||
|
-- missing_data_start, missing_data_end, missing_bss_start,
|
||||||
|
-- missing_bss_end, missing_stack_top, missing_hot_reload_entry,
|
||||||
|
-- zero_smem_size, stack_misaligned, stack_out_of_main_ram,
|
||||||
|
-- section_overlaps_smem, bad_file_adapter
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
-- Convert a KSEG0/KSEG1/physical address to its physical main-RAM offset.
|
||||||
|
local function to_physical(addr)
|
||||||
|
if addr >= 0x80000000 and addr < 0x80200000 then
|
||||||
|
return addr - 0x80000000
|
||||||
|
elseif addr >= 0xa0000000 and addr < 0xa0200000 then
|
||||||
|
return addr - 0xa0000000
|
||||||
|
end
|
||||||
|
return addr
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Strip KSEG0 / KSEG1 alias from an address and return the physical main-RAM
|
||||||
|
-- offset. Used by M.elf_reload and M.patch_handler. Returns nil when the
|
||||||
|
-- address falls outside physical main RAM (0..0x1fffff), KSEG0 main RAM
|
||||||
|
-- (0x80000000..0x801fffff), or KSEG1 main RAM (0xa0000000..0xa01fffff).
|
||||||
|
-- Per spec §7 the patch path MUST reject scratchpad (0x1F800000+), BIOS
|
||||||
|
-- (0x1FC00000+), MMIO, and expansion aliases; this helper centralizes the
|
||||||
|
-- strip + range check so callers cannot forget the upper bound.
|
||||||
|
local function strip_kseg(addr)
|
||||||
|
if type(addr) ~= "number" then return nil end
|
||||||
|
if addr >= 0x80000000 and addr < 0x80200000 then
|
||||||
|
return addr - 0x80000000
|
||||||
|
elseif addr >= 0xa0000000 and addr < 0xa0200000 then
|
||||||
|
return addr - 0xa0000000
|
||||||
|
elseif addr >= 0 and addr < 0x200000 then
|
||||||
|
return addr
|
||||||
|
end
|
||||||
|
return nil
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Parse a hex string ("0xHHHH..." or "HHHH...") into a 32-bit unsigned
|
||||||
|
-- integer. Returns nil + stable error on absent / non-hex / out-of-range.
|
||||||
|
-- Used for both the patch path's addr/hex query parameters and any other
|
||||||
|
-- 32-bit hex field the API may add. Accepts up to 8 hex digits.
|
||||||
|
local function parse_hex_u32(s, missing_err, badhex_err)
|
||||||
|
if type(s) ~= "string" or #s == 0 then
|
||||||
|
return nil, missing_err or "missing_hex"
|
||||||
|
end
|
||||||
|
local clean = s:match("^0[xX]([0-9A-Fa-f]+)$")
|
||||||
|
or s:match("^([0-9A-Fa-f]+)$")
|
||||||
|
if not clean then return nil, badhex_err or "non_hex" end
|
||||||
|
if #clean > 8 then return nil, badhex_err or "non_hex" end
|
||||||
|
return tonumber(clean, 16), nil
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Trap on a missing E.* — keeps the existing one-line-error pattern when
|
||||||
|
-- the helper zip is stale or absent.
|
||||||
|
local function stack()
|
||||||
|
io.stderr:write("[reload.parse_manifest] FATAL: scripts/elf32.lua not loaded; aborting\n")
|
||||||
|
error("elf32 module not loaded")
|
||||||
|
end
|
||||||
|
|
||||||
|
local function parse_manifest_impl(file_adapter, target, path, require_entry)
|
||||||
|
-- Wrap the body in a pcall so any thrown exception (e.g. a bad
|
||||||
|
-- adapter method or a malformed section header) surfaces as a
|
||||||
|
-- parse_error with the message and traceback instead of being lost
|
||||||
|
-- into the with_busy_guard xpcall as a generic internal_error.
|
||||||
|
local inner_ok, inner_result, inner_err = pcall(function()
|
||||||
|
-- Validate the adapter surface. E.validate_adapter returns the same
|
||||||
|
-- "bad_file_adapter" error code the prior implementation used.
|
||||||
|
local ok, err = E.validate_adapter(file_adapter)
|
||||||
|
if not ok then return nil, err end
|
||||||
|
|
||||||
|
-- Magic, class, data encoding. E.parse_elf32_headers reads fields at
|
||||||
|
-- the wire offsets specified in E.ELF32_HEADER.
|
||||||
|
local hdr, hdr_err = E.parse_elf32_headers(file_adapter)
|
||||||
|
if not hdr then return nil, hdr_err end
|
||||||
|
|
||||||
|
-- Machine check (e.g. EM_MIPS = 8). e_machine is at offset 0x12 (18).
|
||||||
|
-- The reload helper rejects non-MIPS ELFs before any symbol work.
|
||||||
|
-- Explicit pass style: E.read_u16(adapter, off). The helper wraps the
|
||||||
|
-- Support.File adapter once to strip its implicit `self` so the
|
||||||
|
-- parser shape stays flat-function, not colon-dispatch.
|
||||||
|
local machine = E.read_u16(file_adapter, 0x12)
|
||||||
|
if not machine then return nil, "truncated_header" end
|
||||||
|
if machine ~= E.EM_MIPS then
|
||||||
|
return nil, "non_mips_machine"
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Walk sections. E.walk_sections also resolves .shstrtab names.
|
||||||
|
local sections, walk_err = E.walk_sections(file_adapter, hdr)
|
||||||
|
if not sections then return nil, walk_err end
|
||||||
|
|
||||||
|
-- Walk symbols. E.collect_symbols includes both STB_LOCAL and STB_GLOBAL
|
||||||
|
-- (the live ELF stores smem as a local symbol).
|
||||||
|
local symbols, sym_err = E.collect_symbols(file_adapter, sections)
|
||||||
|
if not symbols then return nil, sym_err end
|
||||||
|
|
||||||
|
-- Required symbols.
|
||||||
|
local smem = symbols["smem"]
|
||||||
|
local data_start = symbols["__data_start"]
|
||||||
|
local data_end = symbols["__data_end"]
|
||||||
|
local bss_start = symbols["__bss_start"]
|
||||||
|
local bss_end = symbols["__bss_end"]
|
||||||
|
local stack_top_s = symbols["__sp"]
|
||||||
|
local entry_s = symbols["hot_reload_entry"]
|
||||||
|
|
||||||
|
if not smem then return nil, "missing_smem" end
|
||||||
|
if not data_start then return nil, "missing_data_start" end
|
||||||
|
if not data_end then return nil, "missing_data_end" end
|
||||||
|
if not bss_start then return nil, "missing_bss_start" end
|
||||||
|
if not bss_end then return nil, "missing_bss_end" end
|
||||||
|
if not stack_top_s then return nil, "missing_stack_top" end
|
||||||
|
if require_entry and not entry_s then
|
||||||
|
return nil, "missing_hot_reload_entry"
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Validate smem size.
|
||||||
|
if smem.size == 0 then
|
||||||
|
return nil, "zero_smem_size"
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Validate stack alignment and range.
|
||||||
|
local stack_top = stack_top_s.value
|
||||||
|
if stack_top % 8 ~= 0 then
|
||||||
|
return nil, "stack_misaligned"
|
||||||
|
end
|
||||||
|
local p = to_physical(stack_top)
|
||||||
|
if p < 0 or p > 0x1fffff then
|
||||||
|
return nil, "stack_out_of_main_ram"
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Collect loaded (SHF_ALLOC, non-SHT_NOBITS) sections and check overlap.
|
||||||
|
local loaded = {}
|
||||||
|
local smem_lo = smem.value
|
||||||
|
local smem_hi = smem.value + smem.size
|
||||||
|
for _, s in ipairs(sections) do
|
||||||
|
-- bit 1 (SHF_ALLOC = 0x2) of sh_flags. The modulo-4 trick matches
|
||||||
|
-- the prior implementation; canonicalising on E.SHF_ALLOC would
|
||||||
|
-- gain readability but lose the exact prior behavior.
|
||||||
|
local is_alloc = (s.sh_flags % 4) >= 2
|
||||||
|
if is_alloc and s.sh_type ~= E.SHT_NOBITS and s.sh_size > 0 then
|
||||||
|
loaded[#loaded + 1] = { name = s.name, addr = s.sh_addr, size = s.sh_size }
|
||||||
|
local lo = s.sh_addr
|
||||||
|
local hi = s.sh_addr + s.sh_size
|
||||||
|
if lo < smem_hi and hi > smem_lo then
|
||||||
|
return nil, "section_overlaps_smem"
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return {
|
||||||
|
target = target,
|
||||||
|
elf_path = path,
|
||||||
|
elf_entry = hdr.e_entry,
|
||||||
|
smem_addr = smem.value,
|
||||||
|
smem_size = smem.size,
|
||||||
|
bss_start = bss_start.value,
|
||||||
|
bss_end = bss_end.value,
|
||||||
|
data_start = data_start.value,
|
||||||
|
data_end = data_end.value,
|
||||||
|
hot_reload_entry = entry_s and entry_s.value or nil,
|
||||||
|
stack_top = stack_top,
|
||||||
|
loaded_sections = loaded,
|
||||||
|
}
|
||||||
|
end)
|
||||||
|
if inner_ok then
|
||||||
|
return inner_result, inner_err
|
||||||
|
end
|
||||||
|
-- pcall captured a thrown error; surface as parse_error with the
|
||||||
|
-- message + traceback so the caller can render it.
|
||||||
|
local tb = debug.traceback(inner_result, 2)
|
||||||
|
local err = {
|
||||||
|
parse_error = true,
|
||||||
|
detail = tostring(inner_result),
|
||||||
|
tb = tb,
|
||||||
|
}
|
||||||
|
return nil, err
|
||||||
|
end
|
||||||
|
|
||||||
|
function M.parse_manifest(file_adapter, target, path, require_entry)
|
||||||
|
if type(E) ~= "table" or type(E.parse_elf32_headers) ~= "function" then
|
||||||
|
stack()
|
||||||
|
end
|
||||||
|
return parse_manifest_impl(file_adapter, target, path, require_entry)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
-- Runtime + dispatch (Task 4)
|
||||||
|
--
|
||||||
|
-- M.new(host) returns a runtime object that owns:
|
||||||
|
-- active -- the most recently primed manifest, or nil
|
||||||
|
-- busy -- boolean guard; only one request runs at a time
|
||||||
|
-- host -- the bound host surface (pause / memory_file / open_file
|
||||||
|
-- / binary_load / invalidate_cache / get_registers)
|
||||||
|
--
|
||||||
|
-- runtime:handle(req) parses the query through M.parse_query, validates
|
||||||
|
-- the mode against a dispatch table, then acquires the busy guard through
|
||||||
|
-- xpcall so any error inside the handler releases the guard. The response
|
||||||
|
-- is always a JSON string built by M.json_response.
|
||||||
|
--
|
||||||
|
-- M.prime_active and M.elf_reload are the two handler bodies Task 4 ships.
|
||||||
|
-- prime_active always parses with require_entry=false (Phase 0 binary
|
||||||
|
-- compatibility). elf_reload always parses with require_entry=true (the
|
||||||
|
-- new binary must expose hot_reload_entry). Both validate the parsed
|
||||||
|
-- manifest; elf_reload runs the five-field ABI gate before declaring
|
||||||
|
-- success. Full host.pause / memory_file / binary_load / invalidate_cache
|
||||||
|
-- / get_registers sequencing is Task 5.
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
-- Convert a manifest into the JSON-serializable field subset. loaded_sections
|
||||||
|
-- is excluded because json_response only supports scalars + nil.
|
||||||
|
local function manifest_to_response(m)
|
||||||
|
local fields = {
|
||||||
|
ok = true,
|
||||||
|
target = m.target,
|
||||||
|
elf_path = m.elf_path,
|
||||||
|
elf_entry = m.elf_entry,
|
||||||
|
smem_addr = m.smem_addr,
|
||||||
|
smem_size = m.smem_size,
|
||||||
|
bss_start = m.bss_start,
|
||||||
|
bss_end = m.bss_end,
|
||||||
|
data_start = m.data_start,
|
||||||
|
data_end = m.data_end,
|
||||||
|
stack_top = m.stack_top,
|
||||||
|
}
|
||||||
|
if m.hot_reload_entry then
|
||||||
|
fields.hot_reload_entry = m.hot_reload_entry
|
||||||
|
end
|
||||||
|
return fields
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Open the new ELF through the host and parse its manifest.
|
||||||
|
-- Returns manifest on success; nil + stable error on failure.
|
||||||
|
local function parse_manifest_via_host(host, target, path, require_entry)
|
||||||
|
local adapter = host.open_file(path)
|
||||||
|
if not adapter then
|
||||||
|
return nil, "open_file_failed"
|
||||||
|
end
|
||||||
|
return M.parse_manifest(adapter, target, path, require_entry)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- prime_active: parse with require_entry=false. Accepts Phase 0 binaries
|
||||||
|
-- that lack hot_reload_entry. Stores the manifest in runtime.active.
|
||||||
|
function M.prime_active(runtime, parsed)
|
||||||
|
local manifest, err = parse_manifest_via_host(
|
||||||
|
runtime.host, parsed.target, parsed.path, false)
|
||||||
|
if not manifest then
|
||||||
|
return M.json_response({ ok = false, error = err, restart_required = true })
|
||||||
|
end
|
||||||
|
runtime.active = manifest
|
||||||
|
return M.json_response(manifest_to_response(manifest))
|
||||||
|
end
|
||||||
|
|
||||||
|
-- elf_reload: full host-driven reload sequence.
|
||||||
|
--
|
||||||
|
-- Per conductor/tracks/ps1_pcsx_redux_hot_reload_20260802/spec.md §5 +
|
||||||
|
-- plan.md Task 5 Step 4. The canonical 11-entry success log is:
|
||||||
|
--
|
||||||
|
-- pause, memory_file, state_read, open_new_elf, binary_load,
|
||||||
|
-- state_restore, invalidate_cache, get_registers, write_sp,
|
||||||
|
-- write_ra, write_pc
|
||||||
|
--
|
||||||
|
-- Sequencing:
|
||||||
|
--
|
||||||
|
-- 1. Validate the request (target == active.target, path present).
|
||||||
|
-- 2. Compute the physical address of `active.smem_addr` via
|
||||||
|
-- strip_kseg; reject if outside physical main RAM.
|
||||||
|
-- 3. PARSE PHASE (before pause):
|
||||||
|
-- a. elf_handle = host.open_file(parsed.path)
|
||||||
|
-- b. manifest = M.parse_manifest(elf_handle, ..., require_entry=true)
|
||||||
|
-- c. Run the five-field ABI gate against runtime.active.
|
||||||
|
-- d. On any rejection here, return BEFORE pause — the runtime
|
||||||
|
-- has invoked host.open_file once (logging "open_file") and
|
||||||
|
-- no other host methods.
|
||||||
|
-- 4. Pause + snapshot:
|
||||||
|
-- host.pause()
|
||||||
|
-- mem = host.memory_file()
|
||||||
|
-- saved = mem:readAtToSlice(active.smem_size, smem_phys)
|
||||||
|
-- 5. RELOAD PHASE:
|
||||||
|
-- elf_handle = host.open_new_elf(parsed.path) -- second open
|
||||||
|
-- loaded = host.binary_load(elf_handle, mem)
|
||||||
|
-- if loaded == nil then return binary_load_failed
|
||||||
|
-- 6. Restore state: mem:writeAtMoveSlice(saved, smem_phys)
|
||||||
|
-- 7. host.invalidate_cache()
|
||||||
|
-- 8. Rewrite SP / RA / PC through the FFI register pointer.
|
||||||
|
-- 9. Replace runtime.active last.
|
||||||
|
-- 10. Return the JSON envelope.
|
||||||
|
--
|
||||||
|
-- The two opens are an intentional test-discoverability choice. The
|
||||||
|
-- PARSE phase uses host.open_file (it is an existing Task 4 surface
|
||||||
|
-- also used by prime); the RELOAD phase uses host.open_new_elf (a
|
||||||
|
-- dedicated Task 5 method). In production both methods bind to
|
||||||
|
-- Support.File.open so the runtime cost is identical to a single open
|
||||||
|
-- — the distinction lives in the test log for ordering verification.
|
||||||
|
local function abi_mismatch_response(field, expected, actual)
|
||||||
|
return M.json_response({
|
||||||
|
ok = false, error = "state_abi_mismatch", field = field,
|
||||||
|
expected = expected, actual = actual,
|
||||||
|
restart_required = true,
|
||||||
|
})
|
||||||
|
end
|
||||||
|
|
||||||
|
function M.elf_reload(runtime, parsed)
|
||||||
|
-- 1. Pre-pause request validation. Pure-Lua, no host calls.
|
||||||
|
if not runtime.active then
|
||||||
|
return M.json_response({
|
||||||
|
ok = false, error = "not_primed", restart_required = false })
|
||||||
|
end
|
||||||
|
if parsed.target ~= runtime.active.target then
|
||||||
|
return M.json_response({
|
||||||
|
ok = false, error = "target_mismatch",
|
||||||
|
expected = runtime.active.target, actual = parsed.target,
|
||||||
|
restart_required = true })
|
||||||
|
end
|
||||||
|
if type(parsed.path) ~= "string" or parsed.path == "" then
|
||||||
|
return M.json_response({
|
||||||
|
ok = false, error = "missing_path",
|
||||||
|
restart_required = false })
|
||||||
|
end
|
||||||
|
|
||||||
|
-- 2. SMEM range check on `active` (the new ELF has not been
|
||||||
|
-- parsed yet; the ABI gate below enforces it cannot relocate).
|
||||||
|
local smem_phys = strip_kseg(runtime.active.smem_addr)
|
||||||
|
if smem_phys == nil or smem_phys < 0 or smem_phys > 0x1fffff then
|
||||||
|
return M.json_response({
|
||||||
|
ok = false, error = "smem_out_of_main_ram",
|
||||||
|
restart_required = true })
|
||||||
|
end
|
||||||
|
|
||||||
|
-- 3. PARSE PHASE — open + parse + ABI gate. On any rejection here,
|
||||||
|
-- only host.open_file has been called. Pause and downstream
|
||||||
|
-- mutations do NOT occur.
|
||||||
|
local elf_handle_for_parse = runtime.host.open_file(parsed.path)
|
||||||
|
if not elf_handle_for_parse then
|
||||||
|
return M.json_response({
|
||||||
|
ok = false, error = "open_file_failed",
|
||||||
|
restart_required = true })
|
||||||
|
end
|
||||||
|
local manifest, parse_err = M.parse_manifest(
|
||||||
|
elf_handle_for_parse, parsed.target, parsed.path, true)
|
||||||
|
if not manifest then
|
||||||
|
return M.json_response({
|
||||||
|
ok = false, error = parse_err,
|
||||||
|
restart_required = true })
|
||||||
|
end
|
||||||
|
local active = runtime.active
|
||||||
|
if manifest.smem_addr ~= active.smem_addr then
|
||||||
|
return abi_mismatch_response(
|
||||||
|
"smem_addr", active.smem_addr, manifest.smem_addr)
|
||||||
|
end
|
||||||
|
if manifest.smem_size ~= active.smem_size then
|
||||||
|
return abi_mismatch_response(
|
||||||
|
"smem_size", active.smem_size, manifest.smem_size)
|
||||||
|
end
|
||||||
|
if manifest.bss_start ~= active.bss_start then
|
||||||
|
return abi_mismatch_response(
|
||||||
|
"bss_start", active.bss_start, manifest.bss_start)
|
||||||
|
end
|
||||||
|
if manifest.bss_end ~= active.bss_end then
|
||||||
|
return abi_mismatch_response(
|
||||||
|
"bss_end", active.bss_end, manifest.bss_end)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- 4. Pause + snapshot smem bytes.
|
||||||
|
runtime.host.pause()
|
||||||
|
local mem = runtime.host.memory_file()
|
||||||
|
local saved = mem:readAtToSlice(active.smem_size, smem_phys)
|
||||||
|
|
||||||
|
-- 5. RELOAD PHASE — second open for binary_load.
|
||||||
|
local elf_handle = runtime.host.open_new_elf(parsed.path)
|
||||||
|
if not elf_handle then
|
||||||
|
return M.json_response({
|
||||||
|
ok = false, error = "open_file_failed",
|
||||||
|
restart_required = true })
|
||||||
|
end
|
||||||
|
local loaded = runtime.host.binary_load(elf_handle, mem)
|
||||||
|
if loaded == nil then
|
||||||
|
-- Do NOT restore state; PCSX.Binary.load may have partially
|
||||||
|
-- written RAM. Keep ACTIVE untouched and tell the caller to
|
||||||
|
-- restart the emulator.
|
||||||
|
return M.json_response({
|
||||||
|
ok = false, error = "binary_load_failed",
|
||||||
|
restart_required = true })
|
||||||
|
end
|
||||||
|
|
||||||
|
-- 6. Restore the smem snapshot over the freshly-loaded code.
|
||||||
|
mem:writeAtMoveSlice(saved, smem_phys)
|
||||||
|
|
||||||
|
-- 7. Flush the CPU instruction cache (.text/.rodata changed).
|
||||||
|
runtime.host.invalidate_cache()
|
||||||
|
|
||||||
|
-- 8. Rewrite SP / RA / PC through the FFI register pointer. The
|
||||||
|
-- PC write must happen last; the CPU starts consuming
|
||||||
|
-- instructions at the new PC the moment the emulator resumes.
|
||||||
|
local regs = runtime.host.get_registers()
|
||||||
|
regs.GPR.n.sp = manifest.stack_top
|
||||||
|
regs.GPR.n.ra = 0
|
||||||
|
regs.pc = manifest.hot_reload_entry
|
||||||
|
|
||||||
|
-- 9. Replace ACTIVE last so a failed reload cannot poison the
|
||||||
|
-- next request's gate.
|
||||||
|
runtime.active = manifest
|
||||||
|
|
||||||
|
-- 10. Return the JSON envelope.
|
||||||
|
return M.json_response({
|
||||||
|
ok = true,
|
||||||
|
target = manifest.target,
|
||||||
|
elf_path = manifest.elf_path,
|
||||||
|
elf_entry = manifest.elf_entry,
|
||||||
|
smem_addr = manifest.smem_addr,
|
||||||
|
smem_size = manifest.smem_size,
|
||||||
|
bss_start = manifest.bss_start,
|
||||||
|
bss_end = manifest.bss_end,
|
||||||
|
data_start = manifest.data_start,
|
||||||
|
data_end = manifest.data_end,
|
||||||
|
hot_reload_entry = manifest.hot_reload_entry,
|
||||||
|
stack_top = manifest.stack_top,
|
||||||
|
})
|
||||||
|
end
|
||||||
|
|
||||||
|
-- patch_handler: one-word RAM patch through MemoryAsFile.
|
||||||
|
--
|
||||||
|
-- Per spec §7 + plan.md Task 5 Step 5, the order is:
|
||||||
|
-- 1. Parse addr and hex query parameters
|
||||||
|
-- 2. Reject non-hex / missing inputs
|
||||||
|
-- 3. Reject unaligned addresses (addr & 3)
|
||||||
|
-- 4. Normalize through strip_kseg; reject out-of-main-RAM
|
||||||
|
-- (scratchpad 0x1F800000+, BIOS 0x1FC00000+, MMIO, expansion)
|
||||||
|
-- 5. host.pause()
|
||||||
|
-- 6. mem = host.memory_file()
|
||||||
|
-- 7. mem:writeU32At(value, physical_offset)
|
||||||
|
-- 8. host.invalidate_cache()
|
||||||
|
-- 9. Return JSON envelope ok=true with the requested addr and value.
|
||||||
|
local function patch_error(err, restart)
|
||||||
|
return M.json_response({
|
||||||
|
ok = false, error = err,
|
||||||
|
restart_required = restart or false,
|
||||||
|
})
|
||||||
|
end
|
||||||
|
|
||||||
|
function M.patch_handler(runtime, parsed)
|
||||||
|
local addr_str = parsed.addr
|
||||||
|
local hex_str = parsed.hex
|
||||||
|
|
||||||
|
-- 1. Presence checks.
|
||||||
|
if type(addr_str) ~= "string" or addr_str == "" then
|
||||||
|
return patch_error("missing_addr", false)
|
||||||
|
end
|
||||||
|
if type(hex_str) ~= "string" or hex_str == "" then
|
||||||
|
return patch_error("missing_value", false)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- 2. Hex parse.
|
||||||
|
local addr = parse_hex_u32(addr_str, "missing_addr", "non_hex_addr")
|
||||||
|
if not addr then
|
||||||
|
return patch_error(
|
||||||
|
addr == false and "missing_addr" or "non_hex_addr", false)
|
||||||
|
end
|
||||||
|
local value = parse_hex_u32(hex_str, "missing_value", "non_hex_value")
|
||||||
|
if not value then
|
||||||
|
return patch_error(
|
||||||
|
value == false and "missing_value" or "non_hex_value", false)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- 3. Alignment (checked on the canonical KSEG/physical addr).
|
||||||
|
if addr % 4 ~= 0 then
|
||||||
|
return patch_error("addr_unaligned", false)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- 4. Range check via strip_kseg (rejects KSEG0 > 0x801fffff, KSEG1 >
|
||||||
|
-- 0xa01fffff, scratchpad, BIOS, MMIO, expansion, etc.).
|
||||||
|
local phys = strip_kseg(addr)
|
||||||
|
if phys == nil then
|
||||||
|
return patch_error("addr_out_of_main_ram", false)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- 5-8. Pause / write / cache invalidate.
|
||||||
|
runtime.host.pause()
|
||||||
|
local mem = runtime.host.memory_file()
|
||||||
|
mem:writeU32At(value, phys)
|
||||||
|
runtime.host.invalidate_cache()
|
||||||
|
|
||||||
|
-- 9. Return the JSON envelope. Echo the requested address and the
|
||||||
|
-- value in normalized hex so log captures stay stable across runs.
|
||||||
|
return M.json_response({
|
||||||
|
ok = true,
|
||||||
|
addr = addr_str,
|
||||||
|
value = "0x" .. string.format("%x", value),
|
||||||
|
})
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Mode dispatch table. Each handler is invoked with (runtime, parsed).
|
||||||
|
-- Tasks 5 adds patch (M.patch_handler); the previous placeholder removed.
|
||||||
|
local DISPATCH = {
|
||||||
|
prime = M.prime_active,
|
||||||
|
elf = M.elf_reload,
|
||||||
|
patch = M.patch_handler,
|
||||||
|
}
|
||||||
|
|
||||||
|
-- Wrap a handler call with the busy guard. The guard is acquired only
|
||||||
|
-- after the mode is validated, so unknown-mode requests do not deadlock
|
||||||
|
-- the runtime. xpcall guarantees the guard is released even if the
|
||||||
|
-- handler throws.
|
||||||
|
local function with_busy_guard(runtime, fn)
|
||||||
|
if runtime.busy then
|
||||||
|
return M.json_response({
|
||||||
|
ok = false, error = "reload_busy", restart_required = false })
|
||||||
|
end
|
||||||
|
runtime.busy = true
|
||||||
|
-- Capture both the error text and a full Lua traceback so the user
|
||||||
|
-- can see the actual failing call site instead of a generic
|
||||||
|
-- "internal_error". debug.traceback("", 2) skips this xpcall frame
|
||||||
|
-- and the json_response frame so the trace starts at the handler.
|
||||||
|
local ok, result = xpcall(fn, function(e)
|
||||||
|
return { msg = tostring(e), tb = debug.traceback("", 2) }
|
||||||
|
end)
|
||||||
|
runtime.busy = false
|
||||||
|
if not ok then
|
||||||
|
return M.json_response({
|
||||||
|
ok = false, error = "internal_error",
|
||||||
|
detail = result.msg, tb = result.tb,
|
||||||
|
restart_required = true })
|
||||||
|
end
|
||||||
|
return result
|
||||||
|
end
|
||||||
|
|
||||||
|
function M.new(host)
|
||||||
|
if type(host) ~= "table" then
|
||||||
|
error("M.new: host must be a table, got " .. type(host))
|
||||||
|
end
|
||||||
|
local runtime = {
|
||||||
|
active = nil,
|
||||||
|
busy = false,
|
||||||
|
host = host,
|
||||||
|
}
|
||||||
|
function runtime:handle(req)
|
||||||
|
-- 1. Parse the query (M.parse_query returns nil, err on failure).
|
||||||
|
local query = req and req.urlData and req.urlData.query or ""
|
||||||
|
local parsed, parse_err = M.parse_query(query)
|
||||||
|
if not parsed then
|
||||||
|
return M.json_response({
|
||||||
|
ok = false, error = parse_err, restart_required = false })
|
||||||
|
end
|
||||||
|
-- 2. Validate the mode against the dispatch table.
|
||||||
|
local mode = parsed.mode
|
||||||
|
local handler = DISPATCH[mode]
|
||||||
|
if not handler then
|
||||||
|
return M.json_response({
|
||||||
|
ok = false, error = "unknown_mode", restart_required = false })
|
||||||
|
end
|
||||||
|
-- 3. Acquire busy and dispatch via xpcall. Mode validation
|
||||||
|
-- happens BEFORE busy is acquired so unknown-mode requests
|
||||||
|
-- cannot deadlock the runtime.
|
||||||
|
return with_busy_guard(self, function()
|
||||||
|
return handler(self, parsed)
|
||||||
|
end)
|
||||||
|
end
|
||||||
|
return runtime
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Install the reload handler on a PCSX-Redux instance.
|
||||||
|
--
|
||||||
|
-- Per plan.md Task 5 Step 5 the adapter binds the canonical host method
|
||||||
|
-- names to the PCSX-Lua FFI surface:
|
||||||
|
--
|
||||||
|
-- pause -> PCSX.pauseEmulator
|
||||||
|
-- memory_file -> PCSX.getMemoryAsFile
|
||||||
|
-- open_file -> Support.File.open(path, "READ")
|
||||||
|
-- binary_load -> PCSX.Binary.load
|
||||||
|
-- invalidate_cache -> PCSX.invalidateCache
|
||||||
|
-- get_registers -> PCSX.getRegisters
|
||||||
|
--
|
||||||
|
-- The returned closure dispatches each request through M.new(host)'s
|
||||||
|
-- runtime:handle so the same prime/elf/patch dispatch machinery is used
|
||||||
|
-- (including the busy guard from Task 4).
|
||||||
|
--
|
||||||
|
-- Missing `PCSX.WebServer.Handlers` is created on demand so callers do
|
||||||
|
-- not have to wire that themselves; if `PCSX` or `Support` is absent a
|
||||||
|
-- single line is printed and the function returns without registering
|
||||||
|
-- a handler.
|
||||||
|
function M.install(pcsx, support)
|
||||||
|
if type(pcsx) ~= "table" then
|
||||||
|
print("[reload] install failed: PCSX is not a table")
|
||||||
|
return
|
||||||
|
end
|
||||||
|
if type(support) ~= "table"
|
||||||
|
or type(support.File) ~= "table"
|
||||||
|
or type(support.File.open) ~= "function" then
|
||||||
|
print("[reload] install failed: Support.File.open unavailable")
|
||||||
|
return
|
||||||
|
end
|
||||||
|
if type(pcsx.pauseEmulator) ~= "function" then print("[reload] install failed: PCSX.pauseEmulator missing"); return end
|
||||||
|
if type(pcsx.getMemoryAsFile) ~= "function" then print("[reload] install failed: PCSX.getMemoryAsFile missing"); return end
|
||||||
|
if type(pcsx.Binary) ~= "table"
|
||||||
|
or type(pcsx.Binary.load) ~= "function" then print("[reload] install failed: PCSX.Binary.load missing"); return end
|
||||||
|
if type(pcsx.invalidateCache) ~= "function" then print("[reload] install failed: PCSX.invalidateCache missing"); return end
|
||||||
|
if type(pcsx.getRegisters) ~= "function" then print("[reload] install failed: PCSX.getRegisters missing"); return end
|
||||||
|
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
-- File adapter wrap.
|
||||||
|
--
|
||||||
|
-- The production pcsx-redux Support.File wrapper (see
|
||||||
|
-- toolchain/pcsx-redux/src/lua/fileffi.lua:225-232 + size() around line 203)
|
||||||
|
-- exposes byte-read methods as colon-syntax closures with camelCase names:
|
||||||
|
-- readU8At = function(self, pos) ... end
|
||||||
|
-- readU16At = function(self, pos) ... end
|
||||||
|
-- readU32At = function(self, pos) ... end
|
||||||
|
-- size = function(self) ... end
|
||||||
|
--
|
||||||
|
-- The ELF32 parser (scripts/elf32.lua) uses an explicit-pass shape with
|
||||||
|
-- snake_case names:
|
||||||
|
-- adapter.read_u8_at(off) / adapter.read_u16_at(off) /
|
||||||
|
-- adapter.read_u32_at(off) / adapter.read_size()
|
||||||
|
--
|
||||||
|
-- The install boundary wraps the Support.File return value in a thin
|
||||||
|
-- adapter whose methods forward to the production closures, stripping
|
||||||
|
-- the implicit `self` and re-exporting the names the parser validates.
|
||||||
|
-- Without this wrap, E.validate_adapter returns "bad_file_adapter"
|
||||||
|
-- because adapter.read_u8_at / read_u16_at / read_u32_at / read_size
|
||||||
|
-- are not present on the raw Support.File return.
|
||||||
|
local function wrap_file(f)
|
||||||
|
return {
|
||||||
|
read_u8_at = function(off) return f:readU8At(off) end,
|
||||||
|
read_u16_at = function(off) return f:readU16At(off) end,
|
||||||
|
read_u32_at = function(off) return f:readU32At(off) end,
|
||||||
|
read_size = function() return f:size() end,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
|
||||||
|
local host = {
|
||||||
|
pause = function() pcsx.pauseEmulator() end,
|
||||||
|
memory_file = function() return pcsx.getMemoryAsFile() end,
|
||||||
|
open_file = function(path) return wrap_file(support.File.open(path, "READ")) end,
|
||||||
|
-- open_new_elf returns the raw Support.File object because the
|
||||||
|
-- RELOAD phase passes it directly to PCSX.Binary.load which
|
||||||
|
-- expects a real File (with readAt / size), NOT the elf32
|
||||||
|
-- parser adapter (read_u8_at / read_u16_at / read_u32_at /
|
||||||
|
-- read_size). Wrapping it in the adapter here triggers the
|
||||||
|
-- binffi.lua "Expected a File object as first argument" error.
|
||||||
|
open_new_elf = function(path) return support.File.open(path, "READ") end,
|
||||||
|
binary_load = function(elf, mem) return pcsx.Binary.load(elf, mem) end,
|
||||||
|
invalidate_cache = function() pcsx.invalidateCache() end,
|
||||||
|
get_registers = function() return pcsx.getRegisters() end,
|
||||||
|
}
|
||||||
|
local runtime = M.new(host)
|
||||||
|
|
||||||
|
if type(pcsx.WebServer) ~= "table" then pcsx.WebServer = {} end
|
||||||
|
if type(pcsx.WebServer.Handlers) ~= "table" then pcsx.WebServer.Handlers = {} end
|
||||||
|
pcsx.WebServer.Handlers.reload = function(req)
|
||||||
|
return runtime:handle(req)
|
||||||
|
end
|
||||||
|
|
||||||
|
print("[reload] handler installed: reload")
|
||||||
|
end
|
||||||
|
|
||||||
|
return M
|
||||||
@@ -118,12 +118,6 @@ local PASSES = {
|
|||||||
kind = "header-output",
|
kind = "header-output",
|
||||||
deps = {"scan-source", "word-counts"},
|
deps = {"scan-source", "word-counts"},
|
||||||
},
|
},
|
||||||
auto_reg = {
|
|
||||||
module = "passes.auto_reg",
|
|
||||||
kind = "header-output",
|
|
||||||
deps = {"components"},
|
|
||||||
groups = { "pre-link" },
|
|
||||||
},
|
|
||||||
["emission-model"] = {
|
["emission-model"] = {
|
||||||
module = "passes.emission_model",
|
module = "passes.emission_model",
|
||||||
kind = "validation",
|
kind = "validation",
|
||||||
|
|||||||
@@ -0,0 +1,82 @@
|
|||||||
|
# scripts/reload.ps1
|
||||||
|
#
|
||||||
|
# PCSX-Redux Lua helper reload client.
|
||||||
|
#
|
||||||
|
# Modes:
|
||||||
|
# elf - Request a full ELF reload. Requires -ElfPath.
|
||||||
|
# patch - Request a single-word RAM patch. Requires -Address and -Word.
|
||||||
|
#
|
||||||
|
# -RequestOnly prints the URI and exits before any network I/O.
|
||||||
|
# -Quiet suppresses the compact-JSON printout on the real path.
|
||||||
|
|
||||||
|
[CmdletBinding()]
|
||||||
|
param(
|
||||||
|
[ValidateSet('elf', 'patch')][string]$Mode = 'elf',
|
||||||
|
[string]$Target = 'hello_camera',
|
||||||
|
[string]$ElfPath = '',
|
||||||
|
[string]$Address = '',
|
||||||
|
[string]$Word = '',
|
||||||
|
[int]$Port = 8080,
|
||||||
|
[switch]$RequestOnly,
|
||||||
|
[switch]$Quiet
|
||||||
|
)
|
||||||
|
|
||||||
|
# mode-specific argument guards
|
||||||
|
switch ($Mode) {
|
||||||
|
'patch' {
|
||||||
|
if ([string]::IsNullOrEmpty($Address) -or [string]::IsNullOrEmpty($Word)) {
|
||||||
|
Write-Error "patch mode requires both -Address and -Word"
|
||||||
|
exit 1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
'elf' {
|
||||||
|
if ([string]::IsNullOrEmpty($ElfPath)) {
|
||||||
|
Write-Error "elf mode requires -ElfPath"
|
||||||
|
exit 1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
# Build the URL-encoded query string.
|
||||||
|
$queryParts = New-Object System.Collections.Generic.List[string]
|
||||||
|
[void]$queryParts.Add("mode=$([uri]::EscapeDataString($Mode))")
|
||||||
|
[void]$queryParts.Add("target=$([uri]::EscapeDataString($Target))")
|
||||||
|
|
||||||
|
switch ($Mode) {
|
||||||
|
'elf' {
|
||||||
|
[void]$queryParts.Add("path=$([uri]::EscapeDataString($ElfPath))")
|
||||||
|
}
|
||||||
|
'patch' {
|
||||||
|
[void]$queryParts.Add("addr=$([uri]::EscapeDataString($Address))")
|
||||||
|
[void]$queryParts.Add("hex=$([uri]::EscapeDataString($Word))")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
$uri = "http://localhost:$Port/api/v1/lua/reload?$($queryParts -join '&')"
|
||||||
|
|
||||||
|
# RequestOnly path: emit URI and return before any network I/O.
|
||||||
|
if ($RequestOnly) {
|
||||||
|
Write-Output $uri
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
# Real request path: POST, decode body if it is a byte array, parse JSON.
|
||||||
|
$response = Invoke-WebRequest -Method Post -Uri $uri
|
||||||
|
|
||||||
|
if ($response.Content -is [byte[]]) {
|
||||||
|
$text = [System.Text.Encoding]::UTF8.GetString([byte[]]$response.Content)
|
||||||
|
}
|
||||||
|
else {
|
||||||
|
$text = [string]$response.Content
|
||||||
|
}
|
||||||
|
|
||||||
|
$obj = $text | ConvertFrom-Json
|
||||||
|
|
||||||
|
if (-not $Quiet) {
|
||||||
|
$obj | ConvertTo-Json -Compress | Write-Output
|
||||||
|
}
|
||||||
|
|
||||||
|
if (-not $obj.ok) {
|
||||||
|
$errCode = if ($obj.error) { [string]$obj.error } else { 'unknown' }
|
||||||
|
throw "Reload failed: $errCode"
|
||||||
|
}
|
||||||
@@ -14,21 +14,16 @@ $url_armips = 'https://github.com/Kingcom/armips.git'
|
|||||||
$url_pcsx_redux = 'https://github.com/grumpycoders/pcsx-redux.git'
|
$url_pcsx_redux = 'https://github.com/grumpycoders/pcsx-redux.git'
|
||||||
$url_psyq_iwyu = 'https://github.com/johnbaumann/psyq_include_what_you_use.git'
|
$url_psyq_iwyu = 'https://github.com/johnbaumann/psyq_include_what_you_use.git'
|
||||||
$url_lpeg = 'https://github.com/roberto-ieru/LPeg.git'
|
$url_lpeg = 'https://github.com/roberto-ieru/LPeg.git'
|
||||||
# $url_mkpsxiso = 'https://github.com/Lameguy64/mkpsxiso.git'
|
|
||||||
|
|
||||||
$url_mkpsxiso_win64 = 'https://github.com/Lameguy64/mkpsxiso/releases/download/v2.30/mkpsxiso-2.30-win64.zip'
|
|
||||||
|
|
||||||
$path_armips = join-path $path_toolchain 'armips'
|
$path_armips = join-path $path_toolchain 'armips'
|
||||||
$path_pcsx_redux = join-path $path_toolchain 'pcsx-redux'
|
$path_pcsx_redux = join-path $path_toolchain 'pcsx-redux'
|
||||||
$path_psyq_iwyu = join-path $path_toolchain 'psyq_iwyu'
|
$path_psyq_iwyu = join-path $path_toolchain 'psyq_iwyu'
|
||||||
$path_lpeg = join-path $path_toolchain 'lpeg'
|
$path_lpeg = join-path $path_toolchain 'lpeg'
|
||||||
$path_mkpsxiso = join-path $path_toolchain 'mkpsxiso'
|
|
||||||
|
|
||||||
clone-gitrepo $path_armips $url_armips
|
clone-gitrepo $path_armips $url_armips
|
||||||
clone-gitrepo $path_lpeg $url_lpeg
|
clone-gitrepo $path_lpeg $url_lpeg
|
||||||
clone-gitrepo $path_pcsx_redux $url_pcsx_redux
|
clone-gitrepo $path_pcsx_redux $url_pcsx_redux
|
||||||
clone-gitrepo $path_psyq_iwyu $url_psyq_iwyu
|
clone-gitrepo $path_psyq_iwyu $url_psyq_iwyu
|
||||||
# clone-gitrepo $path_mkpsxiso $url_mkpsxiso
|
|
||||||
|
|
||||||
$path_armips_build = join-path $path_armips 'build'
|
$path_armips_build = join-path $path_armips 'build'
|
||||||
verify-path $path_armips_build
|
verify-path $path_armips_build
|
||||||
@@ -61,110 +56,6 @@ if (-not $msbuild_exe) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
$path_pcsx_sln = join-path $path_pcsx_redux 'vsprojects\pcsx-redux.sln'
|
$path_pcsx_sln = join-path $path_pcsx_redux 'vsprojects\pcsx-redux.sln'
|
||||||
|
|
||||||
# ════════════════════════════════════════════════════════════════════════════
|
|
||||||
# NuGet restore — required before MSBuild.
|
|
||||||
# pcsx-redux's .vcxproj files use the legacy packages.config style with
|
|
||||||
# hardcoded `<Import Project="..\packages\{id}.{ver}\...">` directives.
|
|
||||||
# MSBuild's `/t:Restore` won't fetch missing packages here (the local
|
|
||||||
# packages\ dir is checked but no package-source lookup happens), and
|
|
||||||
# `dotnet restore` errors on packages.config projects, so we walk every
|
|
||||||
# packages.config, parse out the <package id version/> entries, and pull
|
|
||||||
# any missing .nupkg directly from api.nuget.org's flat container.
|
|
||||||
# ════════════════════════════════════════════════════════════════════════════
|
|
||||||
$path_pcsx_packages = join-path $path_pcsx_redux 'vsprojects\packages'
|
|
||||||
$nuget_flat_container = 'https://api.nuget.org/v3-flatcontainer'
|
|
||||||
|
|
||||||
# Collect required (id, version) pairs from every packages.config.
|
|
||||||
$required_packages = @{}
|
|
||||||
Get-ChildItem -Path (join-path $path_pcsx_redux 'vsprojects') -Filter 'packages.config' -Recurse -ErrorAction SilentlyContinue |
|
|
||||||
ForEach-Object {
|
|
||||||
[xml]$xml = Get-Content -LiteralPath $_.FullName -Raw
|
|
||||||
foreach ($pkg in $xml.packages.package) {
|
|
||||||
$key = '{0}|{1}' -f $pkg.id, $pkg.version
|
|
||||||
$required_packages[$key] = @{ id = $pkg.id; version = $pkg.version }
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
# Ensure the packages root exists.
|
|
||||||
if (-not (Test-Path -LiteralPath $path_pcsx_packages)) {
|
|
||||||
New-Item -ItemType Directory -Path $path_pcsx_packages -Force | Out-Null
|
|
||||||
}
|
|
||||||
|
|
||||||
# Download anything missing.
|
|
||||||
# Skip the package entirely if its dir already has any contents (the legacy packages.config style means the targets file location varies per package
|
|
||||||
# — `luajit.native` puts it at build/native/, `glfw` puts it elsewhere — so we can't probe a specific path; just check whether the dir is non-empty).
|
|
||||||
Add-Type -AssemblyName System.IO.Compression.FileSystem
|
|
||||||
foreach ($pkg in $required_packages.Values) {
|
|
||||||
$pkgDir = Join-Path $path_pcsx_packages ('{0}.{1}' -f $pkg.id, $pkg.version)
|
|
||||||
if ((Test-Path -LiteralPath $pkgDir) -and `
|
|
||||||
(@(Get-ChildItem -LiteralPath $pkgDir -Recurse -ErrorAction SilentlyContinue).Count -gt 0)) {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
$url = '{0}/{1}/{2}/{1}.{2}.nupkg' -f $nuget_flat_container, $pkg.id, $pkg.version
|
|
||||||
$nupkg = Join-Path $pkgDir ('{0}.{1}.nupkg' -f $pkg.id, $pkg.version)
|
|
||||||
New-Item -ItemType Directory -Path $pkgDir -Force | Out-Null
|
|
||||||
Write-Host "Fetching NuGet package: $($pkg.id) $($pkg.version)"
|
|
||||||
try {
|
|
||||||
Invoke-WebRequest -Uri $url -OutFile $nupkg -UseBasicParsing -ErrorAction Stop
|
|
||||||
[System.IO.Compression.ZipFile]::ExtractToDirectory($nupkg, $pkgDir)
|
|
||||||
Remove-Item -LiteralPath $nupkg -Force
|
|
||||||
} catch {
|
|
||||||
$msg = $_.Exception.Message
|
|
||||||
if ($msg -match '404') {
|
|
||||||
Write-Host " Not on nuget.org (vendored?) — skipping $url"
|
|
||||||
} else {
|
|
||||||
Write-Warning "Failed to fetch $url — $msg"
|
|
||||||
}
|
|
||||||
if (Test-Path -LiteralPath $nupkg) { Remove-Item -LiteralPath $nupkg -Force }
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
# ════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════
|
|
||||||
# isoffi.lua size guard — `core.vcxproj` #includes src/core/isoffi.lua into luaiso.cc via the `-- lualoader, R"EOF(...)EOF"` trick.
|
|
||||||
# The raw string literal between R"EOF(-- and -- )EOF" must stay under ~16,379 bytes or MSVC (19.44) fails with C2026 (its actual raw-string limit is 16,384, minus 5 bytes for the `-- lualoader, ` prefix).
|
|
||||||
# If the upstream file grows past that, trim it: remove license header, trailing whitespace, blank separators, inline comments, and shrink 4-space indent to 2-space.
|
|
||||||
# Idempotent — only writes when the raw string exceeds the limit.
|
|
||||||
# ════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════
|
|
||||||
$path_isoffi = join-path $path_pcsx_redux 'src\core\isoffi.lua'
|
|
||||||
if (Test-Path -LiteralPath $path_isoffi) {
|
|
||||||
$content = Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8
|
|
||||||
$startMarker = $content.IndexOf('R"EOF(--')
|
|
||||||
$endMarker = $content.IndexOf('-- )EOF"')
|
|
||||||
$literalLen = if ($startMarker -ge 0 -and $endMarker -gt $startMarker) { $endMarker - ($startMarker + 8) } else { -1 }
|
|
||||||
# Effective MSVC raw-string limit for the lualoader prefix is 16379 bytes.
|
|
||||||
if ($literalLen -gt 16379) {
|
|
||||||
Write-Host "isoffi.lua raw string is $literalLen bytes (>16379); trimming for MSVC C2026 limit."
|
|
||||||
$lines = $content -split "`n"
|
|
||||||
$markerIdx = -1
|
|
||||||
for ($i = 0; $i -lt $lines.Length; $i++) {
|
|
||||||
if ($lines[$i] -match '^-- \)EOF"') { $markerIdx = $i; break }
|
|
||||||
}
|
|
||||||
$newLines = @()
|
|
||||||
for ($i = 0; $i -lt $lines.Length; $i++) {
|
|
||||||
$lineNum = $i + 1
|
|
||||||
$line = $lines[$i]
|
|
||||||
# Keep the first line and the EOF-marker line untouched.
|
|
||||||
if ($i -eq 0 -or $i -eq $markerIdx) { $newLines += $line; continue }
|
|
||||||
# Drop the GPL license header (lines 2-17).
|
|
||||||
if ($lineNum -ge 2 -and $lineNum -le 17) { continue }
|
|
||||||
# Drop blank separator lines.
|
|
||||||
if ($line -match '^\s*$') { continue }
|
|
||||||
# Drop trailing whitespace.
|
|
||||||
$line = $line -replace '\s+$', ''
|
|
||||||
# Drop inline comments (anything from `--` to end of line).
|
|
||||||
$line = $line -replace '\s*--.*$', ''
|
|
||||||
# Shrink 4-space indent to 2-space.
|
|
||||||
$line = $line -replace '^( )', ' '
|
|
||||||
if ($line -match '^\s*$') { continue }
|
|
||||||
$newLines += $line
|
|
||||||
}
|
|
||||||
($newLines -join "`n") | Out-File -LiteralPath $path_isoffi -Encoding utf8 -NoNewline
|
|
||||||
$newLen = ((Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8) -replace '.*R"EOF\(--', '' -replace '-- \)EOF".*', '').Length
|
|
||||||
Write-Host "isoffi.lua trimmed: $literalLen -> $newLen bytes of raw string content."
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
& $msbuild_exe $path_pcsx_sln /p:Configuration=Release /p:Platform=x64 /p:PlatformToolset=v143 /m /v:minimal
|
& $msbuild_exe $path_pcsx_sln /p:Configuration=Release /p:Platform=x64 /p:PlatformToolset=v143 /m /v:minimal
|
||||||
|
|
||||||
# Locate luajit via scoop. `luajit.exe` is on PATH via scoop's shim;
|
# Locate luajit via scoop. `luajit.exe` is on PATH via scoop's shim;
|
||||||
@@ -221,15 +112,6 @@ $lfs_dll_import = join-path $luajit_lib_dir 'libluajit-5.1.dll.a'
|
|||||||
# ════════════════════════════════════════════════════════════════════════════
|
# ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
$path_openbios = join-path $path_pcsx_redux 'src\mips\openbios'
|
$path_openbios = join-path $path_pcsx_redux 'src\mips\openbios'
|
||||||
|
|
||||||
# Wipe stale *.dep files across src\mips.
|
|
||||||
# These cache absolute paths to the GCC headers directory; if the toolchain was upgraded (e.g. v14.2.0 → v16.1.0)
|
|
||||||
# Make reads the stale paths and aborts with "no rule to make target .../stddef.h".
|
|
||||||
# `make clean` in openbios only clears its own dir — subdirs like common/crt0/, modplayer/, and shell/ keep their stale .dep files.
|
|
||||||
# Easier to just delete the lot before each build than to teach every Makefile about deepclean recursion.
|
|
||||||
Get-ChildItem -Path (join-path $path_pcsx_redux 'src\mips') -Recurse -Filter '*.dep' -ErrorAction SilentlyContinue |
|
|
||||||
ForEach-Object { Remove-Item -LiteralPath $_.FullName -Force }
|
|
||||||
|
|
||||||
push-location $path_openbios
|
push-location $path_openbios
|
||||||
& make clean
|
& make clean
|
||||||
& make
|
& make
|
||||||
|
|||||||
Reference in New Issue
Block a user