mirror of
https://github.com/Ed94/pikuma_ps1.git
synced 2026-09-10 18:29:06 +00:00
Compare commits
167
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a37ffe6f58 | ||
|
|
b61610d819 | ||
|
|
1b950ab5b5 | ||
|
|
b2858b3c73 | ||
|
|
f1801343e2 | ||
|
|
85b2205603 | ||
|
|
2d754650c9 | ||
|
|
e2ffe538b6 | ||
|
|
223d1832eb | ||
|
|
de13bc3ce9 | ||
|
|
2a087f735e | ||
|
|
449216967b | ||
|
|
c226e8a7d3 | ||
|
|
81f37e0098 | ||
|
|
bde829bf59 | ||
|
|
cf78cfa120 | ||
|
|
3440c9b59e | ||
|
|
1cbddc6708 | ||
|
|
b345ccd60e | ||
|
|
bbda5efaea | ||
|
|
290bb0e07a | ||
|
|
86fe189b4e | ||
|
|
da007d342e | ||
|
|
5a4bfb1224 | ||
|
|
d4795cf9de | ||
|
|
e79c364b40 | ||
|
|
18b1d5a04b | ||
|
|
581b00b960 | ||
|
|
3faccfc283 | ||
|
|
1a0d417649 | ||
|
|
3301826f5c | ||
|
|
d9b9241e2c | ||
|
|
a16c727db2 | ||
|
|
8a825a59c7 | ||
|
|
f8b28be02e | ||
|
|
ffc66052f8 | ||
|
|
7764612325 | ||
|
|
1a5b618484 | ||
|
|
d23b6a2a36 | ||
|
|
7ec778a68e | ||
|
|
9ca865d5db | ||
|
|
764ded4557 | ||
|
|
67a84d34f3 | ||
|
|
baaff12f33 | ||
|
|
b695056b9a | ||
|
|
3a4d6304dd | ||
|
|
a535d381ed | ||
|
|
c447bfa877 | ||
|
|
d88e0d0487 | ||
|
|
9a6eca6047 | ||
|
|
5c9c61720f | ||
|
|
b8e31123e4 | ||
|
|
ea3e30a11e | ||
|
|
37f4712237 | ||
|
|
b699b47b28 | ||
|
|
640dab7e61 | ||
|
|
4688566767 | ||
|
|
5ebaa6e083 | ||
|
|
7f0bdefbcb | ||
|
|
d5f28b83ea | ||
|
|
3ea3e8d105 | ||
|
|
6b60cef2e8 | ||
|
|
77f19321cd | ||
|
|
2e07665920 | ||
|
|
9501bbbcc2 | ||
|
|
9b6b5535f5 | ||
|
|
7807047dc0 | ||
|
|
7daeec0ee3 | ||
|
|
3f3b691ac0 | ||
|
|
a2d79d65eb | ||
|
|
bebcc6a585 | ||
|
|
ece21ed368 | ||
|
|
144c605ad8 | ||
|
|
4afd1af0fd | ||
|
|
004a7eff19 | ||
|
|
e42c75a26a | ||
|
|
69f2c0d036 | ||
|
|
b045856dd6 | ||
|
|
68b87f1c8b | ||
|
|
917b764d95 | ||
|
|
773aa44013 | ||
|
|
2b6fe53ce8 | ||
|
|
01f7ceba7c | ||
|
|
6f2eff920d | ||
|
|
f25765a7b7 | ||
|
|
2757aa4330 | ||
|
|
748b58c5c5 | ||
|
|
6441dbc23e | ||
|
|
57fdb9e037 | ||
|
|
b5953a723b | ||
|
|
888ffce859 | ||
|
|
7289e7c89c | ||
|
|
54a5bb9a31 | ||
|
|
e0f4ac873d | ||
|
|
f17fa9165e | ||
|
|
8282f8e902 | ||
|
|
9eb696ece8 | ||
|
|
858e57f293 | ||
|
|
afcd9b86f0 | ||
|
|
43cd4e0344 | ||
|
|
09dde54030 | ||
|
|
315e1b2c5e | ||
|
|
02658d3609 | ||
|
|
dbc459b7e0 | ||
|
|
a704341fc6 | ||
|
|
7421b32fd7 | ||
|
|
e2eb74be19 | ||
|
|
338f1fe46e | ||
|
|
27a9038e0d | ||
|
|
8c8d2e54aa | ||
|
|
80a35aa23a | ||
|
|
f247d56c32 | ||
|
|
590ff1e2ec | ||
|
|
653e18ee28 | ||
|
|
ebb876fe89 | ||
|
|
1b40b16c0e | ||
|
|
9ffd6592bc | ||
|
|
d56adab38f | ||
|
|
08af73d0d2 | ||
|
|
67d54debfa | ||
|
|
3c25306070 | ||
|
|
c3cf05950e | ||
|
|
f6b4d9895e | ||
|
|
e70361b548 | ||
|
|
ed3eb45b1d | ||
|
|
d7770b6e1d | ||
|
|
137549b1c8 | ||
|
|
7d5b13aadb | ||
|
|
2d901003f9 | ||
|
|
b43d22008e | ||
|
|
904889b483 | ||
|
|
f7aa7b75e7 | ||
|
|
aca6e30e20 | ||
|
|
8b0fb1d4e4 | ||
|
|
9f7a4a00ce | ||
|
|
277af1c901 | ||
|
|
97d2f66c5a | ||
|
|
d9406553b3 | ||
|
|
e662d175ab | ||
|
|
5387a07b84 | ||
|
|
65d805e3ba | ||
|
|
987f4dee1e | ||
|
|
df723c691d | ||
|
|
45ac85c038 | ||
|
|
072231c46b | ||
|
|
2b00956862 | ||
|
|
1ffad6cf98 | ||
|
|
318516a354 | ||
|
|
91a91b3495 | ||
|
|
a0d22700db | ||
|
|
51bdf7106b | ||
|
|
531e1cbd58 | ||
|
|
541e52de2b | ||
|
|
eccf17d21c | ||
|
|
0d94632edf | ||
|
|
798807a9c2 | ||
|
|
e9f26f89b8 | ||
|
|
a226b45d18 | ||
|
|
c22e4baa41 | ||
|
|
fa598a41c6 | ||
|
|
a928d06ac9 | ||
|
|
91c2218471 | ||
|
|
27a5f8029f | ||
|
|
7a168137fc | ||
|
|
2ceb2f2a05 | ||
|
|
6103f47f05 | ||
|
|
c824c998eb |
@@ -15,3 +15,9 @@ toolchain/PSn00bSDK
|
||||
*.a
|
||||
.sentry-native
|
||||
.vscode/settings.json
|
||||
toolchain/lfs
|
||||
toolchain/lpeg
|
||||
|
||||
scratch
|
||||
toolchain/libpsn00b
|
||||
scripts/pcsx_debug_helper.zip
|
||||
|
||||
Vendored
+26
@@ -0,0 +1,26 @@
|
||||
# Cozy and Windy
|
||||
|
||||
Editor theme ported from the Rider scheme of the same name.
|
||||
|
||||
It colors the editor surface, C/C++ syntax, and tape-atom DSL keywords
|
||||
emitted by `local.tape-atom-syntax`. It does not change workbench chrome.
|
||||
|
||||
## Install
|
||||
|
||||
```powershell
|
||||
cd C:\projects\Pikuma\ps1\.vscode\cozy-and-windy
|
||||
npm run package
|
||||
code --install-extension .\cozy-and-windy-0.1.0.vsix --force
|
||||
```
|
||||
|
||||
Reload the window. Select **Cozy and Windy** as the color theme, or set
|
||||
`workbench.colorTheme` to `Cozy and Windy` in the PS1 workspace settings.
|
||||
|
||||
Keep `local.tape-atom-syntax` installed. This theme colors those token
|
||||
types; it does not classify them.
|
||||
|
||||
## Inspect
|
||||
|
||||
Open `hello_camera.atom.c` and run **Developer: Inspect Editor Tokens and Scopes**
|
||||
on `MipsAtom_`, an atom name, `atom_info`, `R_PrimCursor`, a `gte_*` call,
|
||||
and a `mac_*` call.
|
||||
Binary file not shown.
Vendored
+25
@@ -0,0 +1,25 @@
|
||||
{
|
||||
"name": "cozy-and-windy",
|
||||
"displayName": "Cozy and Windy",
|
||||
"description": "Editor theme ported from the Rider Cozy and Windy scheme. Colors C/C++ and tape-atom DSL keywords.",
|
||||
"publisher": "local",
|
||||
"version": "0.1.0",
|
||||
"engines": {
|
||||
"vscode": "^1.80.0"
|
||||
},
|
||||
"categories": [
|
||||
"Themes"
|
||||
],
|
||||
"scripts": {
|
||||
"package": "npx --yes @vscode/vsce@3.6.1 package --allow-missing-repository --skip-license --out cozy-and-windy-0.1.0.vsix"
|
||||
},
|
||||
"contributes": {
|
||||
"themes": [
|
||||
{
|
||||
"label": "Cozy and Windy",
|
||||
"uiTheme": "vs-dark",
|
||||
"path": "./themes/cozy-and-windy-color-theme.json"
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,132 @@
|
||||
{
|
||||
"name": "Cozy and Windy",
|
||||
"type": "dark",
|
||||
"semanticHighlighting": true,
|
||||
"colors": {
|
||||
// 121212
|
||||
// 111212
|
||||
// 211f1e
|
||||
// 191817
|
||||
"editor.background": "#191817",
|
||||
"editor.foreground": "#dfc6ba",
|
||||
"editor.lineHighlightBackground": "#1c1c1c",
|
||||
"editor.selectionBackground": "#164371",
|
||||
"editor.selectionForeground": "#c8c8c8",
|
||||
"editorLineNumber.foreground": "#43c3c3",
|
||||
"editorLineNumber.activeForeground": "#00fff4",
|
||||
"editorIndentGuide.background1": "#181818",
|
||||
"editorIndentGuide.activeBackground1": "#202020",
|
||||
"editorRuler.foreground": "#505050",
|
||||
"editorGutter.background": "#211f1e",
|
||||
"editorBracketMatch.background": "#3b514d",
|
||||
"editor.foldBackground": "#0c0c0c6a",
|
||||
"editor.wordHighlightBackground": "#211f1e4d",
|
||||
"editor.wordHighlightStrongBackground": "#303030",
|
||||
"editorCursor.foreground": "#00fff4",
|
||||
"editorWhitespace.foreground": "#181818",
|
||||
// "editorLineHighlightBorder": "#1c1c1c",
|
||||
"editorWidget.background": "#211f1e",
|
||||
"editorSuggestWidget.background": "#2c334b",
|
||||
"editorHoverWidget.background": "#2c334b"
|
||||
},
|
||||
"semanticTokenColors": {
|
||||
"comment": { "foreground": "#868686", "fontStyle": "italic" },
|
||||
"keyword": { "foreground": "#d8bd5b" },
|
||||
"string": { "foreground": "#d46a54" },
|
||||
"number": { "foreground": "#b5cea8" },
|
||||
"operator": { "foreground": "#be8e78" },
|
||||
"class": { "foreground": "#54a4d6" },
|
||||
"struct": { "foreground": "#54a4d6" },
|
||||
"enum": { "foreground": "#54a4d6" },
|
||||
"type": { "foreground": "#54a4d6" },
|
||||
"interface": { "foreground": "#7984ab" },
|
||||
"function": { "foreground": "#cccab5" },
|
||||
// "function": { "foreground": "#6090a9" },
|
||||
"method": { "foreground": "#6090a9" },
|
||||
"variable": { "foreground": "#bc966c" },
|
||||
"parameter": { "foreground": "#ce8365" },
|
||||
"property": { "foreground": "#acb8c8" },
|
||||
"*.static": { "foreground": "#9e95c6" },
|
||||
"macro": { "foreground": "#5ea852" },
|
||||
"namespace": { "foreground": "#8e8e8e" },
|
||||
"typeParameter": { "foreground": "#b8d7a3" },
|
||||
"enumMember": { "foreground": "#a373b0" },
|
||||
"label": { "foreground": "#c8c8c8", "fontStyle": "bold" },
|
||||
"tapeAtomKeyword": { "foreground": "#d8bd5b", "fontStyle": "bold" },
|
||||
"tapeAtomName": { "foreground": "#b1b7d6", "fontStyle": "bold" },
|
||||
"tapeComponentKeyword": { "foreground": "#d68a36", "fontStyle": "bold" },
|
||||
"tapeComponentName": { "foreground": "#b1b7d6", "fontStyle": "bold" },
|
||||
"tapeAnnotation": { "foreground": "#d8bd5b" },
|
||||
"tapeBindType": { "foreground": "#54a4d6" },
|
||||
"tapePhase": { "foreground": "#b8d7a3", "fontStyle": "italic" },
|
||||
"tapeLabel": { "foreground": "#959595", "fontStyle": "bold" },
|
||||
// "tapeCpuInstruction": { "foreground": "#6d9aa0" },
|
||||
// "tapeCpuInstruction": { "foreground": "#cf7539" },
|
||||
// "tapeCpuInstruction": { "foreground": "#d16b3a" },
|
||||
"tapeCpuInstruction": { "foreground": "#d5895a" },
|
||||
"tapeGteInstruction": { "foreground": "#988bcb" },
|
||||
"tapeGpuInstruction": { "foreground": "#bf7dac" },
|
||||
"tapeComponentInstruction": { "foreground": "#8baa5d" },
|
||||
// "tapeGprRegister": { "foreground": "#92d4d9" },
|
||||
"tapeGprRegister": { "foreground": "#a2bfa8" },
|
||||
"tapeCop2Register": { "foreground": "#945cd9" },
|
||||
"tapeDuffleType": { "foreground": "#54a4d6" },
|
||||
"tapeAttribute": { "foreground": "#73a07c" },
|
||||
"tapeGprRegister.tapeRead": { "foreground": "#5bb8b0", "fontStyle": "italic" },
|
||||
"tapeGprRegister.tapeWrite": { "foreground": "#2d8f8c", "fontStyle": "bold" },
|
||||
"tapeCop2Register.tapeRead": { "foreground": "#b08ae0", "fontStyle": "italic" },
|
||||
"tapeCop2Register.tapeWrite": { "foreground": "#7b3ec4", "fontStyle": "bold" },
|
||||
// "*.tapeAuto": { },
|
||||
"tapeControlFlow": { "foreground": "#63d169", "fontStyle": "bold" },
|
||||
"tapeDelaySlot": { "foreground": "#ff5647" }
|
||||
},
|
||||
"tokenColors": [
|
||||
{ "scope": ["comment", "comment.block", "comment.line", "comment.block.documentation"], "settings": { "foreground": "#868686", "fontStyle": "italic" } },
|
||||
{ "scope": ["keyword", "keyword.control", "keyword.other"], "settings": { "foreground": "#d8bd5b" } },
|
||||
{ "scope": ["string", "string.quoted"], "settings": { "foreground": "#d46a54" } },
|
||||
{ "scope": ["string.quoted.other"], "settings": { "foreground": "#d69d85" } },
|
||||
{ "scope": ["constant.numeric"], "settings": { "foreground": "#b5cea8" } },
|
||||
{ "scope": ["punctuation", "keyword.operator"], "settings": { "foreground": "#be8e78" } },
|
||||
{ "scope": ["keyword.operator.overload"], "settings": { "foreground": "#b87e76" } },
|
||||
{ "scope": ["entity.name.type", "entity.name.type.class", "entity.name.type.struct", "entity.name.type.enum"], "settings": { "foreground": "#54a4d6" } },
|
||||
{ "scope": ["entity.name.type.interface"], "settings": { "foreground": "#7984ab" } },
|
||||
{ "scope": ["entity.name.function"], "settings": { "foreground": "#cccab5" } },
|
||||
{ "scope": ["entity.name.function.member"], "settings": { "foreground": "#6090a9" } },
|
||||
{ "scope": ["variable.other.local"], "settings": { "foreground": "#bc966c" } },
|
||||
{ "scope": ["variable.parameter"], "settings": { "foreground": "#ce8365" } },
|
||||
{ "scope": ["variable.other.property"], "settings": { "foreground": "#acb8c8" } },
|
||||
{ "scope": ["variable.other.constant"], "settings": { "foreground": "#9e95c6" } },
|
||||
{ "scope": ["variable.other.global", "variable.other.defaultLibrary"], "settings": { "foreground": "#bf7dac" } },
|
||||
{ "scope": ["support.type", "support.function"], "settings": { "foreground": "#8baa5d" } },
|
||||
{ "scope": ["meta.preprocessor"], "settings": { "foreground": "#5ea852" } },
|
||||
// { "scope": ["variable.parameter.preprocessor"], "settings": { "foreground": "#636363" } },
|
||||
{ "scope": ["variable.parameter.preprocessor"], "settings": { "foreground": "#bc966c" } },
|
||||
{ "scope": ["keyword.control.directive"], "settings": { "foreground": "#d68a36" } },
|
||||
{ "scope": ["entity.name.namespace"], "settings": { "foreground": "#8e8e8e" } },
|
||||
{ "scope": ["entity.name.type.parameter"], "settings": { "foreground": "#b8d7a3" } },
|
||||
{ "scope": ["variable.other.enummember"], "settings": { "foreground": "#a373b0" } },
|
||||
{ "scope": ["entity.name.type.concept"], "settings": { "foreground": "#76ff7d" } },
|
||||
{ "scope": ["entity.name.type.dependent"], "settings": { "foreground": "#448b5a", "fontStyle": "bold" } },
|
||||
{ "scope": ["entity.name.label"], "settings": { "foreground": "#c8c8c8", "fontStyle": "bold" } },
|
||||
{ "scope": ["invalid"], "settings": { "foreground": "#ff5647" } },
|
||||
{ "scope": ["keyword.codetag.todo"], "settings": { "foreground": "#c10000", "fontStyle": "bold italic" } },
|
||||
{ "scope": ["keyword.control.duffle.atom"], "settings": { "foreground": "#d8bd5b", "fontStyle": "bold" } },
|
||||
{ "scope": ["entity.name.function.duffle.atom"], "settings": { "foreground": "#cccab5", "fontStyle": "bold" } },
|
||||
{ "scope": ["keyword.control.duffle.component"], "settings": { "foreground": "#d68a36", "fontStyle": "bold" } },
|
||||
{ "scope": ["entity.name.function.duffle.component"], "settings": { "foreground": "#6090a9", "fontStyle": "bold" } },
|
||||
{ "scope": ["support.function.duffle.annotation"], "settings": { "foreground": "#d8bd5b" } },
|
||||
{ "scope": ["entity.name.type.duffle.bind"], "settings": { "foreground": "#54a4d6" } },
|
||||
{ "scope": ["entity.name.tag.duffle.phase"], "settings": { "foreground": "#b8d7a3", "fontStyle": "italic" } },
|
||||
{ "scope": ["entity.name.label.duffle.atom"], "settings": { "foreground": "#c8c8c8", "fontStyle": "bold" } },
|
||||
{ "scope": ["support.function.duffle.cpu"], "settings": { "foreground": "#6d9aa0" } },
|
||||
{ "scope": ["support.function.duffle.gte"], "settings": { "foreground": "#988bcb" } },
|
||||
{ "scope": ["support.function.duffle.gpu"], "settings": { "foreground": "#bf7dac" } },
|
||||
{ "scope": ["support.function.duffle.component"], "settings": { "foreground": "#8baa5d" } },
|
||||
{ "scope": ["keyword.control.duffle.branch"], "settings": { "foreground": "#76ff7d", "fontStyle": "bold" } },
|
||||
{ "scope": ["keyword.operator.duffle.delayslot"], "settings": { "foreground": "#ff5647" } },
|
||||
{ "scope": ["variable.other.constant.duffle.gpr"], "settings": { "foreground": "#3fa8a6" } },
|
||||
{ "scope": ["variable.other.constant.duffle.cop2"], "settings": { "foreground": "#945cd9" } },
|
||||
{ "scope": ["storage.type.duffle.type"], "settings": { "foreground": "#54a4d6" } },
|
||||
{ "scope": ["storage.modifier.duffle.attr"], "settings": { "foreground": "#73a07c" } }
|
||||
]
|
||||
}
|
||||
Vendored
+43
@@ -0,0 +1,43 @@
|
||||
# Package and install the local VS Code Insiders extensions under .vscode/.
|
||||
# Usage:
|
||||
# .\install_extensions.ps1
|
||||
# .\install_extensions.ps1 -SkipPackage
|
||||
|
||||
param([switch] $SkipPackage)
|
||||
|
||||
$path_vscode = $PSScriptRoot
|
||||
$code_insiders = "C:\apps\Microsoft VS Code Insiders\bin\code-insiders.cmd"
|
||||
if (-not (test-path -literalpath $code_insiders)) {
|
||||
$found = get-command code-insiders -erroraction silentlycontinue
|
||||
if ($found) { $code_insiders = $found.source }
|
||||
}
|
||||
|
||||
if (-not (test-path -literalpath $code_insiders)) { throw "code-insiders not found. Install VS Code Insiders or add it to PATH." }
|
||||
|
||||
$extensions = @(
|
||||
(join-path $path_vscode "tape-atom-syntax"),
|
||||
(join-path $path_vscode "cozy-and-windy")
|
||||
)
|
||||
|
||||
foreach ($extension in $extensions) {
|
||||
$package_json = join-path $extension "package.json"
|
||||
if (-not (test-path -literalpath $package_json)) { throw "missing $package_json" }
|
||||
|
||||
$manifest = get-content -literalpath $package_json -raw | convertfrom-json
|
||||
$vsix = join-path $extension ("{0}-{1}.vsix" -f $manifest.name, $manifest.version)
|
||||
|
||||
if (-not $SkipPackage) {
|
||||
if (-not $manifest.scripts.package) { throw "$package_json has no scripts.package" }
|
||||
write-host "packaging $($manifest.displayName) ($($manifest.name)@$($manifest.version))"
|
||||
& npm --prefix $extension run package
|
||||
if ($LASTEXITCODE -ne 0) { throw "npm run package failed for $extension" }
|
||||
}
|
||||
|
||||
if (-not (test-path -literalpath $vsix)) { throw "missing $vsix" }
|
||||
|
||||
write-host "installing $vsix"
|
||||
& $code_insiders --install-extension $vsix --force
|
||||
if ($LASTEXITCODE -ne 0) { throw "install failed for $vsix" }
|
||||
}
|
||||
|
||||
write-host "done. reload the Insiders window (Developer: Reload Window)."
|
||||
Vendored
+130
-27
@@ -4,7 +4,7 @@
|
||||
// For more information, visit: https://go.microsoft.com/fwlink/?linkid=830387
|
||||
"version": "0.2.0",
|
||||
"configurations": [
|
||||
{
|
||||
{
|
||||
"name": "Debug: Hello Psy-Q!",
|
||||
"type": "gdb",
|
||||
"request": "attach",
|
||||
@@ -12,6 +12,10 @@
|
||||
"remote": true,
|
||||
"cwd": "${workspaceRoot}/build",
|
||||
"valuesFormatting": "parseText",
|
||||
"registerLimit": "1-32",
|
||||
"frameFilters": false,
|
||||
"showDevDebugOutput": false,
|
||||
"printCalls": false,
|
||||
"stopAtConnect": true,
|
||||
"gdbpath": "gdb-multiarch",
|
||||
"windows": {
|
||||
@@ -20,10 +24,17 @@
|
||||
"osx": {
|
||||
"gdbpath": "gdb"
|
||||
},
|
||||
"executable": "${workspaceRoot}/build/hello_psyq.elf",
|
||||
"executable": "${workspaceRoot}/build/hello_gte.elf",
|
||||
"setupCommands": [
|
||||
{ "text": "set mi-async off" },
|
||||
{ "text": "set remotetimeout 0" },
|
||||
{ "text": "set logging file build/gen/hello_gte.gdb.log" },
|
||||
{ "text": "set logging redirect on" }
|
||||
],
|
||||
"autorun": [
|
||||
"monitor reset shellhalt",
|
||||
"load hello_psyq.elf",
|
||||
"load hello_gte.elf",
|
||||
"source scripts/gdb/gdb_tape_atoms.gdb",
|
||||
"tbreak main",
|
||||
"continue"
|
||||
]
|
||||
@@ -36,30 +47,10 @@
|
||||
"remote": true,
|
||||
"cwd": "${workspaceRoot}/build",
|
||||
"valuesFormatting": "parseText",
|
||||
"stopAtConnect": true,
|
||||
"gdbpath": "gdb-multiarch",
|
||||
"windows": {
|
||||
"gdbpath": "gdb-multiarch.exe"
|
||||
},
|
||||
"osx": {
|
||||
"gdbpath": "gdb"
|
||||
},
|
||||
"executable": "${workspaceRoot}/build/hello_gpu.elf",
|
||||
"autorun": [
|
||||
"monitor reset shellhalt",
|
||||
"load hello_gpu.elf",
|
||||
"tbreak main",
|
||||
"continue"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "Debug: Hello GTE Psy-Q!",
|
||||
"type": "gdb",
|
||||
"request": "attach",
|
||||
"target": "localhost:3333",
|
||||
"remote": true,
|
||||
"cwd": "${workspaceRoot}/build",
|
||||
"valuesFormatting": "parseText",
|
||||
"registerLimit": "1-32",
|
||||
"frameFilters": false,
|
||||
"showDevDebugOutput": false,
|
||||
"printCalls": false,
|
||||
"stopAtConnect": true,
|
||||
"gdbpath": "gdb-multiarch",
|
||||
"windows": {
|
||||
@@ -69,12 +60,124 @@
|
||||
"gdbpath": "gdb"
|
||||
},
|
||||
"executable": "${workspaceRoot}/build/hello_gte.elf",
|
||||
"setupCommands": [
|
||||
{ "text": "set mi-async off" },
|
||||
{ "text": "set remotetimeout 0" },
|
||||
{ "text": "set logging file build/gen/hello_gte.gdb.log" },
|
||||
{ "text": "set logging redirect on" }
|
||||
],
|
||||
"autorun": [
|
||||
"monitor reset shellhalt",
|
||||
"load hello_gte.elf",
|
||||
"tbreak main",
|
||||
"continue"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "Debug: Hello GTE!",
|
||||
"type": "gdb",
|
||||
"request": "attach",
|
||||
"target": "localhost:3333",
|
||||
"remote": true,
|
||||
"cwd": "${workspaceRoot}",
|
||||
"valuesFormatting": "parseText",
|
||||
"registerLimit": "1-32",
|
||||
"frameFilters": false,
|
||||
"showDevDebugOutput": false,
|
||||
"printCalls": false,
|
||||
"stopAtConnect": true,
|
||||
"gdbpath": "gdb-multiarch",
|
||||
"windows": {
|
||||
"gdbpath": "gdb-multiarch.exe"
|
||||
},
|
||||
"osx": {
|
||||
"gdbpath": "gdb"
|
||||
},
|
||||
"executable": "${workspaceRoot}/build/hello_gte.dwarf-injected.elf",
|
||||
"setupCommands": [
|
||||
{ "text": "set mi-async off" },
|
||||
{ "text": "set remotetimeout 0" },
|
||||
{ "text": "set logging file build/gen/hello_gte.gdb.log" },
|
||||
{ "text": "set logging redirect on" }
|
||||
],
|
||||
"autorun": [
|
||||
"monitor reset shellhalt",
|
||||
"load build/hello_gte.dwarf-injected.elf",
|
||||
"source scripts/gdb/gdb_tape_atoms.gdb",
|
||||
"tbreak main",
|
||||
"continue"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "Debug: Hello Joypad!",
|
||||
"type": "gdb",
|
||||
"request": "attach",
|
||||
"target": "localhost:3333",
|
||||
"remote": true,
|
||||
"cwd": "${workspaceRoot}",
|
||||
"valuesFormatting": "parseText",
|
||||
"registerLimit": "1-32",
|
||||
"frameFilters": false,
|
||||
"showDevDebugOutput": false,
|
||||
"printCalls": false,
|
||||
"stopAtConnect": true,
|
||||
"gdbpath": "gdb-multiarch",
|
||||
"windows": {
|
||||
"gdbpath": "gdb-multiarch.exe"
|
||||
},
|
||||
"osx": {
|
||||
"gdbpath": "gdb"
|
||||
},
|
||||
"executable": "${workspaceRoot}/build/hello_joypad.dwarf-injected.elf",
|
||||
"setupCommands": [
|
||||
{ "text": "set mi-async off" },
|
||||
{ "text": "set remotetimeout 0" },
|
||||
{ "text": "set logging file build/gen/hello_joypad.gdb.log" },
|
||||
{ "text": "set logging redirect on" }
|
||||
],
|
||||
"autorun": [
|
||||
"monitor reset shellhalt",
|
||||
"load build/hello_joypad.dwarf-injected.elf",
|
||||
"source scripts/gdb/gdb_tape_atoms.gdb",
|
||||
"tbreak main",
|
||||
"continue"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "Debug: Hello Camera!",
|
||||
"type": "gdb",
|
||||
"request": "attach",
|
||||
"target": "localhost:3333",
|
||||
"remote": true,
|
||||
"cwd": "${workspaceRoot}",
|
||||
"valuesFormatting": "parseText",
|
||||
"registerLimit": "1-32",
|
||||
"frameFilters": false,
|
||||
"showDevDebugOutput": false,
|
||||
"printCalls": false,
|
||||
"stopAtConnect": true,
|
||||
"gdbpath": "gdb-multiarch",
|
||||
"windows": {
|
||||
"gdbpath": "gdb-multiarch.exe"
|
||||
},
|
||||
"osx": {
|
||||
"gdbpath": "gdb"
|
||||
},
|
||||
"executable": "${workspaceRoot}/build/hello_camera.dwarf-injected.elf",
|
||||
"setupCommands": [
|
||||
{ "text": "set mi-async off" },
|
||||
{ "text": "set remotetimeout 0" },
|
||||
{ "text": "set logging file build/gen/hello_camera.gdb.log" },
|
||||
{ "text": "set logging redirect on" }
|
||||
],
|
||||
"autorun": [
|
||||
"monitor reset shellhalt",
|
||||
"load build/hello_camera.dwarf-injected.elf",
|
||||
"source scripts/gdb/gdb_tape_atoms.gdb",
|
||||
"tbreak main",
|
||||
"continue"
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
BIN
Binary file not shown.
+222
@@ -0,0 +1,222 @@
|
||||
"use strict";
|
||||
|
||||
const { nearestCall } = require("./lexer");
|
||||
const { mergeIndexes, scanSource } = require("./source-index");
|
||||
|
||||
const TOKEN_TYPES = [
|
||||
"tapeAtomKeyword",
|
||||
"tapeAtomName",
|
||||
"tapeComponentKeyword",
|
||||
"tapeComponentName",
|
||||
"tapeAnnotation",
|
||||
"tapeBindType",
|
||||
"tapePhase",
|
||||
"tapeLabel",
|
||||
"tapeCpuInstruction",
|
||||
"tapeControlFlow",
|
||||
"tapeGteInstruction",
|
||||
"tapeGpuInstruction",
|
||||
"tapeComponentInstruction",
|
||||
"tapeDelaySlot",
|
||||
"tapeGprRegister",
|
||||
"tapeCop2Register",
|
||||
"tapeDuffleType",
|
||||
"tapeAttribute",
|
||||
"keyword",
|
||||
"macro",
|
||||
];
|
||||
|
||||
const TOKEN_MODIFIERS = ["declaration", "tapeRead", "tapeWrite", "tapeAuto"];
|
||||
const TOKEN_TYPE_INDEX = new Map(TOKEN_TYPES.map((name, index) => [name, index]));
|
||||
const TOKEN_MODIFIER_INDEX = new Map(TOKEN_MODIFIERS.map((name, index) => [name, index]));
|
||||
|
||||
const ATOM_KEYWORDS = new Set(["MipsAtom_", "MipsAtom_Proc_"]);
|
||||
const COMPONENT_KEYWORDS = new Set(["MipsAtomComp_", "MipsAtomComp_Proc_"]);
|
||||
const ANNOTATIONS = new Set([
|
||||
"atom_info", "atom_bind", "atom_reads", "atom_writes", "atom_label",
|
||||
"atom_offset", "atom_reg", "atom_type", "atom_ctx", "atom_phase",
|
||||
"atom_auto_reg", "phase_auto_reg", "atom_dbg_skip",
|
||||
]);
|
||||
|
||||
const DSL_KEYWORDS = new Set([
|
||||
"FI_", "I_", "NI_", "Relative_", "Struct_", "Enum_", "Union_", "Array_",
|
||||
"Slice_", "TypeR_", "TypeV_", "align_", "internal", "local_persist", "global",
|
||||
"RO_", "LP_", "gknown", "expect_", "cexpr_",
|
||||
"asm", "asm_words", "asm_rpins", "asm_clobber",
|
||||
"O_", "S_", "C_", "T_", "tmpl", "glue", "r_", "v_", "tr_", "tv_",
|
||||
"rgcc", "r_use", "r_set", "r_mod", "r_imm", "r_mem",
|
||||
"u1_", "u2_", "u4_", "u8_", "s1_", "s2_", "s4_", "s8_",
|
||||
"u1_r", "u2_r", "u4_r", "u8_r", "u1_v", "u2_v", "u4_v", "u8_v",
|
||||
]);
|
||||
|
||||
const DELAY_SLOT_KEYWORDS = new Set(["LdSlot_", "BdSlot_", "DmaSlot_", "GteDelay_"]);
|
||||
|
||||
const CONTROL_FLOW_PREFIXES = /^(?:branch_|jump_|call_)/;
|
||||
|
||||
const ROLE_TO_TYPE = {
|
||||
atomName: "tapeAtomName",
|
||||
componentName: "tapeComponentName",
|
||||
bindType: "tapeBindType",
|
||||
duffleType: "tapeDuffleType",
|
||||
gprRegister: "tapeGprRegister",
|
||||
cop2Register: "tapeCop2Register",
|
||||
};
|
||||
|
||||
function registerType(name, index) {
|
||||
const kind = index.registers.get(name);
|
||||
if (kind === "gpr" || /^R_[A-Za-z0-9_]+$/.test(name)) return "tapeGprRegister";
|
||||
if (kind === "cop2" || /^(?:C2_|gte_cr_)[A-Za-z0-9_]+$/.test(name)) return "tapeCop2Register";
|
||||
return null;
|
||||
}
|
||||
|
||||
function instructionType(name, index) {
|
||||
const domain = index.macros.get(name);
|
||||
if (domain === "control") return "tapeControlFlow";
|
||||
if (domain === "cpu") return "tapeCpuInstruction";
|
||||
if (domain === "gte") return "tapeGteInstruction";
|
||||
if (domain === "gpu") return "tapeGpuInstruction";
|
||||
if (domain === "component") {
|
||||
if (/^mac_gte_/.test(name)) return "tapeGteInstruction";
|
||||
if (/^mac_gp/.test(name)) return "tapeGpuInstruction";
|
||||
if (/^mac_/.test(name)) return "tapeComponentInstruction";
|
||||
return "macro";
|
||||
}
|
||||
if (domain === "utility") return "macro";
|
||||
if (/^gte_(?!cr_)/.test(name)) return "tapeGteInstruction";
|
||||
if (/^gp[01]_/.test(name)) return "tapeGpuInstruction";
|
||||
if (/^mac_gte_/.test(name)) return "tapeGteInstruction";
|
||||
if (/^mac_gp/.test(name)) return "tapeGpuInstruction";
|
||||
if (/^mac_/.test(name)) return "tapeComponentInstruction";
|
||||
return null;
|
||||
}
|
||||
|
||||
function modifierMask(modifiers) {
|
||||
let mask = 0;
|
||||
for (const modifier of modifiers) {
|
||||
const index = TOKEN_MODIFIER_INDEX.get(modifier);
|
||||
if (index !== undefined) mask |= (1 << index);
|
||||
}
|
||||
return mask;
|
||||
}
|
||||
|
||||
function isRegUseAccess(tokens, tokenIndex) {
|
||||
const prev = tokens[tokenIndex - 1];
|
||||
if (!prev || prev.text !== ".") return false;
|
||||
const prevPrev = tokens[tokenIndex - 2];
|
||||
if (!prevPrev || prevPrev.kind !== "identifier") return false;
|
||||
const next = tokens[tokenIndex + 1];
|
||||
if (next && next.text === ".") return false;
|
||||
if (prevPrev.text === "r") return true;
|
||||
const prev3 = tokens[tokenIndex - 3];
|
||||
const prev4 = tokens[tokenIndex - 4];
|
||||
if (prev3 && prev3.text === "." && prev4 && prev4.kind === "identifier" && prev4.text === "r") return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
function classifyDocument(source, filePath, workspaceIndex, shouldCancel = () => false) {
|
||||
const scanned = scanSource(source, filePath);
|
||||
const index = mergeIndexes(workspaceIndex, scanned.index);
|
||||
const spans = [];
|
||||
|
||||
for (let tokenIndex = 0; tokenIndex < scanned.tokens.length; tokenIndex += 1) {
|
||||
if (shouldCancel()) break;
|
||||
const token = scanned.tokens[tokenIndex];
|
||||
if (token.kind !== "identifier") continue;
|
||||
|
||||
let type = null;
|
||||
let modifiers = [];
|
||||
const declaration = scanned.declarations.get(token.start);
|
||||
const context = nearestCall(scanned.contexts, tokenIndex);
|
||||
|
||||
if (declaration) {
|
||||
type = ROLE_TO_TYPE[declaration.role] || null;
|
||||
modifiers = declaration.modifiers.slice();
|
||||
} else if (ATOM_KEYWORDS.has(token.text)) {
|
||||
type = "tapeAtomKeyword";
|
||||
} else if (COMPONENT_KEYWORDS.has(token.text)) {
|
||||
type = "keyword";
|
||||
} else if (ANNOTATIONS.has(token.text)) {
|
||||
type = "tapeAnnotation";
|
||||
} else if (context && context.callee === "atom_bind" && context.argIndex === 0) {
|
||||
type = "tapeBindType";
|
||||
} else if (context && context.callee === "atom_phase" && context.argIndex === 0) {
|
||||
type = "tapePhase";
|
||||
modifiers = ["declaration"];
|
||||
} else if (context && context.callee === "atom_ctx" && context.argIndex === 0) {
|
||||
type = "tapeAtomName";
|
||||
} else if (context && context.callee === "atom_label" && context.argIndex === 0) {
|
||||
type = "tapeLabel";
|
||||
modifiers = ["declaration"];
|
||||
} else if (context && context.callee === "atom_offset" && context.argIndex <= 1) {
|
||||
type = "tapeLabel";
|
||||
} else if (context && context.callee === "atom_reads") {
|
||||
type = registerType(token.text, index);
|
||||
if (type) modifiers = ["tapeRead"];
|
||||
} else if (context && context.callee === "atom_writes") {
|
||||
type = registerType(token.text, index);
|
||||
if (type) modifiers = ["tapeWrite"];
|
||||
} else if (context && context.callee === "atom_auto_reg") {
|
||||
if (context.argIndex === 0) type = "tapeAtomName";
|
||||
if (context.argIndex === 1) {
|
||||
type = "tapeGprRegister";
|
||||
modifiers = ["declaration", "tapeAuto"];
|
||||
}
|
||||
} else if (context && context.callee === "phase_auto_reg") {
|
||||
if (context.argIndex === 0) type = "tapePhase";
|
||||
if (context.argIndex === 1) {
|
||||
type = "tapeGprRegister";
|
||||
modifiers = ["declaration", "tapeAuto"];
|
||||
}
|
||||
}
|
||||
|
||||
if (!type && index.bindTypes.has(token.text)) type = "tapeBindType";
|
||||
if (!type && DSL_KEYWORDS.has(token.text)) type = "keyword";
|
||||
if (!type && index.types.has(token.text)) type = "tapeDuffleType";
|
||||
if (!type && index.attributes.has(token.text)) type = "tapeAttribute";
|
||||
if (!type) type = registerType(token.text, index);
|
||||
if (!type && DELAY_SLOT_KEYWORDS.has(token.text)) type = "tapeDelaySlot";
|
||||
if (!type) {
|
||||
const domain = index.macros.get(token.text);
|
||||
if (domain === "control" || (domain && CONTROL_FLOW_PREFIXES.test(token.text))) {
|
||||
type = "tapeControlFlow";
|
||||
}
|
||||
}
|
||||
if (!type && isRegUseAccess(scanned.tokens, tokenIndex)) type = "tapeGprRegister";
|
||||
if (!type) type = instructionType(token.text, index);
|
||||
if (!type && /^(?:Slice_|A[0-9]+_)/.test(token.text)) type = "tapeDuffleType";
|
||||
if (!type && /_[RV]$/.test(token.text)) type = "tapeDuffleType";
|
||||
if (!type && index.atoms.has(token.text)) type = "tapeAtomName";
|
||||
if (!type && index.components.has(token.text)) type = "tapeComponentName";
|
||||
if (!type && index.phases.has(token.text)) type = "tapePhase";
|
||||
if (!type && index.labels.has(token.text)) type = "tapeLabel";
|
||||
if (!type) continue;
|
||||
|
||||
spans.push({
|
||||
text: token.text,
|
||||
type,
|
||||
typeIndex: TOKEN_TYPE_INDEX.get(type),
|
||||
modifiers,
|
||||
modifierMask: modifierMask(modifiers),
|
||||
start: token.start,
|
||||
length: token.end - token.start,
|
||||
line: token.line,
|
||||
character: token.character,
|
||||
});
|
||||
}
|
||||
|
||||
spans.sort((left, right) => left.start - right.start || left.length - right.length);
|
||||
const nonOverlapping = [];
|
||||
for (const span of spans) {
|
||||
const previous = nonOverlapping[nonOverlapping.length - 1];
|
||||
if (!previous || previous.start + previous.length <= span.start) nonOverlapping.push(span);
|
||||
}
|
||||
|
||||
return { spans: nonOverlapping, errors: scanned.errors };
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
TOKEN_MODIFIERS,
|
||||
TOKEN_TYPES,
|
||||
classifyDocument,
|
||||
modifierMask,
|
||||
};
|
||||
+111
@@ -0,0 +1,111 @@
|
||||
"use strict";
|
||||
|
||||
const vscode = require("vscode");
|
||||
const { TOKEN_MODIFIERS, TOKEN_TYPES, classifyDocument } = require("./classifier");
|
||||
const { createIndex, mergeIndexes, scanSource } = require("./source-index");
|
||||
|
||||
const SOURCE_GLOB = "**/*.{c,h,cc,cpp,cxx,hh,hpp,hxx}";
|
||||
const EXCLUDE_GLOB = "**/{gen,build,.slop_cache,toolchain,node_modules}/**";
|
||||
const EXCLUDED_SEGMENTS = new Set(["gen", "build", ".slop_cache", "toolchain", "node_modules"]);
|
||||
|
||||
function isExcluded(uri) {
|
||||
const segments = uri.fsPath.replaceAll("\\", "/").split("/");
|
||||
return segments.some((segment) => EXCLUDED_SEGMENTS.has(segment));
|
||||
}
|
||||
|
||||
function formatError(filePath, error) {
|
||||
return `${filePath}:${error.offset}: ${error.kind}`;
|
||||
}
|
||||
|
||||
async function activate(context) {
|
||||
const output = vscode.window.createOutputChannel("Tape Atom DSL");
|
||||
const emitter = new vscode.EventEmitter();
|
||||
const legend = new vscode.SemanticTokensLegend(TOKEN_TYPES, TOKEN_MODIFIERS);
|
||||
let workspaceIndex = createIndex();
|
||||
let rebuildGeneration = 0;
|
||||
let debounceHandle = null;
|
||||
|
||||
async function rebuildIndex() {
|
||||
const generation = ++rebuildGeneration;
|
||||
const files = await vscode.workspace.findFiles(SOURCE_GLOB, EXCLUDE_GLOB);
|
||||
let nextIndex = createIndex();
|
||||
|
||||
for (const uri of files) {
|
||||
if (generation !== rebuildGeneration) return;
|
||||
if (isExcluded(uri)) continue;
|
||||
try {
|
||||
const bytes = await vscode.workspace.fs.readFile(uri);
|
||||
const source = Buffer.from(bytes).toString("utf8");
|
||||
const result = scanSource(source, uri.fsPath);
|
||||
nextIndex = mergeIndexes(nextIndex, result.index);
|
||||
for (const error of result.errors) output.appendLine(formatError(uri.fsPath, error));
|
||||
} catch (error) {
|
||||
output.appendLine(`${uri.fsPath}: ${error.stack || error.message || error}`);
|
||||
}
|
||||
}
|
||||
|
||||
if (generation !== rebuildGeneration) return;
|
||||
workspaceIndex = nextIndex;
|
||||
emitter.fire();
|
||||
}
|
||||
|
||||
function scheduleRebuild(uri) {
|
||||
if (uri && isExcluded(uri)) return;
|
||||
if (debounceHandle !== null) clearTimeout(debounceHandle);
|
||||
debounceHandle = setTimeout(() => {
|
||||
debounceHandle = null;
|
||||
rebuildIndex().catch((error) => output.appendLine(error.stack || String(error)));
|
||||
}, 100);
|
||||
}
|
||||
|
||||
const provider = {
|
||||
onDidChangeSemanticTokens: emitter.event,
|
||||
provideDocumentSemanticTokens(document, cancellationToken) {
|
||||
try {
|
||||
const result = classifyDocument(
|
||||
document.getText(),
|
||||
document.uri.fsPath,
|
||||
workspaceIndex,
|
||||
() => cancellationToken.isCancellationRequested
|
||||
);
|
||||
const builder = new vscode.SemanticTokensBuilder(legend);
|
||||
for (const span of result.spans) {
|
||||
if (cancellationToken.isCancellationRequested) break;
|
||||
builder.push(span.line, span.character, span.length, span.typeIndex, span.modifierMask);
|
||||
}
|
||||
for (const error of result.errors) {
|
||||
output.appendLine(formatError(document.uri.fsPath || document.uri.toString(), error));
|
||||
}
|
||||
return builder.build();
|
||||
} catch (error) {
|
||||
output.appendLine(`${document.uri}: ${error.stack || error.message || error}`);
|
||||
return new vscode.SemanticTokensBuilder(legend).build();
|
||||
}
|
||||
},
|
||||
};
|
||||
|
||||
const selector = [
|
||||
{ language: "c", scheme: "file" },
|
||||
{ language: "c", scheme: "untitled" },
|
||||
{ language: "cpp", scheme: "file" },
|
||||
{ language: "cpp", scheme: "untitled" },
|
||||
];
|
||||
const watcher = vscode.workspace.createFileSystemWatcher(SOURCE_GLOB);
|
||||
|
||||
context.subscriptions.push(
|
||||
output,
|
||||
emitter,
|
||||
watcher,
|
||||
watcher.onDidCreate(scheduleRebuild),
|
||||
watcher.onDidChange(scheduleRebuild),
|
||||
watcher.onDidDelete(scheduleRebuild),
|
||||
vscode.languages.registerDocumentSemanticTokensProvider(selector, provider, legend),
|
||||
{ dispose() { if (debounceHandle !== null) clearTimeout(debounceHandle); } }
|
||||
);
|
||||
|
||||
await rebuildIndex();
|
||||
}
|
||||
|
||||
function deactivate() {}
|
||||
|
||||
module.exports = { activate, deactivate };
|
||||
Vendored
+186
@@ -0,0 +1,186 @@
|
||||
"use strict";
|
||||
|
||||
function isIdentifierStart(code) {
|
||||
return code === 95 ||
|
||||
(code >= 65 && code <= 90) ||
|
||||
(code >= 97 && code <= 122);
|
||||
}
|
||||
|
||||
function isIdentifierContinue(code) {
|
||||
return isIdentifierStart(code) || (code >= 48 && code <= 57);
|
||||
}
|
||||
|
||||
function lex(source) {
|
||||
if (typeof source !== "string") throw new TypeError("source must be a string");
|
||||
|
||||
const tokens = [];
|
||||
const errors = [];
|
||||
let offset = 0;
|
||||
let line = 0;
|
||||
let character = 0;
|
||||
|
||||
function advance() {
|
||||
if (source[offset] === "\r" && source[offset + 1] === "\n") {
|
||||
offset += 2;
|
||||
line += 1;
|
||||
character = 0;
|
||||
return;
|
||||
}
|
||||
if (source[offset] === "\n") {
|
||||
offset += 1;
|
||||
line += 1;
|
||||
character = 0;
|
||||
return;
|
||||
}
|
||||
offset += 1;
|
||||
character += 1;
|
||||
}
|
||||
|
||||
function pushToken(kind, start, startLine, startCharacter) {
|
||||
tokens.push({
|
||||
kind,
|
||||
text: source.slice(start, offset),
|
||||
start,
|
||||
end: offset,
|
||||
line: startLine,
|
||||
character: startCharacter,
|
||||
});
|
||||
}
|
||||
|
||||
while (offset < source.length) {
|
||||
const ch = source[offset];
|
||||
|
||||
if (/\s/.test(ch)) {
|
||||
advance();
|
||||
continue;
|
||||
}
|
||||
|
||||
if (ch === "/" && source[offset + 1] === "/") {
|
||||
while (offset < source.length && source[offset] !== "\r" && source[offset] !== "\n") advance();
|
||||
continue;
|
||||
}
|
||||
|
||||
if (ch === "/" && source[offset + 1] === "*") {
|
||||
const start = offset;
|
||||
advance();
|
||||
advance();
|
||||
let closed = false;
|
||||
while (offset < source.length) {
|
||||
if (source[offset] === "*" && source[offset + 1] === "/") {
|
||||
advance();
|
||||
advance();
|
||||
closed = true;
|
||||
break;
|
||||
}
|
||||
advance();
|
||||
}
|
||||
if (!closed) errors.push({ kind: "unterminated-block-comment", offset: start });
|
||||
continue;
|
||||
}
|
||||
|
||||
if (ch === "\"" || ch === "'") {
|
||||
const quote = ch;
|
||||
const start = offset;
|
||||
advance();
|
||||
let closed = false;
|
||||
while (offset < source.length) {
|
||||
if (source[offset] === "\\") {
|
||||
advance();
|
||||
if (offset < source.length) advance();
|
||||
continue;
|
||||
}
|
||||
if (source[offset] === quote) {
|
||||
advance();
|
||||
closed = true;
|
||||
break;
|
||||
}
|
||||
if (source[offset] === "\n" || source[offset] === "\r") break;
|
||||
advance();
|
||||
}
|
||||
if (!closed) errors.push({ kind: "unterminated-literal", offset: start });
|
||||
continue;
|
||||
}
|
||||
|
||||
const code = source.charCodeAt(offset);
|
||||
if (isIdentifierStart(code)) {
|
||||
const start = offset;
|
||||
const startLine = line;
|
||||
const startCharacter = character;
|
||||
advance();
|
||||
while (offset < source.length && isIdentifierContinue(source.charCodeAt(offset))) advance();
|
||||
pushToken("identifier", start, startLine, startCharacter);
|
||||
continue;
|
||||
}
|
||||
|
||||
const start = offset;
|
||||
const startLine = line;
|
||||
const startCharacter = character;
|
||||
advance();
|
||||
pushToken("punctuation", start, startLine, startCharacter);
|
||||
}
|
||||
|
||||
return { tokens, errors };
|
||||
}
|
||||
|
||||
function buildCallContexts(tokens) {
|
||||
const contexts = Array.from({ length: tokens.length }, () => []);
|
||||
const calls = [];
|
||||
const errors = [];
|
||||
const stack = [];
|
||||
|
||||
for (let tokenIndex = 0; tokenIndex < tokens.length; tokenIndex += 1) {
|
||||
const token = tokens[tokenIndex];
|
||||
|
||||
if (token.text === ")") {
|
||||
if (stack.length === 0) {
|
||||
errors.push({ kind: "unmatched-close-paren", offset: token.start });
|
||||
} else {
|
||||
const frame = stack.pop();
|
||||
if (frame.callee !== null) calls.push({ ...frame, closeTokenIndex: tokenIndex });
|
||||
}
|
||||
}
|
||||
|
||||
contexts[tokenIndex] = stack
|
||||
.filter((frame) => frame.callee !== null)
|
||||
.map((frame) => ({
|
||||
callee: frame.callee,
|
||||
calleeTokenIndex: frame.calleeTokenIndex,
|
||||
openTokenIndex: frame.openTokenIndex,
|
||||
argIndex: frame.argIndex,
|
||||
}));
|
||||
|
||||
if (token.text === "(") {
|
||||
const previous = tokens[tokenIndex - 1];
|
||||
const hasCallee = previous && previous.kind === "identifier";
|
||||
stack.push({
|
||||
callee: hasCallee ? previous.text : null,
|
||||
calleeTokenIndex: hasCallee ? tokenIndex - 1 : -1,
|
||||
openTokenIndex: tokenIndex,
|
||||
argIndex: 0,
|
||||
});
|
||||
continue;
|
||||
}
|
||||
|
||||
if (token.text === "," && stack.length > 0) {
|
||||
const frame = stack[stack.length - 1];
|
||||
if (frame.callee !== null) frame.argIndex += 1;
|
||||
}
|
||||
}
|
||||
|
||||
for (const frame of stack) {
|
||||
errors.push({ kind: "unmatched-open-paren", offset: tokens[frame.openTokenIndex].start });
|
||||
}
|
||||
|
||||
return { contexts, calls, errors };
|
||||
}
|
||||
|
||||
function nearestCall(contexts, tokenIndex, callee) {
|
||||
const entries = contexts[tokenIndex] || [];
|
||||
for (let contextIndex = entries.length - 1; contextIndex >= 0; contextIndex -= 1) {
|
||||
const entry = entries[contextIndex];
|
||||
if (callee === undefined || entry.callee === callee) return entry;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
module.exports = { buildCallContexts, lex, nearestCall };
|
||||
+85
@@ -0,0 +1,85 @@
|
||||
{
|
||||
"name": "atomasm-psx",
|
||||
"displayName": "AtomAsm-PSX",
|
||||
"description": "Semantic highlighting for the PS1 Tape/Atom MIPS macro DSL",
|
||||
"publisher": "local",
|
||||
"version": "0.3.0",
|
||||
"engines": { "vscode": "^1.80.0" },
|
||||
"categories": ["Programming Languages"],
|
||||
"activationEvents": ["onLanguage:c", "onLanguage:cpp"],
|
||||
"main": "./extension.js",
|
||||
"files": [
|
||||
"classifier.js",
|
||||
"extension.js",
|
||||
"lexer.js",
|
||||
"source-index.js",
|
||||
"syntaxes/tape_atom.tmLanguage.json"
|
||||
],
|
||||
"scripts": {
|
||||
"test": "node --test test/*.test.js",
|
||||
"package": "npx --yes @vscode/vsce@3.6.1 package --allow-missing-repository --skip-license --out atomasm-psx-0.3.0.vsix"
|
||||
},
|
||||
"contributes": {
|
||||
"semanticTokenTypes": [
|
||||
{ "id": "tapeAtomKeyword", "superType": "keyword", "description": "Tape atom declaration keyword" },
|
||||
{ "id": "tapeAtomName", "superType": "function", "description": "Tape atom name" },
|
||||
{ "id": "tapeComponentKeyword", "superType": "keyword", "description": "Tape atom component declaration keyword" },
|
||||
{ "id": "tapeComponentName", "superType": "function", "description": "Tape atom component name" },
|
||||
{ "id": "tapeAnnotation", "superType": "macro", "description": "Tape atom annotation" },
|
||||
{ "id": "tapeBindType", "superType": "type", "description": "Tape bind structure type" },
|
||||
{ "id": "tapePhase", "superType": "label", "description": "Tape atom phase" },
|
||||
{ "id": "tapeLabel", "superType": "label", "description": "Tape atom branch label" },
|
||||
{ "id": "tapeCpuInstruction", "superType": "macro", "description": "MIPS CPU instruction emitter" },
|
||||
{ "id": "tapeControlFlow", "superType": "keyword", "description": "MIPS branch or jump instruction" },
|
||||
{ "id": "tapeGteInstruction", "superType": "macro", "description": "GTE instruction emitter" },
|
||||
{ "id": "tapeGpuInstruction", "superType": "macro", "description": "GPU command emitter" },
|
||||
{ "id": "tapeComponentInstruction", "superType": "macro", "description": "Tape atom component invocation" },
|
||||
{ "id": "tapeDelaySlot", "superType": "keyword", "description": "Load or branch delay slot annotation" },
|
||||
{ "id": "tapeGprRegister", "superType": "variable", "description": "MIPS GPR alias" },
|
||||
{ "id": "tapeCop2Register", "superType": "variable", "description": "COP2 data or control register alias" },
|
||||
{ "id": "tapeDuffleType", "superType": "type", "description": "Duffle type or type constructor" },
|
||||
{ "id": "tapeAttribute", "superType": "keyword", "description": "Duffle linkage or storage attribute" },
|
||||
{ "id": "keyword", "description": "Standard keyword (DSL built-in macros)" },
|
||||
{ "id": "macro", "description": "Standard macro (utility #define with no instruction domain)" }
|
||||
],
|
||||
"semanticTokenModifiers": [
|
||||
{ "id": "tapeRead", "description": "Register declared in atom_reads" },
|
||||
{ "id": "tapeWrite", "description": "Register declared in atom_writes" },
|
||||
{ "id": "tapeAuto", "description": "Auto-allocated register" }
|
||||
],
|
||||
"semanticTokenScopes": [
|
||||
{
|
||||
"language": "c",
|
||||
"scopes": {
|
||||
"tapeAtomKeyword": ["keyword.control.duffle.atom"],
|
||||
"tapeAtomName": ["entity.name.function.duffle.atom"],
|
||||
"tapeComponentKeyword": ["keyword.control.duffle.component"],
|
||||
"tapeComponentName": ["entity.name.function.duffle.component"],
|
||||
"tapeAnnotation": ["support.function.duffle.annotation"],
|
||||
"tapeBindType": ["entity.name.type.duffle.bind"],
|
||||
"tapePhase": ["entity.name.tag.duffle.phase"],
|
||||
"tapeLabel": ["entity.name.label.duffle.atom"],
|
||||
"tapeCpuInstruction": ["support.function.duffle.cpu"],
|
||||
"tapeControlFlow": ["keyword.control.duffle.branch"],
|
||||
"tapeGteInstruction": ["support.function.duffle.gte"],
|
||||
"tapeGpuInstruction": ["support.function.duffle.gpu"],
|
||||
"tapeComponentInstruction": ["support.function.duffle.component"],
|
||||
"tapeDelaySlot": ["keyword.operator.duffle.delayslot"],
|
||||
"tapeGprRegister": ["variable.other.constant.duffle.gpr"],
|
||||
"tapeCop2Register": ["variable.other.constant.duffle.cop2"],
|
||||
"tapeDuffleType": ["storage.type.duffle.type"],
|
||||
"tapeAttribute": ["storage.modifier.duffle.attr"],
|
||||
"keyword": ["keyword"],
|
||||
"macro": ["entity.name.function.preprocessor"]
|
||||
}
|
||||
}
|
||||
],
|
||||
"grammars": [
|
||||
{
|
||||
"scopeName": "tape_atom.injection",
|
||||
"path": "./syntaxes/tape_atom.tmLanguage.json",
|
||||
"injectTo": ["source.c", "source.cpp"]
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
+341
@@ -0,0 +1,341 @@
|
||||
"use strict";
|
||||
|
||||
const path = require("node:path");
|
||||
const { buildCallContexts, lex, nearestCall } = require("./lexer");
|
||||
|
||||
const BASE_TYPES = [
|
||||
"B1", "B2", "B4", "B8", "F4", "F8", "S1", "S2", "S4", "S8",
|
||||
"U1", "U2", "U4", "U8", "MipsAtom", "MipsCode", "Reg",
|
||||
];
|
||||
|
||||
const C_BUILTINS = new Set([
|
||||
"void", "type", "char", "short", "int", "long", "float", "double",
|
||||
"unsigned", "signed", "bool", "size_t", "uint8_t", "uint16_t", "uint32_t",
|
||||
"int8_t", "int16_t", "int32_t",
|
||||
]);
|
||||
|
||||
const BASE_ATTRIBUTES = [
|
||||
"FI_", "I_", "NI_", "Relative_", "Struct_", "Enum_", "Union_", "Array_",
|
||||
"Slice_", "TypeR_", "TypeV_", "align_", "internal", "local_persist", "global",
|
||||
"RO_", "LP_", "gknown", "expect_", "cexpr_",
|
||||
"asm", "asm_words", "asm_rpins", "asm_clobber",
|
||||
"O_", "S_", "C_", "T_", "tmpl", "glue", "r_", "v_", "tr_", "tv_",
|
||||
"rgcc", "r_use", "r_set", "r_mod", "r_imm", "r_mem",
|
||||
"u1_", "u2_", "u4_", "u8_", "s1_", "s2_", "s4_", "s8_",
|
||||
"u1_r", "u2_r", "u4_r", "u8_r", "u1_v", "u2_v", "u4_v", "u8_v",
|
||||
];
|
||||
|
||||
function createIndex() {
|
||||
return {
|
||||
atoms: new Set(),
|
||||
components: new Set(),
|
||||
componentAliases: new Set(),
|
||||
macros: new Map(),
|
||||
registers: new Map(),
|
||||
bindTypes: new Set(),
|
||||
types: new Set(BASE_TYPES),
|
||||
phases: new Set(),
|
||||
labels: new Set(),
|
||||
attributes: new Set(BASE_ATTRIBUTES),
|
||||
componentCallees: new Map(),
|
||||
};
|
||||
}
|
||||
|
||||
function cloneIndex(source) {
|
||||
const result = createIndex();
|
||||
for (const key of ["atoms", "components", "componentAliases", "bindTypes", "types", "phases", "labels", "attributes"]) {
|
||||
for (const value of source[key]) result[key].add(value);
|
||||
}
|
||||
for (const [name, domain] of source.macros) result.macros.set(name, domain);
|
||||
for (const [name, domain] of source.registers) result.registers.set(name, domain);
|
||||
for (const [name, callees] of source.componentCallees) result.componentCallees.set(name, callees.slice());
|
||||
return result;
|
||||
}
|
||||
|
||||
function mergeIndexes(...sources) {
|
||||
const result = createIndex();
|
||||
for (const source of sources) {
|
||||
if (!source) continue;
|
||||
for (const key of ["atoms", "components", "componentAliases", "bindTypes", "types", "phases", "labels", "attributes"]) {
|
||||
for (const value of source[key]) result[key].add(value);
|
||||
}
|
||||
for (const [name, domain] of source.macros) {
|
||||
const existing = result.macros.get(name);
|
||||
if (!existing || domainRank(domain) >= domainRank(existing)) result.macros.set(name, domain);
|
||||
}
|
||||
for (const [name, domain] of source.registers) result.registers.set(name, domain);
|
||||
for (const [name, callees] of source.componentCallees) {
|
||||
const existing = result.componentCallees.get(name) || [];
|
||||
result.componentCallees.set(name, existing.concat(callees));
|
||||
}
|
||||
}
|
||||
return resolveComponentDomains(result);
|
||||
}
|
||||
|
||||
function domainFromPath(filePath) {
|
||||
const base = path.basename(filePath.replaceAll("\\", "/")).toLowerCase();
|
||||
if (base === "mips.h") return "cpu";
|
||||
if (base === "gte.h") return "gte";
|
||||
if (base === "gp.h") return "gpu";
|
||||
return null;
|
||||
}
|
||||
|
||||
function prefixDomain(name) {
|
||||
if (/^(?:branch_|jump_|call_)/.test(name)) return "control";
|
||||
if (/^gte_(?!cr_)/.test(name) || name.startsWith("mac_gte_") || name.startsWith("ac_gte_")) return "gte";
|
||||
if (/^gp[01]_/.test(name) || name.startsWith("mac_gp_") || name.startsWith("ac_gp_")) return "gpu";
|
||||
return null;
|
||||
}
|
||||
|
||||
function collectBraceIdentifiers(tokens, openBraceIndex) {
|
||||
const names = [];
|
||||
let depth = 0;
|
||||
for (let tokenIndex = openBraceIndex; tokenIndex < tokens.length; tokenIndex += 1) {
|
||||
if (tokens[tokenIndex].text === "{") depth += 1;
|
||||
if (tokens[tokenIndex].text === "}") {
|
||||
depth -= 1;
|
||||
if (depth === 0) break;
|
||||
}
|
||||
if (tokens[tokenIndex].kind === "identifier") names.push(tokens[tokenIndex].text);
|
||||
}
|
||||
return names;
|
||||
}
|
||||
|
||||
function resolveComponentDomains(index) {
|
||||
const hardwareRank = { cpu: 1, gpu: 2, gte: 3, control: 4 };
|
||||
let changed = true;
|
||||
while (changed) {
|
||||
changed = false;
|
||||
for (const [alias, callees] of index.componentCallees) {
|
||||
let best = index.macros.get(alias) || "component";
|
||||
let bestRank = hardwareRank[best] || 0;
|
||||
for (const callee of callees) {
|
||||
const domain = prefixDomain(callee) || index.macros.get(callee);
|
||||
const rank = hardwareRank[domain] || 0;
|
||||
if (rank > bestRank) {
|
||||
best = domain;
|
||||
bestRank = rank;
|
||||
}
|
||||
}
|
||||
if (bestRank > 0 && index.macros.get(alias) !== best) {
|
||||
index.macros.set(alias, best);
|
||||
changed = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
return index;
|
||||
}
|
||||
|
||||
function domainRank(domain) {
|
||||
if (domain === "control") return 4;
|
||||
if (domain === "cpu" || domain === "gte" || domain === "gpu") return 3;
|
||||
if (domain === "component") return 2;
|
||||
return 1;
|
||||
}
|
||||
|
||||
function registerKind(name) {
|
||||
if (/^R_[A-Za-z0-9_]+$/.test(name)) return "gpr";
|
||||
if (/^(?:C2_|gte_cr_)[A-Za-z0-9_]+$/.test(name)) return "cop2";
|
||||
return null;
|
||||
}
|
||||
|
||||
function componentAlias(name) {
|
||||
return name.startsWith("ac_") ? `mac_${name.slice(3)}` : null;
|
||||
}
|
||||
|
||||
function findFunctionNameBefore(tokens, calleeTokenIndex) {
|
||||
let closeIndex = calleeTokenIndex - 1;
|
||||
while (closeIndex >= 0 && tokens[closeIndex].kind === "identifier" && tokens[closeIndex].text === "atom_dbg_skip") {
|
||||
closeIndex -= 1;
|
||||
}
|
||||
if (!tokens[closeIndex] || tokens[closeIndex].text !== ")") return null;
|
||||
|
||||
let depth = 1;
|
||||
for (let tokenIndex = closeIndex - 1; tokenIndex >= 0; tokenIndex -= 1) {
|
||||
if (tokens[tokenIndex].text === ")") depth += 1;
|
||||
if (tokens[tokenIndex].text === "(") depth -= 1;
|
||||
if (depth !== 0) continue;
|
||||
const name = tokens[tokenIndex - 1];
|
||||
return name && name.kind === "identifier" ? name : null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function scanSource(source, filePath) {
|
||||
const lexical = lex(source);
|
||||
const balanced = buildCallContexts(lexical.tokens);
|
||||
const tokens = lexical.tokens;
|
||||
const contexts = balanced.contexts;
|
||||
const index = createIndex();
|
||||
const declarations = new Map();
|
||||
const domain = domainFromPath(filePath);
|
||||
|
||||
function mark(token, role, modifiers = ["declaration"]) {
|
||||
declarations.set(token.start, { role, modifiers });
|
||||
}
|
||||
|
||||
function addComponent(token) {
|
||||
index.components.add(token.text);
|
||||
mark(token, "componentName");
|
||||
const alias = componentAlias(token.text);
|
||||
if (alias) {
|
||||
index.componentAliases.add(alias);
|
||||
index.macros.set(alias, prefixDomain(alias) || prefixDomain(token.text) || "component");
|
||||
}
|
||||
}
|
||||
|
||||
function bindComponentCallees(alias, callees) {
|
||||
if (!alias) return;
|
||||
index.componentAliases.add(alias);
|
||||
index.componentCallees.set(alias, callees);
|
||||
if (!index.macros.has(alias)) index.macros.set(alias, "component");
|
||||
}
|
||||
|
||||
for (let tokenIndex = 0; tokenIndex < tokens.length; tokenIndex += 1) {
|
||||
const token = tokens[tokenIndex];
|
||||
if (token.kind !== "identifier") continue;
|
||||
|
||||
const kind = registerKind(token.text);
|
||||
if (kind) {
|
||||
index.registers.set(token.text, kind);
|
||||
if (tokens[tokenIndex + 1] && tokens[tokenIndex + 1].text === "=") {
|
||||
mark(token, kind === "gpr" ? "gprRegister" : "cop2Register");
|
||||
}
|
||||
}
|
||||
|
||||
const context = nearestCall(contexts, tokenIndex);
|
||||
if (context && context.argIndex === 0) {
|
||||
if (context.callee === "MipsAtom_") {
|
||||
index.atoms.add(token.text);
|
||||
mark(token, "atomName");
|
||||
}
|
||||
if (context.callee === "MipsAtomComp_") addComponent(token);
|
||||
if (context.callee === "atom_bind") index.bindTypes.add(token.text);
|
||||
if (context.callee === "atom_phase" || context.callee === "phase_auto_reg") index.phases.add(token.text);
|
||||
if (context.callee === "atom_label" || context.callee === "atom_offset") index.labels.add(token.text);
|
||||
}
|
||||
|
||||
const isWrappedType = context && (
|
||||
((context.callee === "Struct_" || context.callee === "Union_") && context.argIndex === 0) ||
|
||||
(context.callee === "Enum_" && context.argIndex === 1)
|
||||
);
|
||||
if (isWrappedType) {
|
||||
index.types.add(token.text);
|
||||
mark(token, token.text.startsWith("Binds_") ? "bindType" : "duffleType");
|
||||
if (token.text.startsWith("Binds_")) index.bindTypes.add(token.text);
|
||||
}
|
||||
|
||||
if (context && context.callee === "atom_offset" && context.argIndex === 1) index.labels.add(token.text);
|
||||
|
||||
if (context && context.callee === "atom_auto_reg") {
|
||||
if (context.argIndex === 0) index.atoms.add(token.text);
|
||||
if (context.argIndex === 1) {
|
||||
index.registers.set(token.text, "gpr");
|
||||
mark(token, "gprRegister", ["declaration", "tapeAuto"]);
|
||||
}
|
||||
}
|
||||
|
||||
if (context && context.callee === "phase_auto_reg" && context.argIndex === 1) {
|
||||
index.registers.set(token.text, "gpr");
|
||||
mark(token, "gprRegister", ["declaration", "tapeAuto"]);
|
||||
}
|
||||
|
||||
if (token.text === "define" && tokens[tokenIndex - 1] && tokens[tokenIndex - 1].text === "#") {
|
||||
const name = tokens[tokenIndex + 1];
|
||||
if (name && name.kind === "identifier" && name.line === token.line) {
|
||||
if (/^(?:RegUse_|Struct_|Enum_|Union_|TypeR_|TypeV_|Relative_|Binds_)/.test(name.text)) {
|
||||
index.types.add(name.text);
|
||||
} else if (/^(?:ac_|mac_)/.test(name.text)) {
|
||||
const alias = name.text.startsWith("ac_") ? componentAlias(name.text) : name.text;
|
||||
const rest = [];
|
||||
for (let restIndex = tokenIndex + 2; restIndex < tokens.length && tokens[restIndex].line === name.line; restIndex += 1) {
|
||||
if (tokens[restIndex].kind === "identifier") rest.push(tokens[restIndex].text);
|
||||
}
|
||||
if (alias) {
|
||||
index.componentAliases.add(alias);
|
||||
index.macros.set(alias, prefixDomain(alias) || "component");
|
||||
if (rest.length) index.componentCallees.set(alias, rest);
|
||||
}
|
||||
} else {
|
||||
index.macros.set(name.text, domain || "utility");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (token.text === "typedef") {
|
||||
let endIndex = tokenIndex + 1;
|
||||
let hasBrace = false;
|
||||
let lastIdentifier = null;
|
||||
while (endIndex < tokens.length && tokens[endIndex].text !== ";") {
|
||||
if (tokens[endIndex].text === "{") hasBrace = true;
|
||||
if (tokens[endIndex].kind === "identifier" && !C_BUILTINS.has(tokens[endIndex].text)) lastIdentifier = tokens[endIndex];
|
||||
endIndex += 1;
|
||||
}
|
||||
if (!hasBrace && lastIdentifier && !C_BUILTINS.has(lastIdentifier.text)) {
|
||||
index.types.add(lastIdentifier.text);
|
||||
mark(lastIdentifier, "duffleType");
|
||||
}
|
||||
}
|
||||
|
||||
if (token.text === "MipsAtom_Proc_") {
|
||||
const functionName = findFunctionNameBefore(tokens, tokenIndex);
|
||||
if (functionName) {
|
||||
const atomName = functionName.text.endsWith("_proc")
|
||||
? functionName.text.slice(0, -5)
|
||||
: functionName.text;
|
||||
index.atoms.add(atomName);
|
||||
index.atoms.add(functionName.text);
|
||||
mark(functionName, "atomName");
|
||||
}
|
||||
}
|
||||
|
||||
if (token.text === "MipsAtomComp_Proc_") {
|
||||
const functionName = findFunctionNameBefore(tokens, tokenIndex);
|
||||
if (functionName) addComponent(functionName);
|
||||
}
|
||||
}
|
||||
|
||||
for (const call of balanced.calls) {
|
||||
if (call.callee === "MipsAtomComp_") {
|
||||
const name = tokens[call.openTokenIndex + 1];
|
||||
const brace = tokens[call.closeTokenIndex + 1];
|
||||
if (name && name.kind === "identifier" && brace && brace.text === "{") {
|
||||
bindComponentCallees(componentAlias(name.text), collectBraceIdentifiers(tokens, call.closeTokenIndex + 1));
|
||||
}
|
||||
}
|
||||
if (call.callee === "MipsAtomComp_Proc_") {
|
||||
const functionName = findFunctionNameBefore(tokens, call.calleeTokenIndex);
|
||||
let braceIndex = -1;
|
||||
for (let tokenIndex = call.openTokenIndex + 1; tokenIndex < call.closeTokenIndex; tokenIndex += 1) {
|
||||
if (tokens[tokenIndex].text === "{") {
|
||||
braceIndex = tokenIndex;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (functionName && braceIndex >= 0) {
|
||||
bindComponentCallees(componentAlias(functionName.text), collectBraceIdentifiers(tokens, braceIndex));
|
||||
}
|
||||
}
|
||||
if (!domain) continue;
|
||||
const name = tokens[call.calleeTokenIndex];
|
||||
const after = tokens[call.closeTokenIndex + 1];
|
||||
if (!name || !after || after.text !== "{") continue;
|
||||
if (/^(?:gp0_|gp1_|gte_|mac_)/.test(name.text)) index.macros.set(name.text, domain);
|
||||
}
|
||||
|
||||
return {
|
||||
index: resolveComponentDomains(cloneIndex(index)),
|
||||
declarations,
|
||||
tokens,
|
||||
contexts,
|
||||
errors: [...lexical.errors, ...balanced.errors],
|
||||
};
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
createIndex,
|
||||
domainFromPath,
|
||||
mergeIndexes,
|
||||
resolveComponentDomains,
|
||||
scanSource,
|
||||
};
|
||||
@@ -0,0 +1,71 @@
|
||||
{
|
||||
"scopeName": "tape_atom.injection",
|
||||
"injectionSelector": "L:source.c -comment -string, L:source.cpp -comment -string",
|
||||
"patterns": [
|
||||
{ "include": "#atom-declarations" },
|
||||
{ "include": "#component-declarations" },
|
||||
{ "include": "#annotation-arguments" },
|
||||
{ "include": "#annotations" },
|
||||
{ "include": "#delay-slots" },
|
||||
{ "include": "#types" },
|
||||
{ "include": "#attributes" }
|
||||
],
|
||||
"repository": {
|
||||
"atom-declarations": {
|
||||
"patterns": [
|
||||
{
|
||||
"match": "\\b(MipsAtom_)\\s*\\(\\s*([A-Za-z_][A-Za-z0-9_]*)",
|
||||
"captures": {
|
||||
"1": { "name": "keyword.control.duffle.atom" },
|
||||
"2": { "name": "entity.name.function.duffle.atom" }
|
||||
}
|
||||
},
|
||||
{ "match": "\\bMipsAtom_Proc_\\b", "name": "keyword.control.duffle.atom" },
|
||||
{ "match": "\\b[A-Za-z_][A-Za-z0-9_]*_proc\\b", "name": "entity.name.function.duffle.atom" }
|
||||
]
|
||||
},
|
||||
"component-declarations": {
|
||||
"patterns": [
|
||||
{
|
||||
"match": "\\b(MipsAtomComp_)\\s*\\(\\s*(ac_[A-Za-z0-9_]*)",
|
||||
"captures": {
|
||||
"1": { "name": "keyword" },
|
||||
"2": { "name": "entity.name.function.duffle.component" }
|
||||
}
|
||||
},
|
||||
{ "match": "\\bMipsAtomComp_Proc_\\b", "name": "keyword" }
|
||||
]
|
||||
},
|
||||
"annotation-arguments": {
|
||||
"patterns": [
|
||||
{
|
||||
"match": "\\b(atom_offset)\\s*\\(\\s*([A-Za-z_][A-Za-z0-9_]*)\\s*,\\s*([A-Za-z_][A-Za-z0-9_]*)",
|
||||
"captures": {
|
||||
"1": { "name": "support.function.duffle.annotation" },
|
||||
"2": { "name": "entity.name.label.duffle.atom" },
|
||||
"3": { "name": "entity.name.label.duffle.atom" }
|
||||
}
|
||||
},
|
||||
{ "match": "(?<=\\batom_bind\\()\\s*Binds_[A-Za-z0-9_]+", "name": "entity.name.type.duffle.bind" },
|
||||
{ "match": "(?<=\\batom_phase\\()\\s*[A-Za-z_][A-Za-z0-9_]*", "name": "entity.name.tag.duffle.phase" },
|
||||
{ "match": "(?<=\\batom_label\\()\\s*[A-Za-z_][A-Za-z0-9_]*", "name": "entity.name.label.duffle.atom" }
|
||||
]
|
||||
},
|
||||
"annotations": {
|
||||
"match": "\\b(atom_info|atom_bind|atom_reads|atom_writes|atom_label|atom_offset|atom_reg|atom_type|atom_ctx|atom_phase|atom_auto_reg|phase_auto_reg|atom_dbg_skip)\\b",
|
||||
"name": "support.function.duffle.annotation"
|
||||
},
|
||||
"delay-slots": {
|
||||
"match": "\\b(LdSlot_|BdSlot_|DmaSlot_|GteDelay_)\\b",
|
||||
"name": "keyword.operator.duffle.delayslot"
|
||||
},
|
||||
"types": {
|
||||
"match": "\\b(?:Binds_[A-Za-z0-9_]+|RegUse_[A-Za-z0-9_]+)\\b",
|
||||
"name": "storage.type.duffle.type"
|
||||
},
|
||||
"attributes": {
|
||||
"match": "\\b(?:FI_|I_|NI_|Relative_|Struct_|Enum_|Union_|Array_|Slice_|TypeR_|TypeV_|align_|internal|local_persist|global|RO_|LP_|gknown|expect_|cexpr_|asm|asm_words|asm_rpins|asm_clobber|O_|S_|C_|T_|tmpl|glue|r_|v_|tr_|tv_|rgcc|r_use|r_set|r_mod|r_imm|r_mem|u[1248]_|u[1248]_r|u[1248]_v|s[1248]_)\\b",
|
||||
"name": "keyword"
|
||||
}
|
||||
}
|
||||
}
|
||||
Binary file not shown.
+128
@@ -0,0 +1,128 @@
|
||||
"use strict";
|
||||
|
||||
const assert = require("node:assert/strict");
|
||||
const test = require("node:test");
|
||||
|
||||
const { classifyDocument } = require("../classifier");
|
||||
const { createIndex } = require("../source-index");
|
||||
|
||||
function byText(result, text) {
|
||||
return result.spans.filter((span) => span.text === text);
|
||||
}
|
||||
|
||||
test("classifyDocument distinguishes declaration, annotation, phase, bind, and label roles", () => {
|
||||
const source = [
|
||||
"typedef Struct_(Binds_CubeTri) { U4 PrimCursor; };",
|
||||
"MipsAtom_(cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4),",
|
||||
"\tatom_reads(R_PrimCursor), atom_writes(R_FaceCursor)) {",
|
||||
"\tbranch_le_zero(R_T0, atom_offset(cull, exit)),",
|
||||
"\tatom_label(exit)",
|
||||
"};",
|
||||
].join("\n");
|
||||
|
||||
const result = classifyDocument(source, "C:/x/code/hello_camera/hello_camera.atom.c", createIndex());
|
||||
|
||||
assert.equal(byText(result, "MipsAtom_")[0].type, "tapeAtomKeyword");
|
||||
assert.deepEqual(byText(result, "cube_g4_face")[0].modifiers, ["declaration"]);
|
||||
assert.equal(byText(result, "atom_bind")[0].type, "tapeAnnotation");
|
||||
assert.equal(byText(result, "Binds_CubeTri").at(-1).type, "tapeBindType");
|
||||
assert.equal(byText(result, "cube_g4")[0].type, "tapePhase");
|
||||
assert.deepEqual(byText(result, "cube_g4")[0].modifiers, ["declaration"]);
|
||||
assert.equal(byText(result, "cull")[0].type, "tapeLabel");
|
||||
assert.equal(byText(result, "exit").every((span) => span.type === "tapeLabel"), true);
|
||||
});
|
||||
|
||||
test("classifyDocument applies read and write modifiers to GPRs", () => {
|
||||
const source = "atom_info(atom_reads(R_PrimCursor), atom_writes(R_FaceCursor))";
|
||||
const result = classifyDocument(source, "C:/x/code/test.atom.c", createIndex());
|
||||
|
||||
assert.deepEqual(byText(result, "R_PrimCursor")[0].modifiers, ["tapeRead"]);
|
||||
assert.deepEqual(byText(result, "R_FaceCursor")[0].modifiers, ["tapeWrite"]);
|
||||
});
|
||||
|
||||
test("classifyDocument separates CPU, GTE, GPU, and component domains", () => {
|
||||
const workspace = createIndex();
|
||||
workspace.macros.set("load_word", "cpu");
|
||||
workspace.macros.set("gte_cmdw_rtpt", "gte");
|
||||
workspace.macros.set("gp1_word_DisplayOn", "gpu");
|
||||
workspace.macros.set("mac_yield", "control");
|
||||
workspace.componentAliases.add("mac_yield");
|
||||
|
||||
const source = "load_word(R_T0, R_T1, 0), gte_cmdw_rtpt, gp1_word_DisplayOn(), mac_yield(), C2_MAC0, gte_cr_OFX_Code";
|
||||
const result = classifyDocument(source, "C:/x/code/test.c", workspace);
|
||||
|
||||
assert.equal(byText(result, "load_word")[0].type, "tapeCpuInstruction");
|
||||
assert.equal(byText(result, "gte_cmdw_rtpt")[0].type, "tapeGteInstruction");
|
||||
assert.equal(byText(result, "gp1_word_DisplayOn")[0].type, "tapeGpuInstruction");
|
||||
assert.equal(byText(result, "mac_yield")[0].type, "tapeControlFlow");
|
||||
assert.equal(byText(result, "C2_MAC0")[0].type, "tapeCop2Register");
|
||||
assert.equal(byText(result, "gte_cr_OFX_Code")[0].type, "tapeCop2Register");
|
||||
});
|
||||
|
||||
test("component invocations keep the domain resolved from their emitted instructions", () => {
|
||||
const workspace = createIndex();
|
||||
workspace.macros.set("mac_load_word_imm", "cpu");
|
||||
workspace.macros.set("mac_gcmd_push", "gpu");
|
||||
workspace.macros.set("mac_gte_store_f3", "gte");
|
||||
workspace.macros.set("mac_load_v3s4", "cpu");
|
||||
|
||||
const source = "mac_load_word_imm(dst, imm), mac_gcmd_push(cmd), mac_gte_store_f3(cursor), mac_load_v3s4()";
|
||||
const result = classifyDocument(source, "C:/x/code/hello_camera/hello_camera.atom.c", workspace);
|
||||
|
||||
assert.equal(byText(result, "mac_load_word_imm")[0].type, "tapeCpuInstruction");
|
||||
assert.equal(byText(result, "mac_gcmd_push")[0].type, "tapeGpuInstruction");
|
||||
assert.equal(byText(result, "mac_gte_store_f3")[0].type, "tapeGteInstruction");
|
||||
assert.equal(byText(result, "mac_load_v3s4")[0].type, "tapeCpuInstruction");
|
||||
});
|
||||
|
||||
test("utility macros without a hardware domain use the standard macro token", () => {
|
||||
const workspace = createIndex();
|
||||
workspace.macros.set("load_word", "cpu");
|
||||
workspace.macros.set("assert", "utility");
|
||||
workspace.macros.set("stringify", "utility");
|
||||
workspace.macros.set("u4_hi", "utility");
|
||||
|
||||
const source = "load_word(R_T0, R_T1, 0), assert(ok), stringify(name), u4_hi(imm)";
|
||||
const result = classifyDocument(source, "C:/x/code/hello_camera/hello_camera.c", workspace);
|
||||
|
||||
assert.equal(byText(result, "load_word")[0].type, "tapeCpuInstruction");
|
||||
assert.equal(byText(result, "assert")[0].type, "macro");
|
||||
assert.equal(byText(result, "stringify")[0].type, "macro");
|
||||
assert.equal(byText(result, "u4_hi")[0].type, "macro");
|
||||
});
|
||||
|
||||
test("document-local declarations override an empty workspace index", () => {
|
||||
const source = [
|
||||
"MipsAtomComp_(ac_new_component) { nop };",
|
||||
"MipsAtomComp_Proc_(ab, { nop })",
|
||||
"mac_new_component(),",
|
||||
].join("\n");
|
||||
const result = classifyDocument(source, "C:/x/code/duffle/math.atom.c", createIndex());
|
||||
|
||||
assert.equal(byText(result, "MipsAtomComp_")[0].type, "keyword");
|
||||
assert.equal(byText(result, "MipsAtomComp_Proc_")[0].type, "keyword");
|
||||
assert.equal(byText(result, "ac_new_component")[0].type, "tapeComponentName");
|
||||
assert.equal(byText(result, "mac_new_component")[0].type, "tapeComponentInstruction");
|
||||
});
|
||||
|
||||
test("delay slot markers share the tapeDelaySlot token", () => {
|
||||
const source = "LdSlot_ nop, BdSlot_ nop, DmaSlot_ nop2, GteDelay_ nop";
|
||||
const result = classifyDocument(source, "C:/x/code/duffle/gte.atom.c", createIndex());
|
||||
|
||||
assert.equal(byText(result, "LdSlot_")[0].type, "tapeDelaySlot");
|
||||
assert.equal(byText(result, "BdSlot_")[0].type, "tapeDelaySlot");
|
||||
assert.equal(byText(result, "DmaSlot_")[0].type, "tapeDelaySlot");
|
||||
assert.equal(byText(result, "GteDelay_")[0].type, "tapeDelaySlot");
|
||||
});
|
||||
|
||||
test("classifier returns ordered non-overlapping spans and partial malformed output", () => {
|
||||
const source = "atom_reads(R_A /* broken";
|
||||
const result = classifyDocument(source, "C:/x/code/test.atom.c", createIndex());
|
||||
|
||||
assert.equal(result.errors.some((error) => error.kind === "unterminated-block-comment"), true);
|
||||
for (let spanIndex = 1; spanIndex < result.spans.length; spanIndex += 1) {
|
||||
const previous = result.spans[spanIndex - 1];
|
||||
const current = result.spans[spanIndex];
|
||||
assert.equal(previous.start + previous.length <= current.start, true);
|
||||
}
|
||||
});
|
||||
+88
@@ -0,0 +1,88 @@
|
||||
"use strict";
|
||||
|
||||
const assert = require("node:assert/strict");
|
||||
const fs = require("node:fs");
|
||||
const path = require("node:path");
|
||||
const test = require("node:test");
|
||||
|
||||
const { TOKEN_MODIFIERS, TOKEN_TYPES } = require("../classifier");
|
||||
|
||||
const ROOT = path.resolve(__dirname, "..");
|
||||
|
||||
function readJson(filePath) {
|
||||
const raw = fs.readFileSync(filePath, "utf8");
|
||||
const stripped = raw.replace(/\/\/.*$/gm, "").replace(/,\s*([}\]])/g, "$1");
|
||||
return JSON.parse(stripped);
|
||||
}
|
||||
|
||||
function collectScopeNames(value, output = new Set()) {
|
||||
if (Array.isArray(value)) {
|
||||
for (const entry of value) collectScopeNames(entry, output);
|
||||
return output;
|
||||
}
|
||||
if (!value || typeof value !== "object") return output;
|
||||
if (typeof value.name === "string") output.add(value.name);
|
||||
for (const child of Object.values(value)) collectScopeNames(child, output);
|
||||
return output;
|
||||
}
|
||||
|
||||
test("package semantic legend matches classifier exports", () => {
|
||||
const packageJson = readJson(path.join(ROOT, "package.json"));
|
||||
const contributedTypes = packageJson.contributes.semanticTokenTypes.map((entry) => entry.id);
|
||||
const contributedModifiers = packageJson.contributes.semanticTokenModifiers.map((entry) => entry.id);
|
||||
|
||||
assert.equal(packageJson.version, "0.3.0");
|
||||
assert.deepEqual(contributedTypes, TOKEN_TYPES);
|
||||
assert.deepEqual(contributedModifiers, TOKEN_MODIFIERS.filter((name) => name !== "declaration"));
|
||||
});
|
||||
|
||||
test("package includes runtime files only and acknowledges local-only metadata", () => {
|
||||
const packageJson = readJson(path.join(ROOT, "package.json"));
|
||||
|
||||
assert.deepEqual(packageJson.files, [
|
||||
"classifier.js",
|
||||
"extension.js",
|
||||
"lexer.js",
|
||||
"source-index.js",
|
||||
"syntaxes/tape_atom.tmLanguage.json",
|
||||
]);
|
||||
assert.equal(packageJson.scripts.package.includes("--allow-missing-repository"), true);
|
||||
assert.equal(packageJson.scripts.package.includes("--skip-license"), true);
|
||||
});
|
||||
|
||||
test("every semantic token has a scope mapping; DSL-specific tokens also have grammar scopes", () => {
|
||||
const packageJson = readJson(path.join(ROOT, "package.json"));
|
||||
const grammar = readJson(path.join(ROOT, "syntaxes", "tape_atom.tmLanguage.json"));
|
||||
const mappings = packageJson.contributes.semanticTokenScopes[0].scopes;
|
||||
const grammarScopes = collectScopeNames(grammar);
|
||||
|
||||
const grammarRequired = new Set([
|
||||
"tapeAtomKeyword", "tapeAtomName", "tapeComponentName",
|
||||
"tapeAnnotation", "tapeBindType", "tapePhase", "tapeLabel",
|
||||
"tapeDelaySlot", "tapeDuffleType", "keyword",
|
||||
]);
|
||||
|
||||
for (const tokenType of TOKEN_TYPES) {
|
||||
assert.equal(Array.isArray(mappings[tokenType]), true, `missing scope mapping: ${tokenType}`);
|
||||
if (grammarRequired.has(tokenType)) {
|
||||
assert.equal(mappings[tokenType].some((scope) => grammarScopes.has(scope)), true, `grammar does not emit: ${tokenType}`);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
test("TextMate offset labels stay scoped to atom_offset calls", () => {
|
||||
const grammar = readJson(path.join(ROOT, "syntaxes", "tape_atom.tmLanguage.json"));
|
||||
const serialized = JSON.stringify(grammar);
|
||||
const offsetRule = grammar.repository["annotation-arguments"].patterns
|
||||
.find((rule) => rule.match.includes("atom_offset"));
|
||||
|
||||
assert.equal(serialized.includes("(?<=,)"), false);
|
||||
assert.equal(offsetRule.captures[1].name, "support.function.duffle.annotation");
|
||||
assert.equal(offsetRule.captures[2].name, "entity.name.label.duffle.atom");
|
||||
assert.equal(offsetRule.captures[3].name, "entity.name.label.duffle.atom");
|
||||
});
|
||||
|
||||
test("workspace enables semantic highlighting", () => {
|
||||
const settings = readJson(path.resolve(ROOT, "..", "settings.json"));
|
||||
assert.equal(settings["editor.semanticHighlighting.enabled"], true);
|
||||
});
|
||||
+77
@@ -0,0 +1,77 @@
|
||||
"use strict";
|
||||
|
||||
const assert = require("node:assert/strict");
|
||||
const test = require("node:test");
|
||||
|
||||
const { buildCallContexts, lex, nearestCall } = require("../lexer");
|
||||
|
||||
test("lex skips comments, strings, and character literals", () => {
|
||||
const source = [
|
||||
"MipsAtom_(visible)",
|
||||
"// MipsAtom_(line_comment)",
|
||||
"const char *s = \"atom_reads(R_Hidden)\";",
|
||||
"char c = '\\''; /* gte_cmdw_hidden */",
|
||||
"atom_reads(R_Visible)",
|
||||
].join("\n");
|
||||
|
||||
const result = lex(source);
|
||||
const identifiers = result.tokens
|
||||
.filter((token) => token.kind === "identifier")
|
||||
.map((token) => token.text);
|
||||
|
||||
assert.deepEqual(result.errors, []);
|
||||
assert.equal(identifiers.includes("visible"), true);
|
||||
assert.equal(identifiers.includes("R_Visible"), true);
|
||||
assert.equal(identifiers.includes("line_comment"), false);
|
||||
assert.equal(identifiers.includes("R_Hidden"), false);
|
||||
assert.equal(identifiers.includes("gte_cmdw_hidden"), false);
|
||||
});
|
||||
|
||||
test("lex reports unterminated block comments without returning comment tokens", () => {
|
||||
const result = lex("R_Visible /* atom_reads(R_Hidden)");
|
||||
|
||||
assert.equal(result.tokens.some((token) => token.text === "R_Visible"), true);
|
||||
assert.equal(result.tokens.some((token) => token.text === "R_Hidden"), false);
|
||||
assert.deepEqual(result.errors.map((error) => error.kind), ["unterminated-block-comment"]);
|
||||
});
|
||||
|
||||
test("line comments stop at CRLF boundaries", () => {
|
||||
const result = lex("// atom_reads(R_Hidden)\r\natom_reads(R_Visible)\r\n");
|
||||
const identifiers = result.tokens
|
||||
.filter((token) => token.kind === "identifier")
|
||||
.map((token) => token.text);
|
||||
|
||||
assert.equal(identifiers.includes("R_Hidden"), false);
|
||||
assert.equal(identifiers.includes("R_Visible"), true);
|
||||
});
|
||||
|
||||
test("balanced contexts retain multiline nesting and argument indexes", () => {
|
||||
const source = [
|
||||
"atom_info(",
|
||||
"\tatom_phase(cube_g4),",
|
||||
"\tatom_reads(R_A, nested(R_B, R_C)),",
|
||||
"\tatom_writes(R_D)",
|
||||
")",
|
||||
].join("\n");
|
||||
const lexical = lex(source);
|
||||
const balanced = buildCallContexts(lexical.tokens);
|
||||
|
||||
const byText = new Map();
|
||||
lexical.tokens.forEach((token, index) => {
|
||||
if (token.kind === "identifier") byText.set(token.text, index);
|
||||
});
|
||||
|
||||
assert.equal(nearestCall(balanced.contexts, byText.get("cube_g4")).callee, "atom_phase");
|
||||
assert.equal(nearestCall(balanced.contexts, byText.get("R_A")).callee, "atom_reads");
|
||||
assert.equal(nearestCall(balanced.contexts, byText.get("R_A")).argIndex, 0);
|
||||
assert.equal(nearestCall(balanced.contexts, byText.get("R_C")).callee, "nested");
|
||||
assert.equal(nearestCall(balanced.contexts, byText.get("R_D")).callee, "atom_writes");
|
||||
assert.deepEqual(balanced.errors, []);
|
||||
});
|
||||
|
||||
test("balanced contexts report unmatched parentheses", () => {
|
||||
const lexical = lex("atom_reads(R_A");
|
||||
const balanced = buildCallContexts(lexical.tokens);
|
||||
|
||||
assert.deepEqual(balanced.errors.map((error) => error.kind), ["unmatched-open-paren"]);
|
||||
});
|
||||
+134
@@ -0,0 +1,134 @@
|
||||
"use strict";
|
||||
|
||||
const assert = require("node:assert/strict");
|
||||
const test = require("node:test");
|
||||
|
||||
const {
|
||||
createIndex,
|
||||
domainFromPath,
|
||||
mergeIndexes,
|
||||
scanSource,
|
||||
} = require("../source-index");
|
||||
|
||||
test("scanSource discovers current atom and component forms", () => {
|
||||
const source = [
|
||||
"MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4), atom_reads(R_PrimCursor)) { mac_yield() };",
|
||||
"MipsAtomComp_(ac_load_pair) { load_word(R_T0, R_T1, 0) };",
|
||||
"internal MipsAtom* normalize_proc(AtomArena_R aa) MipsAtom_Proc_(aa, { mac_yield() })",
|
||||
"FI_ void ac_store_pair(MipsAtomBuilder_R ab) atom_dbg_skip MipsAtomComp_Proc_(ab, { store_word(R_T0, R_T1, 0) })",
|
||||
].join("\n");
|
||||
|
||||
const result = scanSource(source, "C:/projects/Pikuma/ps1/code/duffle/mips.atom.c");
|
||||
|
||||
assert.equal(result.index.atoms.has("cube_g4_face"), true);
|
||||
assert.equal(result.index.atoms.has("normalize"), true);
|
||||
assert.equal(result.index.components.has("ac_load_pair"), true);
|
||||
assert.equal(result.index.components.has("ac_store_pair"), true);
|
||||
assert.equal(result.index.componentAliases.has("mac_load_pair"), true);
|
||||
assert.equal(result.index.componentAliases.has("mac_store_pair"), true);
|
||||
assert.equal(result.index.macros.get("mac_store_pair"), "component");
|
||||
assert.equal(result.index.componentCallees.get("mac_store_pair").includes("store_word"), true);
|
||||
assert.equal(result.index.phases.has("cube_g4"), true);
|
||||
assert.equal(result.index.registers.get("R_PrimCursor"), "gpr");
|
||||
assert.deepEqual(result.errors, []);
|
||||
});
|
||||
|
||||
test("scanSource discovers binds, labels, registers, typedefs, and macro domains", () => {
|
||||
const source = [
|
||||
"typedef Struct_(Binds_CubeTri) { U4 PrimCursor; };",
|
||||
"typedef Enum_(U4, PadStatus) { PadStatus_Ok };",
|
||||
"typedef U4 const MipsCode;",
|
||||
"enum { R_PrimCursor = R_T7 atom_reg, C2_Custom = 12, gte_cr_Custom = 13 };",
|
||||
"#define load_word(rt, base, off) enc_i(rt, base, off)",
|
||||
"atom_bind(Binds_CubeTri)",
|
||||
"atom_label(exit)",
|
||||
"atom_offset(entry, exit)",
|
||||
].join("\n");
|
||||
|
||||
const result = scanSource(source, "C:/projects/Pikuma/ps1/code/duffle/mips.h");
|
||||
|
||||
assert.equal(result.index.bindTypes.has("Binds_CubeTri"), true);
|
||||
assert.equal(result.index.types.has("PadStatus"), true);
|
||||
assert.equal(result.index.types.has("MipsCode"), true);
|
||||
assert.equal(result.index.registers.get("R_PrimCursor"), "gpr");
|
||||
assert.equal(result.index.registers.get("C2_Custom"), "cop2");
|
||||
assert.equal(result.index.registers.get("gte_cr_Custom"), "cop2");
|
||||
assert.equal(result.index.macros.get("load_word"), "cpu");
|
||||
assert.equal(result.index.labels.has("entry"), true);
|
||||
assert.equal(result.index.labels.has("exit"), true);
|
||||
});
|
||||
|
||||
test("domainFromPath uses the declaration file rather than parent directory names", () => {
|
||||
assert.equal(domainFromPath("C:/x/code/hello_gte/hello_gte.atom.c"), null);
|
||||
assert.equal(domainFromPath("C:/x/code/duffle/mips.h"), "cpu");
|
||||
assert.equal(domainFromPath("C:/x/code/duffle/gte.h"), "gte");
|
||||
assert.equal(domainFromPath("C:/x/code/duffle/gp.h"), "gpu");
|
||||
});
|
||||
|
||||
test("component aliases inherit the domain of the instructions they emit", () => {
|
||||
const headers = mergeIndexes(
|
||||
scanSource("#define load_word(a,b,c) 1\n#define store_word(a,b,c) 1\n#define shift_aright_var(a,b,c) 1\n#define jump_reg(rd) 1\n", "C:/x/code/duffle/mips.h").index,
|
||||
scanSource("#define gte_sw(rt, base, off) 1\n", "C:/x/code/duffle/gte.h").index
|
||||
);
|
||||
const math = scanSource(
|
||||
[
|
||||
"MipsAtomComp_(ac_load_v3s4) { load_word(R_T0, R_T1, 0) };",
|
||||
"#define mac_load_p3s4 mac_load_v3s4",
|
||||
].join("\n"),
|
||||
"C:/x/code/duffle/math.atom.c"
|
||||
);
|
||||
const shift = scanSource(
|
||||
"MipsAtomComp_(ac_shift_aright_var_v3_self) { shift_aright_var(R_T0, R_T0, R_T1) };",
|
||||
"C:/x/code/duffle/gte.atom.c"
|
||||
);
|
||||
const gte = scanSource(
|
||||
"MipsAtomComp_(ac_gte_store_f3) { gte_sw(C2_SXY0, R_T0, 0) };",
|
||||
"C:/x/code/duffle/gte.atom.c"
|
||||
);
|
||||
const yieldAtom = scanSource(
|
||||
"MipsAtomComp_(ac_yield) { load_word(R_AtomJmp, R_TapePtr, 0), jump_reg(R_AtomJmp), nop };",
|
||||
"C:/x/code/duffle/lottes_tape.h"
|
||||
);
|
||||
|
||||
const merged = mergeIndexes(headers, math.index, shift.index, gte.index, yieldAtom.index);
|
||||
assert.equal(merged.macros.get("mac_load_v3s4"), "cpu");
|
||||
assert.equal(merged.macros.get("mac_load_p3s4"), "cpu");
|
||||
assert.equal(merged.macros.get("mac_shift_aright_var_v3_self"), "cpu");
|
||||
assert.equal(merged.macros.get("mac_gte_store_f3"), "gte");
|
||||
assert.equal(merged.macros.get("mac_yield"), "control");
|
||||
});
|
||||
|
||||
test("scanSource tags utility header defines as utility, not a hardware domain", () => {
|
||||
const source = [
|
||||
"#define assert(cond) ((void)(cond))",
|
||||
"#define stringify(name) #name",
|
||||
"#define u4_hi(imm) ((imm) >> 16)",
|
||||
].join("\n");
|
||||
const result = scanSource(source, "C:/projects/Pikuma/ps1/code/duffle/dsl.h");
|
||||
|
||||
assert.equal(result.index.macros.get("assert"), "utility");
|
||||
assert.equal(result.index.macros.get("stringify"), "utility");
|
||||
assert.equal(result.index.macros.get("u4_hi"), "utility");
|
||||
});
|
||||
|
||||
test("mergeIndexes prefers a hardware domain over a later utility define", () => {
|
||||
const left = createIndex();
|
||||
left.macros.set("sub_s", "utility");
|
||||
const right = createIndex();
|
||||
right.macros.set("sub_s", "cpu");
|
||||
|
||||
assert.equal(mergeIndexes(left, right).macros.get("sub_s"), "cpu");
|
||||
assert.equal(mergeIndexes(right, left).macros.get("sub_s"), "cpu");
|
||||
});
|
||||
|
||||
test("mergeIndexes preserves domain-specific aliases", () => {
|
||||
const left = createIndex();
|
||||
left.macros.set("load_word", "cpu");
|
||||
const right = createIndex();
|
||||
right.componentAliases.add("mac_gte_store");
|
||||
right.macros.set("mac_gte_store", "gte");
|
||||
|
||||
const merged = mergeIndexes(left, right);
|
||||
assert.equal(merged.macros.get("load_word"), "cpu");
|
||||
assert.equal(merged.macros.get("mac_gte_store"), "gte");
|
||||
});
|
||||
@@ -1,24 +1,17 @@
|
||||
This is free and unencumbered software released into the public domain.
|
||||
Copyright (C) 2026 Edward R. Gonzalez
|
||||
|
||||
Anyone is free to copy, modify, publish, use, compile, sell, or
|
||||
distribute this software, either in source code form or as a compiled
|
||||
binary, for any purpose, commercial or non-commercial, and by any
|
||||
means.
|
||||
This software is provided 'as-is', without any express or implied
|
||||
warranty. In no event will the authors be held liable for any damages
|
||||
arising from the use of this software.
|
||||
|
||||
In jurisdictions that recognize copyright laws, the author or authors
|
||||
of this software dedicate any and all copyright interest in the
|
||||
software to the public domain. We make this dedication for the benefit
|
||||
of the public at large and to the detriment of our heirs and
|
||||
successors. We intend this dedication to be an overt act of
|
||||
relinquishment in perpetuity of all present and future rights to this
|
||||
software under copyright law.
|
||||
Permission is granted to anyone to use this software for any purpose,
|
||||
including commercial applications, and to alter it and redistribute it
|
||||
freely, subject to the following restrictions:
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||
IN NO EVENT SHALL THE AUTHORS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
OTHER DEALINGS IN THE SOFTWARE.
|
||||
|
||||
For more information, please refer to <https://unlicense.org>
|
||||
1. The origin of this software must not be misrepresented; you must not
|
||||
claim that you wrote the original software. If you use this software
|
||||
in a product, an acknowledgment in the product documentation would be
|
||||
appreciated but is not required.
|
||||
2. Altered source versions must be plainly marked as such, and must not be
|
||||
misrepresented as being the original software.
|
||||
3. This notice may not be removed or altered from any source distribution.
|
||||
|
||||
@@ -1,467 +0,0 @@
|
||||
/*
|
||||
* atom_dsl.h
|
||||
* ============================================================================
|
||||
*
|
||||
* ATOM DSL — annotation layer for tape atoms (lottes_tape.h).
|
||||
*
|
||||
* This header turns `__attribute__((annotate(...)))` and `_Pragma(...)` into
|
||||
* a small named DSL that the metaprogram can validate against.
|
||||
*
|
||||
* The C compiler treats every macro below as a no-op:
|
||||
* - atom_init / atom_terminate / atom_bind / atom_setup / atom_commit /
|
||||
* atom_annot all expand to `__attribute__((annotate("..."))) MipsAtom_(name)`
|
||||
* — accepted by GCC (with -Wno-attributes), absent at runtime.
|
||||
* - atom_resource / atom_region / atom_group / atom_cadence / atom_async
|
||||
* expand to `_Pragma("...")` — accepted by any C11 preprocessor.
|
||||
*
|
||||
* The metaprogram (tape_atom_annotation_pass.lua) reads the source-as-written
|
||||
* and validates:
|
||||
* - every MipsAtom_ has one atom_*() annotation (no orphans)
|
||||
* - phase is recognized (init/bind/setup/work/commit/terminate)
|
||||
* - reads/writes reference canonical wave-context registers
|
||||
* - rbind atoms reference a real Binds_* struct declaration
|
||||
* - word-counts in tapre metadata agree with the body's actual .word count
|
||||
* - resource/region/group/cadence/async pragmas are spelled correctly and
|
||||
* reference known enum values
|
||||
*
|
||||
* ============================================================================
|
||||
*
|
||||
* PUTTING IT ON AN ATOM — the canonical pattern
|
||||
*
|
||||
* _tape_resources_
|
||||
* atom_resource(cube_tri, "model_ship_cube")
|
||||
* atom_region (cube_tri, PRIM_ARENA)
|
||||
* atom_group (cube_tri, GROUP_RENDER_PRIMS)
|
||||
* atom_cadence (cube_tri, CADENCE_FRAME)
|
||||
*
|
||||
* atom_annot(cube_tri, phase_work,
|
||||
* tape_regs(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
||||
* tape_regs(R_PrimCursor, R_FaceCursor))
|
||||
* internal MipsAtom_(cube_tri) {
|
||||
* atom_label(culling),
|
||||
* // ... atom body ...
|
||||
* atom_label(bounds_chk),
|
||||
* };
|
||||
*
|
||||
* atom_offset(culling, bounds_chk) // ← branch target, validated
|
||||
*
|
||||
* RBIND pattern — `Binds_*` is the contract
|
||||
*
|
||||
* // Wave-context register layout (declarative):
|
||||
* typedef struct Binds_TrackFaceBatch {
|
||||
* U4 R_PrimCursor, R_FaceCursor,
|
||||
* R_VertBase, R_OtBase;
|
||||
* } Binds_TrackFaceBatch;
|
||||
*
|
||||
* atom_resource(rbind_track_face_batch, "track_face_batch_42")
|
||||
* atom_region (rbind_track_face_batch, HEAP_3D)
|
||||
* atom_group (rbind_track_face_batch, GROUP_LOAD_FACES)
|
||||
* atom_cadence (rbind_track_face_batch, CADENCE_ONDEMAND)
|
||||
* atom_async (rbind_track_face_batch, true)
|
||||
*
|
||||
* atom_bind(rbind_track_face_batch, Binds_TrackFaceBatch,
|
||||
* tape_regs(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase))
|
||||
* internal MipsAtom_(rbind_track_face_batch) { ... };
|
||||
*
|
||||
* Annotation rules
|
||||
* ----------------
|
||||
* 1. Each MipsAtom_(name) needs EXACTLY ONE atom_*() macro on the line
|
||||
* immediately above. No annotation = orphan (warning). Two annotations
|
||||
* on the same name = duplicate (error).
|
||||
*
|
||||
* 2. atom_init and atom_terminate take only the name.
|
||||
*
|
||||
* 3. atom_setup and atom_commit take name + reads.
|
||||
*
|
||||
* 4. atom_bind takes name + Binds_* type + writes.
|
||||
*
|
||||
* 5. atom_annot takes name + phase token + reads + writes.
|
||||
* Phase tokens: phase_init / phase_bind / phase_setup / phase_work /
|
||||
* phase_commit / phase_terminate.
|
||||
*
|
||||
* 6. Optional pragmas (atom_resource / atom_region / atom_group /
|
||||
* atom_cadence / atom_async) attach metadata to the atom. They can
|
||||
* appear in any order, with one per atom. They're independent of the
|
||||
* atom_*() macro — multiple pragmatics are fine.
|
||||
*
|
||||
* ============================================================================
|
||||
*
|
||||
* WHY A SEPARATE LAYER (not just put everything in source comments)?
|
||||
*
|
||||
* Source comments are invisible to the compiler. Annotations live in the
|
||||
* source as actual C tokens, so:
|
||||
* - they can never silently get out of sync with the code (the build
|
||||
* fails at preprocessing if the metaprogram disagrees)
|
||||
* - they can be cross-validated against metadata (build fails if a
|
||||
* WORD_COUNT entry drifts away from the .word count in source)
|
||||
* - they make the C compiler a witness ("there's a marker here, and
|
||||
* it's labelled, and it has arguments") without making the C compile
|
||||
* itself do any work
|
||||
*
|
||||
* ============================================================================
|
||||
*/
|
||||
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
#pragma once
|
||||
// #include <stdint.h>
|
||||
#endif
|
||||
|
||||
/* ============================================================================
|
||||
* PHASE TOKENS — strings, used as the second arg to atom_annot(...)
|
||||
*
|
||||
* Why strings? They preserve the metaprogram's ability to read phase directly
|
||||
* from the source-as-written, even when the macro isn't expanded. The Lua
|
||||
* tool also has a MACRO_EXPANSION table for resolving phase_* source-level
|
||||
* references.
|
||||
*
|
||||
* atom_annot(cube_tri, phase_work, ...) ← legal
|
||||
* atom_annot(cube_tri, "work", ...) ← legal (and equivalent)
|
||||
* atom_annot(cube_tri, phase_setup, ...) ← legal
|
||||
*
|
||||
* ============================================================================*/
|
||||
|
||||
#define phase_init "init"
|
||||
#define phase_bind "bind"
|
||||
#define phase_setup "setup"
|
||||
#define phase_work "work"
|
||||
#define phase_commit "commit"
|
||||
#define phase_terminate "terminate"
|
||||
|
||||
/* ============================================================================
|
||||
* WAVE-CONTEXT REGISTERS — canonical register set for the tape wave model.
|
||||
*
|
||||
* The tape-atom runtime carries four registers across a wave:
|
||||
*
|
||||
* R_PrimCursor output pointer into the prim arena (next OT entry to write)
|
||||
* R_FaceCursor input pointer into the face array (next face to consume)
|
||||
* R_VertBase base pointer into the vertex arena (this wave's vertices)
|
||||
* R_OtBase base pointer into the ordering table (this wave's OT slot)
|
||||
*
|
||||
* Each atom declares its reads/writes against this canonical set. The Lua
|
||||
* tool rejects wave-context positions that reference any other register
|
||||
* (warning today — the C compiler's R_T4..R_T7 / R_RA / etc. aliases are
|
||||
* implementation details and not part of the typed surface).
|
||||
*
|
||||
* If your atom needs to touch GTE / SP / DMA / other side state, declare it
|
||||
* at the source level as you normally would — but DO NOT put those registers
|
||||
* in tape_regs(...). Wave-context is a closed set.
|
||||
*
|
||||
* ============================================================================*/
|
||||
|
||||
/* ============================================================================
|
||||
* REGION TOKENS — memory regions atoms may allocate from or write into.
|
||||
*
|
||||
* Use atom_region(name, REGION) to declare. The Lua tool validates that the
|
||||
* region is in this set, AND that:
|
||||
* - rbind atoms declare the source region (usually HEAP_3D or CDROM_STREAM)
|
||||
* - work atoms declare the destination region (the arena they push to)
|
||||
* - commit atoms must declare a region equal to what setup wrote, so the
|
||||
* C-side mirror is consistent
|
||||
*
|
||||
* Add new regions by extending this list and the metaprogram's KNOWN_REGIONS.
|
||||
* Don't add regions ad-hoc — every new region becomes part of the contract.
|
||||
*
|
||||
* ============================================================================*/
|
||||
#define REGION_PRIM_ARENA prim_arena /* OT/prim packet arena */
|
||||
#define REGION_FACE_ARENA face_arena /* face index array */
|
||||
#define REGION_VERTEX_ARENA vertex_arena /* vertex pool */
|
||||
#define REGION_OT_ARENA ot_arena /* ordering-table array */
|
||||
#define REGION_HEAP_3D heap_3d_models /* loaded model heap */
|
||||
#define REGION_CDROM_STREAM cdrom_stream /* CDROM read buffer */
|
||||
#define REGION_VRAM vram_heap /* VRAM texture/GPU buffer */
|
||||
|
||||
/* ============================================================================
|
||||
* CADENCE TOKENS — how often the atom runs.
|
||||
*
|
||||
* frame runs every vsync (rendering, input poll)
|
||||
* once runs exactly once per process lifetime (init, terminate)
|
||||
* ondemand runs when triggered by event (CDROM load, async DMA complete)
|
||||
*
|
||||
* Used as a hint for the metaprogram to flag:
|
||||
* - frame-cadence atoms that have side effects (they'll be hit many times,
|
||||
* so avoid global state mutation unless it's idempotent)
|
||||
* - once-cadence atoms inside "if (frame_count == 0)" guards (the guard
|
||||
* is then provably one-shot, the metaprogram can lift initialization)
|
||||
* - ondemand atoms that are missed by the wave scheduler (forces async
|
||||
* and discards yield results without further processing)
|
||||
*
|
||||
* ============================================================================*/
|
||||
#define CADENCE_FRAME frame
|
||||
#define CADENCE_ONCE once
|
||||
#define CADENCE_ONDEMAND ondemand
|
||||
|
||||
/* ============================================================================
|
||||
* tape_regs(...) — wave-context register list
|
||||
*
|
||||
* tape_regs(R_PrimCursor, R_FaceCursor) → (R_PrimCursor, R_FaceCursor)
|
||||
*
|
||||
* The macro produces a comma-evaluated expression that the C compiler
|
||||
* silently discards (it's wrapped in parentheses in the call argument
|
||||
* position — the result is never bound). The Lua tool pattern-matches the
|
||||
* "tape_regs(...)" token to extract the list.
|
||||
*
|
||||
* You can have at most one tape_regs(...) in the reads slot and one in the
|
||||
* writes slot of atom_annot. To declare multiple disjoint sets (rare), just
|
||||
* declare the union — the metaprogram doesn't track which reads need which
|
||||
* writes at this granularity.
|
||||
*
|
||||
* ============================================================================*/
|
||||
#define atom_reads(...) (__VA_ARGS__)
|
||||
#define atom_writes(...) (__VA_ARGS__)
|
||||
|
||||
/* ============================================================================
|
||||
* ATOM ANNOTATION MACROS
|
||||
*
|
||||
* Each expands to `__attribute__((annotate("kind"))) MipsAtom_(name)` —
|
||||
* the GCC attribute is accepted under -Wno-attributes (already in your
|
||||
* build flags) and stripped at runtime. The annotation string is just the
|
||||
* macro kind ("atom_annot", "atom_bind", etc.) — the metaprogram reads
|
||||
* the macro call's full args list from the source-as-written.
|
||||
*
|
||||
* ============================================================================*/
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* atom_init — entry into tape_runtime_main
|
||||
*
|
||||
* atom_init(tape_main)
|
||||
* internal MipsAtom_(tape_main) { ... };
|
||||
*
|
||||
* Implies: no reads, no writes (wave-context not established yet).
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_init(name) __attribute__((annotate("atom_init")))
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* atom_terminate — exit from tape_runtime_main
|
||||
*
|
||||
* atom_terminate(tape_exit)
|
||||
* internal MipsAtom_(tape_exit) { ... };
|
||||
*
|
||||
* Implies: no reads, no writes (wave-context destroyed at this point).
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_terminate(name) __attribute__((annotate("atom_terminate")))
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* atom_setup — pre-work atom: prepares engine state (e.g., set_gte_world)
|
||||
*
|
||||
* atom_setup(set_gte_world, tape_regs(R_TapePtr))
|
||||
* internal MipsAtom_(set_gte_world) { ... };
|
||||
*
|
||||
* Reads: anything (the engine state you're reading)
|
||||
* Writes: engine state (GTE / DMA / etc. — declared in source, not part of
|
||||
* wave-context, so doesn't go in tape_regs)
|
||||
*
|
||||
* The metaprogram checks that setup is followed (in atomic order) by a work
|
||||
* atom in the same wave — there's no point in setting up state if no one
|
||||
* reads it.
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_setup(name, reads) __attribute__((annotate("atom_setup")))
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* atom_commit — post-work atom: flushes wave-context back to C-side state
|
||||
*
|
||||
* atom_commit(sync_prim_cursor, tape_regs(R_PrimCursor))
|
||||
* internal MipsAtom_(sync_prim_cursor) { ... };
|
||||
*
|
||||
* Reads: wave-context registers (the ones you sync back to C)
|
||||
* Writes: C-side mirror (declared in source — not part of wave-context)
|
||||
*
|
||||
* The metaprogram checks that commit is preceded (in atomic order) by a
|
||||
* work atom that wrote the registers this commit is reading.
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_commit(name, reads) __attribute__((annotate("atom_commit")))
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* atom_bind — rbind atom: read wave-context registers from tape pointer
|
||||
*
|
||||
* atom_bind(rbind_cube_tri, Binds_CubeTri,
|
||||
* tape_regs(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase))
|
||||
* internal MipsAtom_(rbind_cube_tri) { ... };
|
||||
*
|
||||
* The binds_struct MUST be a typedef'd type (declared via
|
||||
* `typedef struct Binds_X { ... } Binds_X;` somewhere in the source).
|
||||
* The Lua tool cross-references this. Missing struct = error.
|
||||
*
|
||||
* Implicit: reads R_TapePtr, writes the four wave-context registers.
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_bind(name, binds_struct, writes) __attribute__((annotate("atom_bind")))
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* atom_annot — generic work atom with explicit phase
|
||||
*
|
||||
* atom_annot(cube_tri, phase_work,
|
||||
* tape_regs(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
||||
* tape_regs(R_PrimCursor, R_FaceCursor))
|
||||
* internal MipsAtom_(cube_tri) { ... };
|
||||
*
|
||||
* Use this for the bulk of your atoms. For init/setup/commit/bind, prefer
|
||||
* the convenience macros above — they pin the phase for you.
|
||||
*
|
||||
* The phase arg is one of: phase_init / phase_bind / phase_setup /
|
||||
* phase_work / phase_commit / phase_terminate. Spelling mistakes are errors.
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_annot(name, phase, reads, writes) __attribute__((annotate("atom_annot")))
|
||||
|
||||
/* ============================================================================
|
||||
* RESOURCE / GROUP / CADENCE / REGION / ASYNC — optional atom metadata
|
||||
*
|
||||
* These don't annotate the atom semantically (phase/reads/writes do that).
|
||||
* They attach extra context that the metaprogram uses to catch:
|
||||
* - same resource loaded twice in different ways
|
||||
* - atoms that span multiple regions (likely bug — pick one)
|
||||
* - frame-cadence atoms that should be once-cadence (perf / correctness)
|
||||
* - ondemand atoms that aren't async (CDROM races)
|
||||
*
|
||||
* You can use as many as apply to a given atom, in any order, immediately
|
||||
* above the atom_*() macro.
|
||||
*
|
||||
* ============================================================================*/
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* atom_resource — name the logical resource the atom references
|
||||
*
|
||||
* atom_resource(cube_tri, "model_ship_cube")
|
||||
* atom_resource(load_track_faces, "track_lavender_field_0x42")
|
||||
* atom_resource(play_engine_sfx, "sfx_engine_loop")
|
||||
*
|
||||
* Use any human-readable string. The metaprogram:
|
||||
* - validates resource strings are non-empty and don't contain control chars
|
||||
* - flags duplicates across atoms with the same name (two atoms claiming
|
||||
* ownership of a resource is usually a refactor artifact or bug)
|
||||
* - flags references to resources that no atom actually defines
|
||||
*
|
||||
* The arg is a STRING LITERAL, so it can't accidentally alias a variable.
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_resource(name, res_id) //_Pragma("atom " #name " resource=" res_id)
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* atom_region — name the memory region the atom touches
|
||||
*
|
||||
* atom_region(cube_tri, REGION_PRIM_ARENA)
|
||||
* atom_region(load_faces, REGION_HEAP_3D)
|
||||
* atom_region(load_tex, REGION_VRAM)
|
||||
*
|
||||
* Use REGION_* tokens above. The metaprogram enforces the closed set.
|
||||
*
|
||||
* Edge cases the metaprogram catches:
|
||||
* - rbind atom that doesn't declare a SOURCE region (where is it loading from?)
|
||||
* - work atom with no destination region (where is it pushing to?)
|
||||
* - region that disagrees with the Binds_* struct layout (you said it's a
|
||||
* prim_arena rbind but the struct has 4 faces in it — wait, that's wrong)
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_region(name, region) //_Pragma("atom " #name " region=" #region)
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* atom_group — bundle atoms into a logical batch (track-load, sound-load, etc.)
|
||||
*
|
||||
* atom_group(load_track_face_42, GROUP_LOAD_FACES)
|
||||
* atom_group(load_track_face_43, GROUP_LOAD_FACES)
|
||||
* atom_group(swap_face_42_43, GROUP_VISIBILITY_SWAP)
|
||||
*
|
||||
* Use any token as the group id. The metaprogram:
|
||||
* - validates all atoms in a group emit their waves in the same tb_group
|
||||
* (no spawning other waves inside a group)
|
||||
* - flags groups with only one member (probably a typo — meant to be a group?)
|
||||
* - validates cross-group edges (no atom reads what another group writes,
|
||||
* unless explicitly grouped together)
|
||||
*
|
||||
* Useful when:
|
||||
* - subdivisible work (track-face batches, polygon subdivision) needs to
|
||||
* confirm that all batches of one logical visible scene are emitted
|
||||
* together
|
||||
* - async loads (CDROM -> VRAM) need to be grouped so all batches complete
|
||||
* before the swap
|
||||
*
|
||||
* Use GROUPS for sound effects to track which sound plays during which atom,
|
||||
* which is needed if the sound tool ever has to validate "this atom is the
|
||||
* trigger for an audio play".
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_group(name, group_id) //_Pragma("atom " #name " group=" #group_id)
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* atom_cadence — declare execution frequency
|
||||
*
|
||||
* atom_cadence(render_frame, CADENCE_FRAME) // every vsync
|
||||
* atom_cadence(load_track_faces, CADENCE_ONDEMAND) // on demand
|
||||
* atom_cadence(init_heap, CADENCE_ONCE) // process lifetime
|
||||
*
|
||||
* Default (no atom_cadence call) is CADENCE_FRAME — most atoms run every
|
||||
* frame. Override explicitly when not.
|
||||
*
|
||||
* The metaprogram's checks:
|
||||
* - CADENCE_ONCE atoms inside `if (frame == 0)` or `if (!initialized)` are
|
||||
* tagged, validating that guards are required (or warning if missing)
|
||||
* - CADENCE_FRAME atoms that mutate state outside the wave context get
|
||||
* flagged (likely a bug — state should persist through commits)
|
||||
* - CADENCE_ONDEMAND atoms must have atom_async — otherwise the trigger
|
||||
* mechanism is undefined
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_cadence(name, cadence) //_Pragma("atom " #name " cadence=" #cadence)
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* atom_async — declare whether the atom yields / interacts with CDROM DMA
|
||||
*
|
||||
* atom_async(load_track_tex, true) // CDROM read yield
|
||||
* atom_async(load_vram, true) // VRAM upload DMA
|
||||
* atom_async(render_frame, false) // pure compute, no async
|
||||
*
|
||||
* The metaprogram requires this for CADENCE_ONDEMAND atoms. For
|
||||
* CADENCE_FRAME, it's optional but documents intent.
|
||||
*
|
||||
* Note: CDROM ATOMS in Psy-Q are typically implemented as a chain of
|
||||
* "async-init" atom followed by a "wait-for-completion" atom. Both atoms
|
||||
* should be marked async=true, and both should have the same resource/group
|
||||
* tag (so the metaprogram can verify they're paired).
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_async(name, is_async) //_Pragma("atom " #name " async=" #is_async)
|
||||
|
||||
/* ============================================================================
|
||||
* WORD-COUNT ANNOTATION FOR A #define MAC
|
||||
*
|
||||
* tape_words(mac_yield, 1)
|
||||
* #define mac_yield() \
|
||||
* load_word(R_AtomJmp, R_TapePtr, 0), \
|
||||
* add_ui_self(R_TapePtr, 4), \
|
||||
* jump_reg(R_AtomJmp), \
|
||||
* nop
|
||||
*
|
||||
* The compiler accepts the unknown _Pragma. The Lua tool reads it and
|
||||
* cross-checks against WORD_COUNT(mac_yield, 1) in tape_atom.metadata.h.
|
||||
* If they disagree, build fails.
|
||||
*
|
||||
* Use sparingly — only on multi-word macros (single-word ones don't need
|
||||
* drift tracking; they're checked by the .word-count pass anyway).
|
||||
*
|
||||
* ============================================================================*/
|
||||
#define tape_words(name, n) //_Pragma(#name " tape_atom words=" #n)
|
||||
|
||||
/* ============================================================================
|
||||
* atom_label / atom_offset — branch target machinery
|
||||
*
|
||||
* atom_label(culling) ← nothing in C; anchor only
|
||||
* ... body ...
|
||||
* atom_label(bounds_chk) ← another anchor
|
||||
*
|
||||
* atom_offset(culling, bounds_chk) ← resolved by gen/.offsets.h
|
||||
*
|
||||
* The metaprogram generates gen/atom_offsets.h with one
|
||||
* #define atom_offset__culling__bounds_chk ((target - branch_pos - 1))
|
||||
* per atom_offset(F, T) call. The preprocessor then expands your call to
|
||||
* the right immediate value.
|
||||
*
|
||||
* If gen/atom_offsets.h is stale (or atom_label(name) is undefined),
|
||||
* `atom_offset__F__T` becomes an undefined macro and the C build fails.
|
||||
* This catches:
|
||||
* - typo in atom_label (no anchor → metaprogram doesn't emit the macro)
|
||||
* - .offsets.h not regenerated after body edits
|
||||
* - body edit that broke the offset math (recompile + retest picks it up
|
||||
* in CPU emulator)
|
||||
*
|
||||
* ============================================================================*/
|
||||
|
||||
#define atom_offset(F, T) atom_offset_ ## F ## _ ## T
|
||||
/* atom_label is a pure annotation for the metaprogram's offset calculations.
|
||||
* The macro expands to a C comment, so the C preprocessor strips it to
|
||||
* whitespace — NO instruction word is emitted in the asm. The metaprogram
|
||||
* still recognises the literal `atom_label(name)` token in source and
|
||||
* records the marker at the current pos. */
|
||||
#define atom_label(name) /* atom_label anchor: name */
|
||||
@@ -0,0 +1,15 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
#endif
|
||||
|
||||
enum {
|
||||
bios_init_pad_2 = 0x12,
|
||||
bios_start_pad_2 = 0x13,
|
||||
bios_flushcache = 0x44,
|
||||
bios_table_addr = 0xA0,
|
||||
bios_btable_addr = 0xB0,
|
||||
};
|
||||
|
||||
enum {
|
||||
bios_pad_buffer_size = 0x22,
|
||||
};
|
||||
@@ -0,0 +1,181 @@
|
||||
/*
|
||||
* dsl.atom.h
|
||||
* ============================================================================
|
||||
*
|
||||
* ATOM DSL: Annotation layer for tape atoms (lottes_tape.h).
|
||||
* The metaprogram (scripts/passes/annotation.lua) reads source-as-written and validates:
|
||||
* - atom_info(...) shape: up to three sub-calls (atom_bind(Binds_X), atom_reads(...), atom_writes(...)) in any order and are optional.
|
||||
* - rbind atoms (atom_info(..., atom_bind(Binds_X), ...)) reference a real Binds_* struct declaration.
|
||||
* - atom word-counts in word_counts.metadata.h match the body's actual .word count.
|
||||
*
|
||||
* Pure macro anntation.
|
||||
* ---------------
|
||||
* Don't want to constraint the macro usage to some attribute placment constraint, etc, don't want ot dela with the compiler.
|
||||
* atom_info, atom_bind, atom_reads, atom_writes, atom_label, atom_dbg_skip each expand to a C comment or to nothing
|
||||
* (C preprocessor strips them to whitespace).
|
||||
*
|
||||
* ============================================================================
|
||||
* Usage:
|
||||
* MipsAtom_(cube_tri) atom_info(
|
||||
* atom_reads (R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
* , atom_writes(R_PrimCursor, R_FaceCursor)
|
||||
* ){
|
||||
* atom_label(culling),
|
||||
* // ... atom body ...
|
||||
* atom_offset(culling, bounds_chk) // branch target, validated
|
||||
* // ... atom body ...
|
||||
* atom_label(bounds_chk),
|
||||
* };
|
||||
*
|
||||
*
|
||||
* Data Binding pattern -- atom_bind as a sub-call of atom_info
|
||||
*
|
||||
* // Wave-context register layout (declarative):
|
||||
* typedef Struct_(Binds_TrackFaceBatch) {
|
||||
* U4 PrimCursor;
|
||||
* U4 FaceCursor;
|
||||
* U4 VertBase;
|
||||
* U4 OtBase;
|
||||
* };
|
||||
* MipsAtom_(rbind_track_face_batch) atom_info(
|
||||
* atom_bind(Binds_TrackFaceBatch)
|
||||
* , atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
* ){ ... };
|
||||
*
|
||||
* Annotation rules
|
||||
* ----------------
|
||||
* 1. atom_info(...) is OPTIONAL. Atoms without atom_info are silently skipped by the metaprogram.
|
||||
* 2. If present, atom_info takes up to three sub-calls, all order-independent within the arg list:
|
||||
* - atom_bind(Binds_X)
|
||||
* - atom_reads(...)
|
||||
* - atom_writes(...)
|
||||
* 3. atom_bind(Binds_X): metaprogram cross-references Binds_X against the `typedef struct Binds_X { ... } Binds_X;` declaration.
|
||||
* 4. atom_reads(...) and atom_writes(...): Used to to check if registers are used correctly in macros: R_PrimCursor / R_FaceCursor / R_VertBase / R_OtBase.
|
||||
* 5. atom_label(name: Utilize with atom_offset as a target location.
|
||||
* 6. atom_offset(F, T): Resolved by gen/atom_offsets.h, generated from the atom_label markers. Calculated during the offset pass of the lua metaprogram.
|
||||
*/
|
||||
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
#pragma once
|
||||
#endif
|
||||
|
||||
/* ============================================================================
|
||||
* atom_reads(...) / atom_writes(...)
|
||||
*
|
||||
* Used during the static analysis pass of the metaprogram to do
|
||||
* ============================================================================*/
|
||||
#define atom_reads(...) (__VA_ARGS__)
|
||||
#define atom_writes(...) (__VA_ARGS__)
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* atom_reg (per-enum opt-in marker for the DWARF register-alias registry)
|
||||
*
|
||||
* Bare `atom_reg` token adjacent to an enum entry that alias as debug-visible for scan_source's register_alias_registry.
|
||||
* Lua scanner reads the bare token.
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_reg /* atom_reg: opt the preceding enum entry into the DWARF registry */
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// atom_auto_reg(atom, sym) — per-atom auto-allocated GPR binding.
|
||||
// enum {
|
||||
// atom_auto_reg(cube_g4_face, R_Fwdx), // expands to: R_Fwdx = R_Fwdx_Code /* atom_auto_reg: cube_g4_face */,
|
||||
// atom_auto_reg(cube_g4_face, R_Eye_z) atom_type(S4), // atom_type chains after
|
||||
// };
|
||||
// (The macro IS the entire enum entry — no separate LHS=RHS. The `atom` scope is
|
||||
// preserved in a trailing C-comment on the RHS so the Lua scanner can recover
|
||||
// it after preprocessing strips the macro form. R_<Sym>_Code is resolved from gen/auto_reg.h which the .c file #include's before the enum declaration.)
|
||||
#define atom_auto_reg(atom, sym) sym = sym ## _Code /* atom_auto_reg: atom */
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// phase_auto_reg(phase, sym) — per-phase auto-allocated GPR binding.
|
||||
// enum {
|
||||
// phase_auto_reg(cube_g4, R_Temp0), // expands to: R_Temp0 = R_Temp0_Code /* phase_auto_reg: cube_g4 */,
|
||||
// phase_auto_reg(cube_g4, R_Temp1),
|
||||
// };
|
||||
// (Same macro-as-enum-entry form as atom_auto_reg above; the `phase` scope is preserved in a trailing C-comment on the RHS for the Lua scanner to recover.)
|
||||
#define phase_auto_reg(phase, sym) sym = sym ## _Code /* phase_auto_reg: phase */
|
||||
|
||||
/* ============================================================================
|
||||
* atom_info :
|
||||
* MipsAtom_(cube_tri) atom_info(
|
||||
* atom_reads (R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
* , atom_writes(R_PrimCursor, R_FaceCursor)
|
||||
* ){ ... };
|
||||
*
|
||||
* - atom_bind(Binds_X): metaprogram cross-references Binds_X against the `typedef struct Binds_X { ... } Binds_X;` declaration.
|
||||
* - atom_reads(...): comma-list of registers
|
||||
* - atom_writes(...): comma-list of registers
|
||||
* ============================================================================*/
|
||||
#define atom_info(...) /* atom_info(__VA_ARGS__) */
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* DEBUG SOURCE-STEP MARKER
|
||||
*
|
||||
* Place `atom_dbg_skip` (BARE) before a MipsAtom_, MipsAtomComp_, or MipsAtomComp_Proc_.
|
||||
* The following declaration kind determines whether the marker selects a whole atom or a component inline view.
|
||||
* The source scanner associates the marker with that declaration; placement diagnostics are handled by the annotation pass.
|
||||
*
|
||||
* Example:
|
||||
* atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(rret_addr), nop };
|
||||
* atom_dbg_skip MipsAtomComp_(ac_yield) { ... };
|
||||
* atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, { ... });
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_dbg_skip /* atom_dbg_skip: skip the following atom or component source view */
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* Typed-view annotations (Registry for DWARF RR_<R_X> chain resolution)
|
||||
* atom_type(<T>) -- overloaded:
|
||||
* (a) enum-site default: `R_Foo = R_Tn, atom_reg atom_type(T)`
|
||||
* Sets the per-alias default typed view in the register_alias_registry.
|
||||
* Consumed by the DWARF chain step (e) when no per-atom atom_ctx / atom_phase / atom_type callsite provides a stronger resolution.
|
||||
* (b) callsite override: `atom_reads(R_Foo atom_type(T), ...)` Overrides the per-alias default for THIS atom only.
|
||||
* Last-write-wins per R_Name; conflict -> error.
|
||||
* atom_ctx(<atom_name>) -- atom-info sub-call:
|
||||
* Propagate another atom's atom.rbind.fields (its Binds_* typed fields) into THIS atom's typed-view resolution.
|
||||
* The named atom must be an rbind atom (have `atom_bind(Binds_X)` in its `atom_info`).
|
||||
* Used as the escape hatch when atom_phase is not the natural correlation.
|
||||
* atom_phase(<label>) -- atom-info sub-call:
|
||||
* Free-form C-identifier label for grouping atoms.
|
||||
* Within a phase, the FIRST atom in source-order that owns its own atom.rbind provides
|
||||
* the Binds_* field types used by all other atoms in the same phase.
|
||||
* The preferred correlation mechanism; atom_ctx is the escape hatch for non-natural cases.
|
||||
*
|
||||
* All three expand to C comments
|
||||
* (the bare-token convention matching `atom_reg` and `atom_dbg_skip`).
|
||||
* The Lua scanner reads the bare tokens in source-as-written; the C preprocessor strips them.
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_type(T) /* atom_type: associate <T> with the preceding enum entry (enum site) or this register (atom-info site) */
|
||||
#define atom_ctx(atom_name) /* atom_ctx: propagate <atom_name>'s Binds_* field types into this atom's typed views */
|
||||
#define atom_phase(label) /* atom_phase: tag this atom with <label> for grouped typed-view resolution */
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* atom_bind(Binds_X) -- rbind sub-call of atom_info
|
||||
*
|
||||
* MipsAtom_(rbind_cube_tri) atom_info(
|
||||
* atom_bind(Binds_CubeTri)
|
||||
* , atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
* ){ ... };
|
||||
*
|
||||
* The Binds_X MUST be a typedef'd type (declared via `typedef struct Binds_X { ... } Binds_X;` somewhere in the source).
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_bind(binds_struct) /* atom_bind(binds_struct) */
|
||||
|
||||
#define Binds_(type) (tmpl(Binds,type)) // TODO(Ed): Do we want to use this?
|
||||
|
||||
/* ============================================================================
|
||||
* atom_label / atom_offset — branch target machinery
|
||||
*
|
||||
* atom_label(culling) ← nothing in C; anchor only
|
||||
* ... body ...
|
||||
* atom_label(bounds_chk) ← another anchor
|
||||
*
|
||||
* atom_offset(culling, bounds_chk) ← resolved by gen/offsets.h
|
||||
*
|
||||
* The metaprogram generates gen/offsets.h with one #define with the offset value per atom_offset(F, T) call.
|
||||
* The preprocessor then expands the call to the right immediate value.
|
||||
*
|
||||
* If gen/offsets.h is stale (or atom_label(name) is undefined), `atom_offset_F_T` becomes an undefined macro and the C build fails.
|
||||
* ============================================================================*/
|
||||
#define atom_offset(F, T) atom_offset_ ## F ## _ ## T
|
||||
// atom_label is a pure annotation for the metaprogram's offset calculations.
|
||||
#define atom_label(name) /* atom_label anchor: name */
|
||||
+35
-21
@@ -3,7 +3,7 @@
|
||||
# include "assert.h"
|
||||
#endif
|
||||
|
||||
#define offset_of(type, member) cast(U8,__builtin_offsetof(type,member))
|
||||
#define offset_of(type, member) cast(U8,__builtin_offsetof(type,member)) // Compiler builtin version of O_
|
||||
#define static_assert _Static_assert
|
||||
#define typeof __typeof__
|
||||
#define typeof_ptr(ptr) typeof((ptr)[0])
|
||||
@@ -28,8 +28,9 @@
|
||||
#define internal static // internal
|
||||
|
||||
#define asm __asm__
|
||||
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
||||
|
||||
#define A_(data) (& data)
|
||||
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
||||
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
||||
#define C_(type,data) ((type)(data)) // for enforced precedence
|
||||
#define expect_(x, y) __builtin_expect(x, y) // so compiler knows the common path
|
||||
@@ -43,7 +44,9 @@
|
||||
|
||||
#define R_ restrict
|
||||
#define V_ volatile
|
||||
// Fictional, used for intiution.
|
||||
|
||||
#pragma region Fictional //, used for intiution
|
||||
|
||||
#define EUB_ restrict // Execute Unit Bound: Data is siloed in the ALU Register File. The Load/Store Unit is bypassed. (Route to Execution Unit. Keep in registers)
|
||||
#define ISO_ restrict // Isolated Provenance: Alternative to Exu_. Guarantees electrical memory isolation,
|
||||
// unlocking the compiler’s ability to safely pack data across multiple parallel SIMD lanes (vectorization).
|
||||
@@ -67,7 +70,8 @@
|
||||
#define latch_load_anchor(ptr) //__atomic_load_n(ptr, ooo_anchor_)
|
||||
#define latch_store_drain(ptr, val) //__atomic_store_n(ptr, val, ooo_drain_)
|
||||
#define pulse_xchg_weld(ptr, val) //__atomic_exchange_n(ptr, val, ooo_weld_)
|
||||
//end of: Fictional.
|
||||
|
||||
#pragma endreigon Fictional
|
||||
|
||||
|
||||
// R_ (restrict) establishes an "Eigen" or "Proprius" mapping.
|
||||
@@ -87,12 +91,13 @@
|
||||
#define PtrSet_(type) TypeR_(type); typedef TypeV_(type)
|
||||
#define TSet_(type) type; typedef PtrSet_(type)
|
||||
|
||||
#define array_len(a) (U4)(sizeof(a) / sizeof(typeof((a)[0])))
|
||||
#define array_decl(type, ...) (type[]){__VA_ARGS__}
|
||||
#define Array_len(a) (U4)(sizeof(a) / sizeof(typeof((a)[0])))
|
||||
#define Array_decl(type, ...) (type[]){__VA_ARGS__}
|
||||
#define Array_sym(type,len) A ## len ## _ ## type
|
||||
#define Array_expand(type,len) type Array_sym(type, len)[len]; typedef PtrSet_(Array_sym(type, len))
|
||||
#define Array_(type,len) Array_expand(type,len)
|
||||
#define Bit_(id,b) id = (1 << b), tmpl(id,pos) = b
|
||||
#define Bitmask_(b) (1u << b)
|
||||
#define Enum_(underlying_type, symbol) underlying_type TSet_(symbol); enum symbol
|
||||
#define Proc_(symbol) symbol
|
||||
#define Relative_(symbol) // Does nothing but annotate that a symbol is associated with another.
|
||||
@@ -130,20 +135,21 @@ typedef __UINT32_TYPE__ TSet_(B4);
|
||||
#define u4_v(value) C_(U4 V_*, value)
|
||||
enum { false = 0, true = 1, true_overflow, };
|
||||
|
||||
#define u4_lo(value) ((value) & 0xFFFFU)
|
||||
#define u4_hi(value) ((value) >> 12)
|
||||
#define u4_lo(value) (u4_(value) & 0xFFFFU)
|
||||
#define u4_hi(value) (u4_(value) >> (S_(U2) * 8))
|
||||
|
||||
typedef void Proc_(VoidFn) (void);
|
||||
|
||||
#define kilo(n) (C_(U4, n) << 10)
|
||||
#define mega(n) (C_(U4, n) << 20)
|
||||
#define giga(n) (C_(U4, n) << 30)
|
||||
#define tera(n) (C_(U4, n) << 40)
|
||||
#define Kilo_(n) (C_(U4, n) << 10)
|
||||
#define Mega_(n) (C_(U4, n) << 20)
|
||||
#define Giga_(n) (C_(U4, n) << 30)
|
||||
#define Tera_(n) (C_(U4, n) << 40)
|
||||
|
||||
#define null C_(U4, 0)
|
||||
#define nullptr C_(void*, 0)
|
||||
#define O_(type, field) (C_(U4, & C_(type*,0)->field))
|
||||
|
||||
#define OT_(field) O_(typeof_ptr(& field), filed))
|
||||
#define O_(type, field) C_(U4, & C_(type*,0)->field)
|
||||
#define OA_(type, aexpr) C_(U4, & C_(type*,0) aexpr)
|
||||
#define OT_(field) O_(typeof_ptr(& field), field))
|
||||
#define S_(data) C_(U4, sizeof(data))
|
||||
|
||||
#define sop_1(op,a,b) C_(U1, s1_(a) op s1_(b))
|
||||
@@ -164,6 +170,8 @@ def_signed_ops(le, <=)
|
||||
#undef def_signed_ops
|
||||
#undef def_signed_op
|
||||
|
||||
// Unused, we arent' doing any C-like asm since we have the asm dsl. We'll keep the non-generics if we somehow do.
|
||||
#if 0
|
||||
#define def_generic_sop(op, a, ...) _Generic((a), U1: op ## _s1, U2: op ## _s2, U4: op ## _s4) (a, __VA_ARGS__)
|
||||
#define add_s(a,b) def_generic_sop(add,a,b)
|
||||
#define sub_s(a,b) def_generic_sop(sub,a,b)
|
||||
@@ -173,11 +181,12 @@ def_signed_ops(le, <=)
|
||||
#define ge_s(a,b) def_generic_sop(ge, a,b)
|
||||
#define le_s(a,b) def_generic_sop(le, a,b)
|
||||
#undef def_generic_sop
|
||||
#endif
|
||||
|
||||
#define alignas _Alignas
|
||||
#define alignof _Alignof
|
||||
#define byte_pad(amount, ...) B1 glue(_PAD_, __VA_ARGS__) [amount]
|
||||
#define pcast(type, data) (C_(type*, & (data)) [0])
|
||||
#define C_ptr(type, data) (C_(type*, & (data)) [0])
|
||||
|
||||
#define dbg_args(...) __VA_ARGS__
|
||||
|
||||
@@ -192,6 +201,8 @@ def_signed_ops(le, <=)
|
||||
#define defer_info(type,expr, ...) for(type info= {__VA_ARGS__}; info.once!=1;++info.once,(expr)) // Defer with tracked state
|
||||
|
||||
#define do_while(cond) for (U8 once=0; once!=1 || (cond); ++once)
|
||||
|
||||
#define Jmp_nZero_(cond,label) if (cond) goto label;
|
||||
#pragma endregion Control Flow & Iteration
|
||||
|
||||
#define span_iter(type, iter, m_begin, op, m_end) ( \
|
||||
@@ -208,13 +219,16 @@ def_signed_ops(le, <=)
|
||||
typedef Span_(S4);
|
||||
typedef Span_(U4);
|
||||
|
||||
#if 0
|
||||
#pragma region Debug
|
||||
#define debug_trap() __builtin_debugtrap()
|
||||
#define debug_trap() __builtin_trap()
|
||||
#if BUILD_DEBUG
|
||||
IA_ void assert(U8 cond) { if(cond){return;} else{debug_trap(); ms_exit_process(1);} }
|
||||
#define assert(cond) if(cond == false){debug_trap();}
|
||||
#else
|
||||
#define assert(cond)
|
||||
# ifndef assert
|
||||
# include <assert.h>
|
||||
# endif
|
||||
#endif
|
||||
#pragma endregion Debug
|
||||
#endif
|
||||
|
||||
#define GCC_OPTIMIZATION_DISABLE _Pragma("GCC push_options") _Pragma("GCC optimize(\"O0\")")
|
||||
#define GCC_OPTIMIZATION_ENABLE _Pragma("GCC pop_options")
|
||||
|
||||
+14
-22
@@ -50,17 +50,13 @@
|
||||
#define asm_words(...) m_expand(glue(GCC_ASM_INL_, GCC_ASM_COUNT_ARGS(__VA_ARGS__))(__VA_ARGS__))
|
||||
// Very nasty macro expansion. See the Cruft pragma region after all the DSL defines
|
||||
|
||||
/* reg_str(n) — Stringify an integer register id into the GCC asm
|
||||
* string form (e.g. 12 → "$12"). Use this anywhere GCC's parser
|
||||
* expects a literal string identifying a register: clobber lists,
|
||||
* asm templates, etc. The two-level macro is the standard preprocessor
|
||||
* idiom for forcing one level of expansion before stringify — without
|
||||
* it, `#n` would stringify the macro name `R_T4` to `"R_T4"` instead
|
||||
* of expanding `R_T4` to its value first.
|
||||
/* reg_str(n) — Stringify an integer register id into the GCC asm string form (e.g. 12 → "$12").
|
||||
* Use this anywhere GCC's parser expects a literal string identifying a register: clobber lists,
|
||||
* asm templates, etc. The two-level macro is the standard preprocessor idiom for forcing one level of expansion before stringify —
|
||||
* without it, `#n` would stringify the macro name `R_T4` to `"R_T4"` instead of expanding `R_T4` to its value first.
|
||||
*
|
||||
* For declaring a register variable bound to a specific GPR, use the
|
||||
* `rgcc(n)` bundle from gcc_asm.h instead — it adds the `__asm__()`
|
||||
* qualifier around the string.
|
||||
* For declaring a register variable bound to a specific GPR, use the `rgcc(n)` bundle from gcc_asm.h instead —
|
||||
* it adds the `__asm__()` qualifier around the string.
|
||||
*
|
||||
* register V3_S2* p0 __asm__(reg_str(R_T4)) = ...; // verbose
|
||||
* register V3_S2* p0 rgcc(R_T4) = ...; // bundled
|
||||
@@ -85,7 +81,7 @@
|
||||
* - The string "$12" is derived from it via reg_str, so they cannot drift apart.
|
||||
* - Spelling `__asm__(reg_str(R_T4_Code))` at every call site is noise.
|
||||
*
|
||||
* tmpl defined in dsl.h (the token-paste glue).
|
||||
* tmpl defined in dsl.h (token-paste glue).
|
||||
* rgcc define here (gcc_asm.h) because the `__asm__` keyword is GCC-specific.
|
||||
* Anyone porting to a different compiler's asm dialect overrides rgcc,
|
||||
* and the integer→string derivation in rlit can be retargeted in one place.
|
||||
@@ -94,12 +90,10 @@
|
||||
* ------------------------------------------------------------------------ */
|
||||
#define rgcc(n) __asm__(rlit(n))
|
||||
|
||||
/* rgcc_ref(n) — GCC operand-reference form "%N". Not currently used
|
||||
* by the placeholder-pun macros (the .word bodies are fully baked
|
||||
* at compile time and have no runtime operand references), but kept
|
||||
* here for completeness in case a future asm template needs to refer
|
||||
* to a runtime input by position. Mirror of rgcc but produces "%N"
|
||||
* instead of "$N". */
|
||||
/* rgcc_ref(n) — GCC operand-reference form "%N". Not currently used by the placeholder-pun macros
|
||||
* (the .word bodies are fully baked at compile time and have no runtime operand references),
|
||||
* but kept here for completeness in case a future asm template needs to refer to a runtime input by position.
|
||||
* Mirror of rgcc but produces "%N" instead of "$N". */
|
||||
#define rgcc_ref_(n) "%" #n
|
||||
#define rgcc_ref(n) rgcc_ref_(n)
|
||||
|
||||
@@ -147,11 +141,9 @@
|
||||
9, 8, 7, 6, 5, 4, 3, 2, 1, 0))
|
||||
|
||||
/* --- 2. String Concatenation Helpers --- *
|
||||
* NOTE: we use `%0`, `%1`, ... not `%c0`, `%c1`, ... because GCC's
|
||||
* asm-parser rejects `%cN` in this position with "invalid use of '%c'".
|
||||
* The `%cN` form is for printing *character* constants; for arbitrary
|
||||
* integer immediates (the only kind `"i"(...)` produces), the plain
|
||||
* `%N` form is the right one. Both expand to the bare immediate.
|
||||
* NOTE: we use `%0`, `%1`, ... not `%c0`, `%c1`, ... because GCC's asm-parser rejects `%cN` in this position with "invalid use of '%c'".
|
||||
* The `%cN` form is for printing *character* constants; for arbitrary integer immediates (the only kind `"i"(...)` produces),
|
||||
* the plain `%N` form is the right one. Both expand to the bare immediate.
|
||||
*/
|
||||
#define GCC_ASM_W1 "%0"
|
||||
#define GCC_ASM_W2 GCC_ASM_W1 ", %1"
|
||||
|
||||
@@ -1,128 +0,0 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
#pragma once
|
||||
#endif
|
||||
// Auto-generated by tape_atom_annotation_pass.lua — DO NOT EDIT
|
||||
// Source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
||||
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
|
||||
// + auto word-counts (so tape_atom.metadata.h stays manual-only
|
||||
// for encoding macros).
|
||||
|
||||
#ifndef WORD_COUNT
|
||||
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
||||
#endif
|
||||
|
||||
#define mac_yield(...) \
|
||||
load_word(R_AtomJmp, R_TapePtr, 0) \
|
||||
, add_ui_self( R_TapePtr, S_(MipsCode)) \
|
||||
, jump_reg( R_AtomJmp) \
|
||||
, nop
|
||||
WORD_COUNT(mac_yield, 4)
|
||||
|
||||
/* Words: 3; Loads 3 S2 indices from the face array */
|
||||
#define mac_load_tri_indices(...) \
|
||||
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)) \
|
||||
, load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)) \
|
||||
, load_half_u(R_T2, R_FaceCursor, 2 * S_(S2))
|
||||
WORD_COUNT(mac_load_tri_indices, 3)
|
||||
|
||||
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
|
||||
#define mac_load_tri_verts(...) \
|
||||
shift_lleft(R_AT, R_T0, v3s2_byteoff) \
|
||||
, add_u_self(R_AT, R_VertBase) \
|
||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||
, gte_mv_to_data_r(R_V0, C2_VXY0) \
|
||||
, gte_mv_to_data_r(R_V1, C2_VZ0) \
|
||||
, shift_lleft(R_AT, R_T1, v3s2_byteoff) \
|
||||
, add_u_self(R_AT, R_VertBase) \
|
||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||
, gte_mv_to_data_r(R_V0, C2_VXY1) \
|
||||
, gte_mv_to_data_r(R_V1, C2_VZ1) \
|
||||
, shift_lleft(R_AT, R_T2, v3s2_byteoff) \
|
||||
, add_u_self(R_AT, R_VertBase) \
|
||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||
, gte_mv_to_data_r(R_V0, C2_VXY2) \
|
||||
, gte_mv_to_data_r(R_V1, C2_VZ2)
|
||||
WORD_COUNT(mac_load_tri_verts, 18)
|
||||
|
||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
|
||||
* Hardcoded for Poly_F3 (5 words). For Poly_G4, use ac_insert_ot_tag_g4. */
|
||||
#define mac_insert_ot_tag_f3(...) \
|
||||
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
|
||||
, add_u_self( R_T1, R_OtBase) /* T1 = & OrderingTable[OTZ] */ \
|
||||
, load_word( R_AT, R_T1, O_(PolyTag,bf_addr_len)) /* AT = old_ot_head */ \
|
||||
, load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << polytag_len_bits) /* V0 = (5 - 1) << 24 = 4 << 24 */ \
|
||||
, mask_upper( R_AT, R_AT, S_(polytag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
|
||||
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
|
||||
, store_word( R_AT, R_PrimCursor, O_(PolyTag,bf_addr_len)) /* prim->tag = packed(prim_length, old_addr) */ \
|
||||
, shift_lleft( R_AT, R_PrimCursor, S_(polytag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
|
||||
, shift_lright(R_AT, R_AT, S_(polytag_len_bits)) \
|
||||
, store_word( R_AT, R_T1, O_(PolyTag,bf_addr_len)) /* OrderingTable[OTZ] = PrimCursor */
|
||||
WORD_COUNT(mac_insert_ot_tag_f3, 11)
|
||||
|
||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
|
||||
* Hardcoded for Poly_G4 (9 words). For Poly_F3, use ac_insert_ot_tag_f3. */
|
||||
#define mac_insert_ot_tag_g4(...) \
|
||||
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
|
||||
, add_u_self( R_T1, R_OtBase) /* T1 = & OrderingTable[OTZ] */ \
|
||||
, load_word( R_AT, R_T1, O_(PolyTag,bf_addr_len)) /* AT = old_ot_head */ \
|
||||
, load_upper_i(R_V0, (S_(Poly_G4)/S_(U4) - S_(PolyTag)/S_(U4)) << polytag_len_bits) /* V0 = (9 - 1) << 24 = 8 << 24 */ \
|
||||
, mask_upper( R_AT, R_AT, S_(polytag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
|
||||
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
|
||||
, store_word( R_AT, R_PrimCursor, O_(PolyTag,bf_addr_len)) /* prim->tag = packed(prim_length, old_addr) */ \
|
||||
, shift_lleft( R_AT, R_PrimCursor, S_(polytag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
|
||||
, shift_lright(R_AT, R_AT, S_(polytag_len_bits)) \
|
||||
, store_word( R_AT, R_T1, O_(PolyTag,bf_addr_len)) /* OrderingTable[OTZ] = PrimCursor */
|
||||
WORD_COUNT(mac_insert_ot_tag_g4, 11)
|
||||
|
||||
#define mac_pack_color_word(off, code, r, g, b) \
|
||||
load_upper_i(R_AT, (code) << 8 | (b)) \
|
||||
, or_i_self( R_AT, ((g) << 8) | (r)) \
|
||||
, store_word( R_AT, R_PrimCursor, (off))
|
||||
WORD_COUNT(mac_pack_color_word, 3)
|
||||
|
||||
#define mac_format_f3_color(r, g, b) \
|
||||
mac_pack_color_word(O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b)
|
||||
WORD_COUNT(mac_format_f3_color, 3)
|
||||
|
||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
|
||||
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
|
||||
#define mac_gte_store_f3_post_rtpt(...) \
|
||||
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_F3,p0)) \
|
||||
, gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_F3,p1)) \
|
||||
, gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_F3,p2))
|
||||
WORD_COUNT(mac_gte_store_f3_post_rtpt, 3)
|
||||
|
||||
#define mac_format_g4_color(r0, g0, b0, r1, g1, b1, r2, g2, b2, r3, g3, b3) \
|
||||
mac_pack_color_word(O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0) \
|
||||
, mac_pack_color_word(O_(Poly_G4,c1), 0, r1,g1,b1) \
|
||||
, mac_pack_color_word(O_(Poly_G4,c2), 0, r2,g2,b2) \
|
||||
, mac_pack_color_word(O_(Poly_G4,c3), 0, r3,g3,b3)
|
||||
WORD_COUNT(mac_format_g4_color, 12)
|
||||
|
||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
|
||||
* G4 triangle portion to p0/p1/p2.
|
||||
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen,
|
||||
* SXY2=v2.screen). MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2
|
||||
* get overwritten with v3 (RTPS writes only to SXY2, but to keep the
|
||||
* three registers aligned with v0/v1/v2 you must store before RTPS).
|
||||
* The macro name declares the pipeline position; check #6 (GTE state-
|
||||
* machine validation) verifies the call site matches the declaration. */
|
||||
#define mac_gte_store_g4_p012_post_rtpt_pre_rtps(...) \
|
||||
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_G4,p0)) \
|
||||
, gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_G4,p1)) \
|
||||
, gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2))
|
||||
WORD_COUNT(mac_gte_store_g4_p012_post_rtpt_pre_rtps, 3)
|
||||
|
||||
/* Words: 1; Stores the V3 screen coord to the G4's p3 slot.
|
||||
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its
|
||||
* single-vertex result to SXY2; SXY0 still holds v0.screen from the
|
||||
* earlier RTPT — DO NOT read SXY0 here, that's the bug this name
|
||||
* prevents).
|
||||
*/
|
||||
#define mac_gte_store_g4_p3_post_rtps(...) \
|
||||
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p3))
|
||||
WORD_COUNT(mac_gte_store_g4_p3_post_rtps, 1)
|
||||
|
||||
@@ -1,9 +0,0 @@
|
||||
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
|
||||
// Source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
||||
#pragma once
|
||||
|
||||
#pragma region lottes_tape
|
||||
|
||||
|
||||
#pragma endregion lottes_tape
|
||||
|
||||
@@ -0,0 +1,408 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
#pragma once
|
||||
#endif
|
||||
// Auto-generated by ps1_meta.lua — DO NOT EDIT
|
||||
// Directory: C:\projects\Pikuma\ps1\code\duffle/
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\word_count.metadata.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\memory.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\math.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\gcc_asm.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\mips.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\gp.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\gte.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.atom.c
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.atom.c
|
||||
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
|
||||
|
||||
#ifndef WORD_COUNT
|
||||
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
||||
#endif
|
||||
|
||||
/* atom_dbg_skip */
|
||||
/* ---------------------------------------------------------------------------
|
||||
* MACRO ATOM Components (Reusable Assembly Components)
|
||||
* These do NOT yield. They are expanded inline inside Tape Atoms.
|
||||
* ---------------------------------------------------------------------------*/
|
||||
// The 'Yield' sequence for Tape Atoms (mac_yield).
|
||||
#define mac_yield(...) \
|
||||
load_word(R_AtomJmp, R_TapePtr, 0) \
|
||||
LdSlot_ \
|
||||
, add_ui_self( R_TapePtr, S_(MipsCode)) \
|
||||
, jump_reg( R_AtomJmp) \
|
||||
, BdSlot_ nop
|
||||
WORD_COUNT(mac_yield, 4)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_yield_load(...) \
|
||||
load_word(R_AtomJmp, R_TapePtr, 0)
|
||||
WORD_COUNT(mac_yield_load, 1)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_yield_tail(...) \
|
||||
add_ui_self(R_TapePtr, S_(MipsCode)) \
|
||||
, jump_reg( R_AtomJmp) \
|
||||
, BdSlot_ nop
|
||||
WORD_COUNT(mac_yield_tail, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_load_half_v3(tx, ty, tz, base, offset) \
|
||||
load_half(tx, base, offset + OA_(U2,[0])) \
|
||||
, load_half(ty, base, offset + OA_(U2,[1])) \
|
||||
, load_half(tz, base, offset + OA_(U2,[2]))
|
||||
WORD_COUNT(mac_load_half_v3, 3)
|
||||
|
||||
#define mac_load_v3s2(transfer, base, offset) \
|
||||
mac_load_half_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||
WORD_COUNT(mac_load_v3s2, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_load_v2s2(rs_x, rs_y, r_base, offset) \
|
||||
load_half(rs_x, r_base, offset + O_(V3_S2,x)) \
|
||||
, load_half(rs_y, r_base, offset + O_(V3_S2,y))
|
||||
WORD_COUNT(mac_load_v2s2, 2)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_store_v2s2(rt_x, rt_y, base, offset) \
|
||||
store_half(rt_x, base, offset + O_(V2_S2,x)) \
|
||||
, store_half(rt_y, base, offset + O_(V2_S2,y))
|
||||
WORD_COUNT(mac_store_v2s2, 2)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_load_word_v3(tx, ty, tz, base, offset) \
|
||||
load_word(tx, base, offset + OA_(U4,[0])) \
|
||||
, load_word(ty, base, offset + OA_(U4,[1])) \
|
||||
, load_word(tz, base, offset + OA_(U4,[2]))
|
||||
WORD_COUNT(mac_load_word_v3, 3)
|
||||
|
||||
#define mac_load_v3s4(transfer, base, offset) \
|
||||
mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||
WORD_COUNT(mac_load_v3s4, 3)
|
||||
|
||||
#define mac_load_p3s4(transfer, base, offset) \
|
||||
mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||
WORD_COUNT(mac_load_p3s4, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_store_half_v3(tx, ty, tz, base, offset) \
|
||||
store_half(tx, base, offset + OA_(U2,[0])) \
|
||||
, store_half(ty, base, offset + OA_(U2,[1])) \
|
||||
, store_half(tz, base, offset + OA_(U2,[2]))
|
||||
WORD_COUNT(mac_store_half_v3, 3)
|
||||
|
||||
#define mac_store_v3s2(transfer, base, offset) \
|
||||
mac_store_half_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||
WORD_COUNT(mac_store_v3s2, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_store_word_v3(tx, ty, tz, base, offset) \
|
||||
store_word(tx, base, offset + OA_(U4,[0])) \
|
||||
, store_word(ty, base, offset + OA_(U4,[1])) \
|
||||
, store_word(tz, base, offset + OA_(U4,[2]))
|
||||
WORD_COUNT(mac_store_word_v3, 3)
|
||||
|
||||
#define mac_store_v3s4(transfer, base, offset) \
|
||||
mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||
WORD_COUNT(mac_store_v3s4, 3)
|
||||
|
||||
#define mac_store_p3s4(transfer, base, offset) \
|
||||
mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||
WORD_COUNT(mac_store_p3s4, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_add_si_v3s4(rt_x, rt_y, rt_z, base, offset) \
|
||||
add_si(rt_x, base, O_(V3_S4,x)) \
|
||||
, add_si(rt_y, base, O_(V3_S4,y)) \
|
||||
, add_si(rt_z, base, O_(V3_S4,z))
|
||||
WORD_COUNT(mac_add_si_v3s4, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_sub_s_v3(dx, dy, dz, sx, sy, sz, tx, ty, tz) \
|
||||
sub_s(dx, sx, tx) \
|
||||
, sub_s(dy, sy, ty) \
|
||||
, sub_s(dz, sz, tz)
|
||||
WORD_COUNT(mac_sub_s_v3, 3)
|
||||
|
||||
#define mac_sub_v3s4(d, s, t) \
|
||||
mac_sub_s_v3(d.x, d.y, d.z, s.x, s.y, s.z, t.x, t.y, t.z)
|
||||
WORD_COUNT(mac_sub_v3s4, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_sub_s_v3_self(ds_x, ds_y, ds_z, tx, ty, tz) \
|
||||
sub_s(ds_x, ds_x, tx) \
|
||||
, sub_s(ds_y, ds_y, ty) \
|
||||
, sub_s(ds_z, ds_z, tz)
|
||||
WORD_COUNT(mac_sub_s_v3_self, 3)
|
||||
|
||||
#define mac_sub_v3s4_self(ds, t) \
|
||||
mac_sub_s_v3_self(ds.x, ds.y, ds.z, t.x, t.y, t.z)
|
||||
WORD_COUNT(mac_sub_v3s4_self, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_store_rects2(rt_x, rt_y, rt_width, rt_height, base, offset) \
|
||||
store_half(rt_x, base, offset + O_(Rect_S2,x)) \
|
||||
, store_half(rt_y, base, offset + O_(Rect_S2,y)) \
|
||||
, store_half(rt_width, base, offset + O_(Rect_S2,width)) \
|
||||
, store_half(rt_height, base, offset + O_(Rect_S2,height))
|
||||
WORD_COUNT(mac_store_rects2, 4)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_load_word_imm(dst, imm) \
|
||||
load_upper_i(dst, u4_hi(imm)) \
|
||||
, or_i_self( dst, u4_lo(imm))
|
||||
WORD_COUNT(mac_load_word_imm, 2)
|
||||
|
||||
#define mac_shift_aright_v3_self(dt_x, dt_y, dt_z, shift_amount) \
|
||||
shift_aright(dt_x, dt_x, shift_amount) \
|
||||
, shift_aright(dt_y, dt_y, shift_amount) \
|
||||
, shift_aright(dt_z, dt_z, shift_amount)
|
||||
WORD_COUNT(mac_shift_aright_v3_self, 3)
|
||||
|
||||
#define mac_shift_aright_v3s4_self(dt, shift) \
|
||||
mac_shift_aright_v3_self(dt.x, dt.y, dt.z, shift)
|
||||
WORD_COUNT(mac_shift_aright_v3s4_self, 3)
|
||||
|
||||
#define mac_shift_aright_var_v3(rd_v0, rd_v1, rd_v2, rs_v0, rs_v1, rs_v2, r_shift) \
|
||||
shift_aright_var(rd_v0, rs_v0, r_shift) \
|
||||
, shift_aright_var(rd_v1, rs_v1, r_shift) \
|
||||
, shift_aright_var(rd_v2, rs_v2, r_shift)
|
||||
WORD_COUNT(mac_shift_aright_var_v3, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_shift_aright_var_v3_self(rds_v0, rds_v1, rds_v2, r_shift) \
|
||||
shift_aright_var(rds_v0, rds_v0, r_shift) \
|
||||
, shift_aright_var(rds_v1, rds_v1, r_shift) \
|
||||
, shift_aright_var(rds_v2, rds_v2, r_shift)
|
||||
WORD_COUNT(mac_shift_aright_var_v3_self, 3)
|
||||
|
||||
#define mac_shift_aright_var_v3s4_self(ds, shift) \
|
||||
mac_shift_aright_var_v3_self(ds.x, ds.y, ds.z, shift)
|
||||
WORD_COUNT(mac_shift_aright_var_v3s4_self, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_load_tri_indices(r_face_cusor, r_i0, r_i1, r_i2) \
|
||||
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)) \
|
||||
, load_half_u(r_i1, r_face_cusor, 1 * S_(S2)) \
|
||||
, load_half_u(r_i2, r_face_cusor, 2 * S_(S2))
|
||||
WORD_COUNT(mac_load_tri_indices, 3)
|
||||
|
||||
#define mac_gte_mv_to_cr_diag_v3s4(v) \
|
||||
gte_mv_to_ctrl_r(v.y, gte_cr_RT13) \
|
||||
, gte_mv_to_ctrl_r(v.z, gte_cr_RT22) \
|
||||
, gte_mv_to_ctrl_r(v.x, gte_cr_RT11)
|
||||
WORD_COUNT(mac_gte_mv_to_cr_diag_v3s4, 3)
|
||||
|
||||
#define mac_gte_ld_ir123_v3s4(v) \
|
||||
gte_mv_to_data_r(v.x, C2_IR1) \
|
||||
, gte_mv_to_data_r(v.y, C2_IR2) \
|
||||
, gte_mv_to_data_r(v.z, C2_IR3)
|
||||
WORD_COUNT(mac_gte_ld_ir123_v3s4, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_gte_op_cross_v3s4(a, b) \
|
||||
mac_gte_mv_to_cr_diag_v3s4(a) \
|
||||
GteDelay_ /* RT diagonal: D1 = a.x, D2 = a.y, D3 = a.z */ \
|
||||
, mac_gte_ld_ir123_v3s4(b) \
|
||||
GteDelay_ /* IR: second operand (b.xyz) */ \
|
||||
, gte_cmdw_cross /* OP: MAC1/2/3 = a × b (S12.20) */ \
|
||||
, mac_gte_mv_from_mac123_v3s4(a) \
|
||||
GteDelay_ /* Read MAC1/2/3 → a.xyz (overwrites source-A's load targets) */ \
|
||||
, mac_shift_aright_v3s4_self(a, 12) /* Right-shift MAC by 12 (S12.20 → S12.0 OuterProduct12) */
|
||||
WORD_COUNT(mac_gte_op_cross_v3s4, 13)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_gte_store_f3(r_primitive_cursor) \
|
||||
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)) \
|
||||
, gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_F3,p1)) \
|
||||
, gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_F3,p2))
|
||||
WORD_COUNT(mac_gte_store_f3, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_gte_load_tri_verts(vbase, v0, v1, v2) \
|
||||
shift_lleft(R_AT, v0, v3s2_byteoff) \
|
||||
, add_u_self(R_AT, vbase) \
|
||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||
, LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY0) \
|
||||
, gte_mv_to_data_r(R_V1, C2_VZ0) \
|
||||
, shift_lleft(R_AT, v1, v3s2_byteoff) \
|
||||
, add_u_self(R_AT, vbase) \
|
||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||
, LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY1) \
|
||||
, gte_mv_to_data_r(R_V1, C2_VZ1) \
|
||||
, shift_lleft(R_AT, v2, v3s2_byteoff) \
|
||||
, add_u_self(R_AT, vbase) \
|
||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||
, LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY2) \
|
||||
, gte_mv_to_data_r(R_V1, C2_VZ2)
|
||||
WORD_COUNT(mac_gte_load_tri_verts, 18)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_gte_store_g4_p012(r_primitive_cursor) \
|
||||
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)) \
|
||||
, gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)) \
|
||||
, gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2))
|
||||
WORD_COUNT(mac_gte_store_g4_p012, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_gte_store_g4_p3(r_primitive_cursor) \
|
||||
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3))
|
||||
WORD_COUNT(mac_gte_store_g4_p3, 1)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_gte_sqr_v3(r_sx, r_sy, r_sz, r_sq_x, r_sq_y, r_sq_z) \
|
||||
mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop) \
|
||||
, gte_mv_from_data_r(r_sq_x, C2_MAC1) \
|
||||
, gte_mv_from_data_r(r_sq_y, C2_MAC2) \
|
||||
, gte_mv_from_data_r(r_sq_z, C2_MAC3)
|
||||
WORD_COUNT(mac_gte_sqr_v3, 8)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, delay_slot) \
|
||||
gte_mv_to_data_r(r_sx, C2_IR1) \
|
||||
, gte_mv_to_data_r(r_sy, C2_IR2) \
|
||||
, gte_mv_to_data_r(r_sz, C2_IR3) \
|
||||
, delay_slot \
|
||||
, gte_cmdw_sqr
|
||||
WORD_COUNT(mac_gte_sqr_v3s4, 5)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_gte_gpf_scale(r_sx, r_sy, r_sz, r_recip_est, r_shift, r_dx, r_dy, r_dz) \
|
||||
gte_mv_to_data_r(r_recip_est, C2_IR0) \
|
||||
, gte_mv_to_data_r(r_sx, C2_IR1) \
|
||||
, gte_mv_to_data_r(r_sy, C2_IR2) \
|
||||
, gte_mv_to_data_r(r_sz, C2_IR3) \
|
||||
, GteDelay_ nop2 /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */ \
|
||||
, gte_cmdw_gpf \
|
||||
, gte_mv_from_data_r(r_dx, C2_MAC1) \
|
||||
, gte_mv_from_data_r(r_dy, C2_MAC2) \
|
||||
, gte_mv_from_data_r(r_dz, C2_MAC3) \
|
||||
, shift_aright_var(r_dx, r_dx, r_shift) \
|
||||
, shift_aright_var(r_dy, r_dy, r_shift) \
|
||||
, shift_aright_var(r_dz, r_dz, r_shift)
|
||||
WORD_COUNT(mac_gte_gpf_scale, 13)
|
||||
|
||||
#define mac_trans_mt3s3s4(r_mtx, r_off, r_t0, r_t1, r_t2) \
|
||||
load_word( r_t0, r_off, O_(V3_S4,x)) \
|
||||
, load_word( r_t1, r_off, O_(V3_S4,y)) \
|
||||
, load_word( r_t2, r_off, O_(V3_S4,z)) \
|
||||
, store_word(r_t0, r_mtx, O_(MT3_S2S4,t[0])) \
|
||||
, store_word(r_t1, r_mtx, O_(MT3_S2S4,t[1])) \
|
||||
, store_word(r_t2, r_mtx, O_(MT3_S2S4,t[2]))
|
||||
WORD_COUNT(mac_trans_mt3s3s4, 6)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_lzcr_round_even_half_shift(r_shift, r_mag_sq, r_mag_sq_copy) \
|
||||
and_i(r_shift, r_shift, gte_lzcr_even_mask) \
|
||||
, or_u(r_mag_sq_copy, r_mag_sq, 0) \
|
||||
, li_s( r_mag_sq, 31) \
|
||||
, sub_s( r_mag_sq, r_mag_sq, r_shift) \
|
||||
, shift_aright(r_mag_sq, r_mag_sq, 1)
|
||||
WORD_COUNT(mac_lzcr_round_even_half_shift, 5)
|
||||
|
||||
#define mac_gte_general_purpose_interopolation(to_ir0, to_ir1, to_ir2, to_ir3, fr_mac1, fr_mac2, fr_mac3, nop_slot1, nop_slot2) \
|
||||
gte_mv_to_data_r(to_ir0, C2_IR0) \
|
||||
, gte_mv_to_data_r(to_ir1, C2_IR1) /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */ \
|
||||
, gte_mv_to_data_r(to_ir2, C2_IR2) \
|
||||
, gte_mv_to_data_r(to_ir3, C2_IR3) /* IR3 = src.z (reloaded) */ \
|
||||
, GteDelay_ nop_slot1 \
|
||||
, GteDelay_ nop_slot2 \
|
||||
, gte_cmdw_gpf \
|
||||
, gte_mv_from_data_r(fr_mac1, C2_MAC1) \
|
||||
, gte_mv_from_data_r(fr_mac2, C2_MAC2) \
|
||||
, gte_mv_from_data_r(fr_mac3, C2_MAC3)
|
||||
WORD_COUNT(mac_gte_general_purpose_interopolation, 10)
|
||||
|
||||
#define mac_gte_mv_from_data_r_mac123(fr_mac1, fr_mac2, fr_mac3) \
|
||||
gte_mv_from_data_r(fr_mac1, C2_MAC1) \
|
||||
, gte_mv_from_data_r(fr_mac2, C2_MAC2) \
|
||||
, gte_mv_from_data_r(fr_mac3, C2_MAC3)
|
||||
WORD_COUNT(mac_gte_mv_from_data_r_mac123, 3)
|
||||
|
||||
#define mac_gte_mv_from_mac123_v3s4(v) \
|
||||
mac_gte_mv_from_data_r_mac123(v.x, v.y, v.z)
|
||||
WORD_COUNT(mac_gte_mv_from_mac123_v3s4, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \
|
||||
mac_load_word_imm(reg_transfer, cmd) \
|
||||
, store_word( reg_transfer, reg_base, port)
|
||||
WORD_COUNT(mac_gcmd_push, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_store_rgb8(rr, rg, rb, base, offset) \
|
||||
store_byte(rr, base, offset + O_(RGB8,r)) \
|
||||
, store_byte(rg, base, offset + O_(RGB8,g)) \
|
||||
, store_byte(rb, base, offset + O_(RGB8,b))
|
||||
WORD_COUNT(mac_store_rgb8, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_pack_color_word(r_base, off, cmd, r, g, b) \
|
||||
load_upper_i(R_AT, (cmd) << 8 | (b)) \
|
||||
, or_i_self( R_AT, ((g) << 8) | (r)) \
|
||||
, store_word( R_AT, r_base, (off))
|
||||
WORD_COUNT(mac_pack_color_word, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_format_f3_color(r_base, r, g, b) \
|
||||
mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b)
|
||||
WORD_COUNT(mac_format_f3_color, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_format_g4_color(r_prim_cursor, r0, g0, b0, r1, g1, b1, r2, g2, b2, r3, g3, b3) \
|
||||
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0) \
|
||||
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1) \
|
||||
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c2), 0, r2,g2,b2) \
|
||||
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3)
|
||||
WORD_COUNT(mac_format_g4_color, 12)
|
||||
|
||||
#define mac_insert_ot_tag(r_ot_base, r_prim_cursor, poly_size) \
|
||||
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
|
||||
, add_u_self( R_T1, r_ot_base) /* T1 = & OrderingTable[OTZ] */ \
|
||||
, load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \
|
||||
, load_upper_i(R_V0, (poly_size/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) \
|
||||
, mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
|
||||
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
|
||||
, store_word( R_AT, r_prim_cursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \
|
||||
, shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
|
||||
, shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \
|
||||
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
|
||||
WORD_COUNT(mac_insert_ot_tag, 11)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_pad_set_centered_axes(state, scratch) \
|
||||
load_upper_i(scratch, (PadAxis_Centered >> 16) & 0xFFFF) \
|
||||
, or_i_self( scratch, PadAxis_Centered & 0xFFFF) /* mac_load_word_imm(scratch, PadAxis_Centered), */ \
|
||||
, store_word( scratch, state, O_(PadState,axes))
|
||||
WORD_COUNT(mac_pad_set_centered_axes, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_pad_set_id_byte(state, r_id, id_value) \
|
||||
add_ui( r_id, R_0, id_value) \
|
||||
, store_byte(r_id, state, O_(PadState,id))
|
||||
WORD_COUNT(mac_pad_set_id_byte, 2)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_pad_set_status(r_tmp, r_state, pad_status) \
|
||||
add_ui( r_tmp, R_0, pad_status) \
|
||||
, store_word(r_tmp, r_state, O_(PadState,status))
|
||||
WORD_COUNT(mac_pad_set_status, 2)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_pad_store_inverted_buttons(r_buttons, r_pad_state) \
|
||||
nor_u( r_buttons, r_buttons, R_0) \
|
||||
, store_half(r_buttons, r_pad_state, O_(PadState,buttons))
|
||||
WORD_COUNT(mac_pad_store_inverted_buttons, 2)
|
||||
|
||||
@@ -0,0 +1,73 @@
|
||||
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
|
||||
// Directory: C:\projects\Pikuma\ps1\code\duffle\
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\word_count.metadata.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\memory.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\math.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\gcc_asm.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\mips.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\gp.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\gte.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.atom.c
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.atom.c
|
||||
#pragma once
|
||||
|
||||
#pragma region duffle
|
||||
|
||||
|
||||
// --- atom: example_atom_proc (10 words) ---
|
||||
|
||||
#define _atom_offset_example_atom_proc_skip 2
|
||||
|
||||
enum {
|
||||
atom_offset_example_atom_proc_skip = _atom_offset_example_atom_proc_skip,
|
||||
};
|
||||
|
||||
// --- atom: normalize_v3s4 (62 words) ---
|
||||
|
||||
#define _atom_offset_aligned_done_srav_path 3
|
||||
#define _atom_offset_srav_path_aligned_done 4
|
||||
|
||||
enum {
|
||||
atom_offset_aligned_done_srav_path = _atom_offset_aligned_done_srav_path,
|
||||
atom_offset_srav_path_aligned_done = _atom_offset_srav_path_aligned_done,
|
||||
};
|
||||
|
||||
// --- atom: pad_bios_snapshot (84 words) ---
|
||||
|
||||
#define _atom_offset_snap_root_skip_disconnected 10
|
||||
#define _atom_offset_disconnected_snap_end 65
|
||||
#define _atom_offset_case_2_id_dispatch 9
|
||||
#define _atom_offset_pending_snap_end 54
|
||||
#define _atom_offset_id_dispatch_try_analog_stick 12
|
||||
#define _atom_offset_id_dispatch_snap_end 40
|
||||
#define _atom_offset_try_analog_stick_try_analog_pad 13
|
||||
#define _atom_offset_analog_stick_snap_end 25
|
||||
#define _atom_offset_try_analog_pad_try_unsupported 12
|
||||
#define _atom_offset_analog_pad_snap_end 10
|
||||
|
||||
enum {
|
||||
atom_offset_snap_root_skip_disconnected = _atom_offset_snap_root_skip_disconnected,
|
||||
atom_offset_disconnected_snap_end = _atom_offset_disconnected_snap_end,
|
||||
atom_offset_case_2_id_dispatch = _atom_offset_case_2_id_dispatch,
|
||||
atom_offset_pending_snap_end = _atom_offset_pending_snap_end,
|
||||
atom_offset_id_dispatch_try_analog_stick = _atom_offset_id_dispatch_try_analog_stick,
|
||||
atom_offset_id_dispatch_snap_end = _atom_offset_id_dispatch_snap_end,
|
||||
atom_offset_try_analog_stick_try_analog_pad = _atom_offset_try_analog_stick_try_analog_pad,
|
||||
atom_offset_analog_stick_snap_end = _atom_offset_analog_stick_snap_end,
|
||||
atom_offset_try_analog_pad_try_unsupported = _atom_offset_try_analog_pad_try_unsupported,
|
||||
atom_offset_analog_pad_snap_end = _atom_offset_analog_pad_snap_end,
|
||||
};
|
||||
|
||||
#pragma endregion duffle
|
||||
|
||||
@@ -0,0 +1,61 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# include "dsl.h"
|
||||
# include "gp.h"
|
||||
# include "lottes_tape.h"
|
||||
#endif
|
||||
|
||||
ATOM_FILE_DEBUGGER_LINE_MARKER(gp_atom_c);
|
||||
|
||||
#pragma region MACs (Mips Atom Components)
|
||||
|
||||
FI_ Slice_MipsCode ac_gcmd_push(AtomBuilder_R ab, U4 cmd, U4 reg_transfer, U4 reg_base, U2 port)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
mac_load_word_imm(reg_transfer, cmd),
|
||||
store_word( reg_transfer, reg_base, port),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_store_rgb8(AtomBuilder_R ab, U1 rr, U1 rg, U1 rb, U4 base, U4 offset)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
store_byte(rr, base, offset + O_(RGB8,r)),
|
||||
store_byte(rg, base, offset + O_(RGB8,g)),
|
||||
store_byte(rb, base, offset + O_(RGB8,b)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_pack_color_word(AtomBuilder_R ab, U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
load_upper_i(R_AT, (cmd) << 8 | (b)),
|
||||
or_i_self( R_AT, ((g) << 8) | (r)),
|
||||
store_word( R_AT, r_base, (off)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_format_f3_color(AtomBuilder_R ab, U4 r_base, U1 r, U1 g, U1 b)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ab, { mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
|
||||
|
||||
FI_ Slice_MipsCode ac_format_g4_color(AtomBuilder_R ab, U4 r_prim_cursor,
|
||||
U1 r0, U1 g0, U1 b0,
|
||||
U1 r1, U1 g1, U1 b1,
|
||||
U1 r2, U1 g2, U1 b2,
|
||||
U1 r3, U1 g3, U1 b3)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0),
|
||||
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1),
|
||||
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c2), 0, r2,g2,b2),
|
||||
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3),
|
||||
})
|
||||
|
||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list. */
|
||||
// TODO(Ed): Expose R_T1 as a r_t0, r_V0 as r_t2
|
||||
I_ Slice_MipsCode ac_insert_ot_tag(AtomBuilder_R ab, Reg r_ot_base, Reg r_prim_cursor, U2 poly_size) MipsAtomComp_Proc_(ab, {
|
||||
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
|
||||
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
|
||||
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
|
||||
load_upper_i(R_V0, (poly_size/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits),
|
||||
mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
|
||||
or_u( R_AT, R_AT, R_V0), // Merge length
|
||||
store_word( R_AT, r_prim_cursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr)
|
||||
shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)), // AT = (prim_length << 24) | old_addr
|
||||
shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
|
||||
store_word( R_AT, R_T1, O_(PolyTag,code)), // OrderingTable[OTZ] = PrimCursor
|
||||
})
|
||||
|
||||
#pragma endregion MACs (Mips Atom Components)
|
||||
+365
-366
@@ -1,7 +1,6 @@
|
||||
/* ============================================================================
|
||||
* duffle DSL Suffix Conventions
|
||||
* ============================================================================
|
||||
*
|
||||
* Every mnemonic in this header follows the same suffix grammar:
|
||||
*
|
||||
* Primitive commands: gp0_cmd_poly_f3 = 0x20 (byte opcode)
|
||||
@@ -22,12 +21,11 @@
|
||||
* 4. Semantic encoders gp0_word_poly_f3(r,g,b)
|
||||
* 3. Composite encoders enc_color_word(cmd, r, g, b)
|
||||
* 2. Per-field encoders enc_gp0_color_r(r), enc_gp0_color_g(g), ...
|
||||
* 1. Bitfield layout consts gp0_color_red_shift = 0, gp0_color_red_mask = 0xFF
|
||||
* 1. Bitfield layout consts gp0_color_red_pos = 0, gp0_color_red_width = 8
|
||||
* 0. Opcode IDs gp0_cmd_poly_f3 = 0x20
|
||||
*
|
||||
* Vendor mnemonics (gte_mtc2, gte_mfc2, etc.) are NOT in this header.
|
||||
* They live in the opt-in `gp_vendor_sym.h` for users who prefer the
|
||||
* PSYQ-style names.
|
||||
* They live in the opt-in `gp_vendor_sym.h` for users who prefer the PSYQ-style names.
|
||||
* ============================================================================ */
|
||||
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
@@ -41,33 +39,28 @@
|
||||
/* ============================================================================
|
||||
* Hardware MMIO Addresses
|
||||
* ============================================================================
|
||||
*
|
||||
* PSX GPU has two 32-bit ports in the I/O register region at KSEG2
|
||||
* 0x1F800000+. GP0 (offset 0x10) is the data port (commands + params).
|
||||
* PSX GPU has two 32-bit ports in the I/O register region at KSEG2 0x1F800000+.
|
||||
* GP0 (offset 0x10) is the data port (commands + params).
|
||||
* GP1 (offset 0x14) is the control port (status, ctrl writes).
|
||||
* ============================================================================ */
|
||||
/* IO base address (KSEG2 0x1F800000+ for the I/O register region).
|
||||
* The 16-bit upper half `IO_BASE_ADDR_HI16` is the form used by
|
||||
* tape-side macros that pin a register to hold the IO base and access
|
||||
* ports via offsets — `lui $reg, 0x1F80` (1 word) then `sw $data,
|
||||
* GPIO_PORT*_OFFSET($reg)` (1 word). Mirrors the `IO_BASE_ADDR equ
|
||||
* 0x1F80` + `gpio_port0 equ 0x1810` pattern from graphics_hello/gp.s.
|
||||
*
|
||||
* See lottes_tape.h `R_GpIoBase` + `mac_gp0_send_imm` for the
|
||||
* wave-context form that composes these primitives. */
|
||||
* The 16-bit upper half `IO_BASE_ADDR_HI16` is the form used by tape-side macros that pin a register
|
||||
* to hold the IO base and access ports via offsets:
|
||||
* `lui $reg, 0x1F80` (1 word) then `sw $data, GPIO_PORT*_OFFSET($reg)` (1 word).
|
||||
* Mirrors the `IO_BASE_ADDR equ 0x1F80` + `gpio_port0 equ 0x1810` pattern from graphics_hello/gp.s. */
|
||||
enum {
|
||||
IO_BASE_ADDR = 0x1F800000, /* full 32-bit I/O region base */
|
||||
IO_BASE_ADDR_HI16 = 0x1F80, /* fits in a single `lui $reg, 0x1F80` */
|
||||
IO_BASE_ADDR = 0x1F800000, /* full 32-bit I/O region base */
|
||||
IO_BASE_ADDR_HI16 = 0x1F80, /* fits in a single `lui $reg, 0x1F80` */
|
||||
|
||||
/* Offsets from IO_BASE_ADDR to each port. Used by tape-side macros
|
||||
* that pin a register to IO_BASE_ADDR and access ports via offsets:
|
||||
* sw $data, GPIO_PORT0_OFFSET($io_base) ; write GP0
|
||||
* sw $data, GPIO_PORT1_OFFSET($io_base) ; write GP1 */
|
||||
GPIO_PORT0_OFFSET = 0x1810,
|
||||
GPIO_PORT1_OFFSET = 0x1814,
|
||||
/* Offsets from IO_BASE_ADDR to each port. Used by tape-side macros
|
||||
* that pin a register to IO_BASE_ADDR and access ports via offsets:
|
||||
* sw $data, GPIO_PORT0_OFFSET($io_base) ; write GP0
|
||||
* sw $data, GPIO_PORT1_OFFSET($io_base) ; write GP1 */
|
||||
GPIO_PORT0_OFFSET = 0x1810,
|
||||
GPIO_PORT1_OFFSET = 0x1814,
|
||||
|
||||
HW_GP0_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT0_OFFSET,
|
||||
HW_GP1_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT1_OFFSET,
|
||||
HW_GP0_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT0_OFFSET,
|
||||
HW_GP1_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT1_OFFSET,
|
||||
};
|
||||
|
||||
#define HW_GP0 C_(U4 V_*, HW_GP0_ADDR)
|
||||
@@ -75,97 +68,87 @@ enum {
|
||||
|
||||
#define gp0_send(word) (HW_GP0[0] = (word))
|
||||
#define gp1_send(word) (HW_GP1[0] = (word))
|
||||
#define DmaSlot_ // Annotate an instruction as filling a CPU <-> Command DMA delay slot/s
|
||||
|
||||
/* ============================================================================
|
||||
* GP0 command byte constants + Layer 1 (GPU bitfield shifts)
|
||||
* ============================================================================
|
||||
*
|
||||
* 8-bit GP0 opcodes (the upper byte of a primitive's first word).
|
||||
* These are the BYTE only; pre-baked 32-bit words are in §10.4.
|
||||
* The layer-1 bitfield-layout constants live in the same enum block
|
||||
* so the encoder in §10.4 can reference them by name. NO macro body
|
||||
* past this point uses a raw shift or raw mask — every shift/width/mask
|
||||
* is named here, named once. Mirrors the OPCODE_SHIFT / RS_SHIFT /
|
||||
* REG_MASK convention from mips.h.
|
||||
* 8-bit GP0 opcodes (the upper byte of a primitive's first word). These are the BYTE only.
|
||||
* NO macro body past this point uses a raw shift or raw mask.
|
||||
* Mirrors the OPCODE_POS / RS_POS convention from mips.h.
|
||||
* ============================================================================ */
|
||||
enum {
|
||||
gp0_cmd_Nop = 0x00,
|
||||
gp0_cmd_Nop = 0x00,
|
||||
|
||||
/* Cache management */
|
||||
gp0_cmd_ClearCache = 0x01,
|
||||
gp0_cmd_FillVram = 0x02,
|
||||
gp0_cmd_CopyVram = 0x80,
|
||||
gp0_cmd_CopyVramChained = 0x81,
|
||||
gp0_cmd_ReadVram = 0xC0,
|
||||
/* Cache management */
|
||||
gp0_cmd_ClearCache = 0x01,
|
||||
gp0_cmd_FillVram = 0x02,
|
||||
gp0_cmd_CopyVram = 0x80,
|
||||
gp0_cmd_CopyVramChained = 0x81,
|
||||
gp0_cmd_ReadVram = 0xC0,
|
||||
|
||||
/* Polygons */
|
||||
gp0_cmd_poly_f3 = 0x20, /* Flat Triangle */
|
||||
gp0_cmd_poly_ft3 = 0x24, /* Flat Textured Triangle */
|
||||
gp0_cmd_poly_g3 = 0x30, /* Gouraud Triangle */
|
||||
gp0_cmd_poly_gt3 = 0x34, /* Gouraud Textured Tri */
|
||||
gp0_cmd_poly_f4 = 0x28, /* Flat Quad */
|
||||
gp0_cmd_poly_ft4 = 0x2C, /* Flat Textured Quad */
|
||||
gp0_cmd_poly_g4 = 0x38, /* Gouraud Quad */
|
||||
gp0_cmd_poly_gt4 = 0x3C, /* Gouraud Textured Quad */
|
||||
/* Polygons */
|
||||
gp0_cmd_poly_f3 = 0x20, /* Flat Triangle */
|
||||
gp0_cmd_poly_ft3 = 0x24, /* Flat Textured Triangle */
|
||||
gp0_cmd_poly_g3 = 0x30, /* Gouraud Triangle */
|
||||
gp0_cmd_poly_gt3 = 0x34, /* Gouraud Textured Tri */
|
||||
gp0_cmd_poly_f4 = 0x28, /* Flat Quad */
|
||||
gp0_cmd_poly_ft4 = 0x2C, /* Flat Textured Quad */
|
||||
gp0_cmd_poly_g4 = 0x38, /* Gouraud Quad */
|
||||
gp0_cmd_poly_gt4 = 0x3C, /* Gouraud Textured Quad */
|
||||
|
||||
/* Lines */
|
||||
gp0_cmd_line_f2 = 0x40,
|
||||
gp0_cmd_line_g2 = 0x50,
|
||||
/* Lines */
|
||||
gp0_cmd_line_f2 = 0x40,
|
||||
gp0_cmd_line_g2 = 0x50,
|
||||
|
||||
/* Sprites + Tiles + Rects */
|
||||
gp0_cmd_sprt_1 = 0x64,
|
||||
gp0_cmd_sprt_8 = 0x74,
|
||||
gp0_cmd_sprt_16 = 0x7C,
|
||||
gp0_cmd_tile_1 = 0x60,
|
||||
gp0_cmd_tile_8 = 0x68,
|
||||
gp0_cmd_tile_16 = 0x70,
|
||||
/* Sprites + Tiles + Rects */
|
||||
gp0_cmd_sprt_1 = 0x64,
|
||||
gp0_cmd_sprt_8 = 0x74,
|
||||
gp0_cmd_sprt_16 = 0x7C,
|
||||
gp0_cmd_tile_1 = 0x60,
|
||||
gp0_cmd_tile_8 = 0x68,
|
||||
gp0_cmd_tile_16 = 0x70,
|
||||
|
||||
/* State setters (not drawing primitives; set render context).
|
||||
* Per PSX-SPX graphicsprocessingunitgpu.md §"GP0 Other Commands". */
|
||||
gp0_cmd_DrawModeSetting = 0xE1, /* TPage / draw-mode (semi-trans, dither, etc.) */
|
||||
gp0_cmd_SetTextureWindow = 0xE2,
|
||||
gp0_cmd_SetDrawArea_TopLeft = 0xE3,
|
||||
gp0_cmd_SetDrawArea_BotRight = 0xE4,
|
||||
gp0_cmd_SetDrawOffset = 0xE5,
|
||||
gp0_cmd_SetMaskBit = 0xE6,
|
||||
/* State setters (not drawing primitives; set render context). */
|
||||
gp0_cmd_DrawModeSetting = 0xE1, /* TPage / draw-mode (semi-trans, dither, etc.) */
|
||||
gp0_cmd_SetTextureWindow = 0xE2,
|
||||
gp0_cmd_SetDrawArea_TopLeft = 0xE3,
|
||||
gp0_cmd_SetDrawArea_BotRight = 0xE4,
|
||||
gp0_cmd_SetDrawOffset = 0xE5,
|
||||
gp0_cmd_SetMaskBit = 0xE6,
|
||||
|
||||
/* bitfield shifts / widths / masks ----
|
||||
*
|
||||
* Generic GP0/GP1 command byte (upper 8 bits of every word sent
|
||||
* to either port). Used by `enc_gp0_cmd(cmd)` and friends below. */
|
||||
gp0_cmd_shift = 24,
|
||||
gp0_cmd_width = 8,
|
||||
gp0_cmd_mask = 0xFF,
|
||||
/* bitfield offset pos / widths ----
|
||||
* Generic GP0/GP1 command byte (upper 8 bits of every word sent to either port). */
|
||||
gp0_cmd_pos = 24,
|
||||
gp0_cmd_width = 8,
|
||||
|
||||
/* Color word layout (lives in Poly_F3.color, Poly_G4.c0..c3, etc.):
|
||||
* bits 31..24 = command byte
|
||||
* bits 23..16 = BLUE
|
||||
* bits 15..08 = GREEN
|
||||
* bits 07..00 = RED (PSX GPU is BGR, NOT RGB) */
|
||||
gp0_color_cmd_shift = 24, gp0_color_cmd_width = 8, gp0_color_cmd_mask = 0xFF,
|
||||
gp0_color_blue_shift = 16, gp0_color_blue_width = 8, gp0_color_blue_mask = 0xFF,
|
||||
gp0_color_green_shift = 8, gp0_color_green_width = 8, gp0_color_green_mask = 0xFF,
|
||||
gp0_color_red_shift = 0, gp0_color_red_width = 8, gp0_color_red_mask = 0xFF,
|
||||
/* Color word layout (lives in Poly_F3.color, Poly_G4.c0..c3, etc.):
|
||||
* bits 31..24 = command byte
|
||||
* bits 23..16 = BLUE
|
||||
* bits 15..08 = GREEN
|
||||
* bits 07..00 = RED (PSX GPU is BGR, NOT RGB) */
|
||||
gp0_color_cmd_pos = 24, gp0_color_cmd_width = 8,
|
||||
gp0_color_blue_pos = 16, gp0_color_blue_width = 8,
|
||||
gp0_color_green_pos = 8, gp0_color_green_width = 8,
|
||||
gp0_color_red_pos = 0, gp0_color_red_width = 8,
|
||||
};
|
||||
|
||||
/* ============================================================================
|
||||
* Layer 1.5 (per-field encoders) + Layer 2 (composite) + Layer 3 (semantic GP0 word builders)
|
||||
* ============================================================================
|
||||
*
|
||||
* Layer 1.5 encoders take one field's value, mask it to its own width,
|
||||
* and shift it to its own position. Mirrors `enc_op` / `enc_rs` /
|
||||
* `enc_rt` in mips.h and `enc_gte_sf` / `enc_gte_mx` in gte.h. Layer-2 composite encoders
|
||||
* OR the per-field encoders together; layer-3 semantic macros delegate to the composites.
|
||||
* Layer 1.5 encoders take one field's value, mask it to its own width, and shift it to its own position.
|
||||
* Mirrors `enc_op` / `enc_rs` / `enc_rt` in mips.h and `enc_gte_sf` / `enc_gte_mx` in gte.h.
|
||||
* Layer-2 composite encoders OR the per-field encoders together; layer-3 semantic macros delegate to the composites.
|
||||
* No raw shifts or magic numbers in any macro body below this point.
|
||||
* ============================================================================ */
|
||||
|
||||
/* ---- Layer 1.5: per-field encoders ---- */
|
||||
#define enc_gp0_cmd(cmd) (((cmd) & gp0_cmd_mask) << gp0_cmd_shift)
|
||||
#define enc_gp0_cmd(cmd) ((cmd) << gp0_cmd_pos)
|
||||
|
||||
#define enc_gp0_color_cmd(cmd) (((cmd) & gp0_color_cmd_mask) << gp0_color_cmd_shift)
|
||||
#define enc_gp0_color_r(r) (((r) & gp0_color_red_mask) << gp0_color_red_shift)
|
||||
#define enc_gp0_color_g(g) (((g) & gp0_color_green_mask) << gp0_color_green_shift)
|
||||
#define enc_gp0_color_b(b) (((b) & gp0_color_blue_mask) << gp0_color_blue_shift)
|
||||
#define enc_gp0_color_cmd(cmd) ((cmd) << gp0_color_cmd_pos)
|
||||
#define enc_gp0_color_r(r) ((r) << gp0_color_red_pos)
|
||||
#define enc_gp0_color_g(g) ((g) << gp0_color_green_pos)
|
||||
#define enc_gp0_color_b(b) ((b) << gp0_color_blue_pos)
|
||||
|
||||
/* ---- Layer 2: composite encoders ---- */
|
||||
#define enc_color_word(cmd, r, g, b) (enc_gp0_color_cmd(cmd) | enc_gp0_color_r(r) | enc_gp0_color_g(g) | enc_gp0_color_b(b))
|
||||
@@ -186,80 +169,80 @@ enum {
|
||||
#define gp0_word_poly_gt4(r,g,b) enc_color_word(gp0_cmd_poly_gt4, (r),(g),(b))
|
||||
|
||||
/* Cache management — bare-cmd words (no color/range payload). */
|
||||
#define gp0_word_clear_cache() enc_gp0_cmd_word(gp0_cmd_ClearCache)
|
||||
#define gp0_word_fill_vram() enc_gp0_cmd_word(gp0_cmd_FillVram)
|
||||
#define gp0_word_copy_vram() enc_gp0_cmd_word(gp0_cmd_CopyVram)
|
||||
#define gp0_word_read_vram() enc_gp0_cmd_word(gp0_cmd_ReadVram)
|
||||
#define gp0_word_clear_cache() enc_gp0_cmd_word(gp0_cmd_ClearCache)
|
||||
#define gp0_word_fill_vram() enc_gp0_cmd_word(gp0_cmd_FillVram)
|
||||
#define gp0_word_copy_vram() enc_gp0_cmd_word(gp0_cmd_CopyVram)
|
||||
#define gp0_word_read_vram() enc_gp0_cmd_word(gp0_cmd_ReadVram)
|
||||
|
||||
/* NOP — bare-cmd word (no effect; used as DR_ENV padding). */
|
||||
#define gp0_word_nop() enc_gp0_cmd_word(gp0_cmd_Nop)
|
||||
|
||||
/* ============================================================================
|
||||
* GP1 command byte constants + Layer 1 (display-mode + range + draw-area bitfield shifts)
|
||||
* ============================================================================
|
||||
*
|
||||
* GP1 status bits are read from HW_GP1; ctrl writes use GP1 commands
|
||||
* packed into 32-bit words (cmd byte in the upper 8 bits via
|
||||
* `enc_gp0_cmd(cmd)` — never a raw shift).
|
||||
* GP1 status bits are read from HW_GP1;
|
||||
* ctrl writes use GP1 commands packed into 32-bit words
|
||||
* (cmd byte in the upper 8 bits via `enc_gp0_cmd(cmd)`).
|
||||
* ============================================================================ */
|
||||
enum {
|
||||
gp1_cmd_Reset = 0x00,
|
||||
gp1_cmd_ResetCmdBuffer = 0x01,
|
||||
gp1_cmd_AcknowledgeIRQ = 0x02,
|
||||
gp1_cmd_DisplayEnable = 0x03,
|
||||
gp1_cmd_DMADirection = 0x04,
|
||||
gp1_cmd_StartDisplayArea = 0x05,
|
||||
gp1_cmd_HorizontalDisplayRange = 0x06,
|
||||
gp1_cmd_VerticalDisplayRange = 0x07,
|
||||
gp1_cmd_DisplayMode = 0x08,
|
||||
/* Note: GP1 only has commands 0x00..0x08.
|
||||
* The state-setter commands (SetTextureWindow, * SetDrawArea*,
|
||||
* SetDrawOffset, SetMaskBit) live in the GP0 enum as * 0xE1..0xE6.
|
||||
* DrawArea word builders are below as GP0s * macros
|
||||
* (since they emit GP0 commands). */
|
||||
gp1_cmd_Reset = 0x00,
|
||||
gp1_cmd_ResetCmdBuffer = 0x01,
|
||||
gp1_cmd_AcknowledgeIRQ = 0x02,
|
||||
gp1_cmd_DisplayEnable = 0x03,
|
||||
gp1_cmd_DMADirection = 0x04,
|
||||
gp1_cmd_StartDisplayArea = 0x05,
|
||||
gp1_cmd_HorizontalDisplayRange = 0x06,
|
||||
gp1_cmd_VerticalDisplayRange = 0x07,
|
||||
gp1_cmd_DisplayMode = 0x08,
|
||||
/* Note: GP1 only has commands 0x00..0x08.
|
||||
* The state-setter commands (SetTextureWindow, * SetDrawArea*, SetDrawOffset, SetMaskBit)
|
||||
* live in the GP0 enum as * 0xE1..0xE6.
|
||||
* DrawArea word builders are below as GP0s * macros (since they emit GP0 commands). */
|
||||
|
||||
/* ---- Display-mode payload flags (per PSX-SPX §"GP1 Display Mode").
|
||||
* Bit positions match the encoder shifts below; values are the
|
||||
* *payload* bits only (the cmd byte is OR'd in by enc_gp1_disp_mode_word). */
|
||||
gp1_disp_HRes_256 = 0x0,
|
||||
gp1_disp_HRes_320 = 0x1,
|
||||
gp1_disp_HRes_512 = 0x2,
|
||||
gp1_disp_HRes_640 = 0x3,
|
||||
gp1_disp_VRes_240 = 0x0,
|
||||
gp1_disp_VRes_480 = 0x1,
|
||||
gp1_disp_Color15 = 0x0,
|
||||
gp1_disp_Color24 = 0x1,
|
||||
gp1_disp_VInterlace = 0x1,
|
||||
/* ---- Display-mode payload flags (per PSX-SPX §"GP1 Display Mode").
|
||||
* Bit positions match the encoder shifts below; values are the *payload* bits only (cmd byte is OR'd in by enc_gp1_disp_mode_word). */
|
||||
gp1_disp_HRes_256 = 0x0,
|
||||
gp1_disp_HRes_320 = 0x1,
|
||||
gp1_disp_HRes_512 = 0x2,
|
||||
gp1_disp_HRes_640 = 0x3,
|
||||
gp1_disp_VRes_240 = 0x0,
|
||||
gp1_disp_VRes_480 = 0x1,
|
||||
gp1_disp_Color15 = 0x0,
|
||||
gp1_disp_Color24 = 0x1,
|
||||
gp1_disp_VInterlace = 0x1,
|
||||
|
||||
/* ---- Layer 1: GP1 display-mode + range + draw-area shifts/masks ---- */
|
||||
gp1_disp_hres_shift = 0, gp1_disp_hres_width = 2, gp1_disp_hres_mask = 0x3,
|
||||
gp1_disp_vres_shift = 2, gp1_disp_vres_width = 1, gp1_disp_vres_mask = 0x1,
|
||||
gp1_disp_color_shift = 4, gp1_disp_color_width = 1, gp1_disp_color_mask = 0x1,
|
||||
gp1_disp_interlace_shift = 5, gp1_disp_interlace_width = 1, gp1_disp_interlace_mask = 0x1,
|
||||
/* ---- Layer 1: GP1 display-mode + range + draw-area shifts/widths ---- */
|
||||
gp1_disp_hres_pos = 0, gp1_disp_hres_width = 2,
|
||||
gp1_disp_vres_pos = 2, gp1_disp_vres_width = 1,
|
||||
gp1_disp_color_pos = 4, gp1_disp_color_width = 1,
|
||||
gp1_disp_interlace_pos = 5, gp1_disp_interlace_width = 1,
|
||||
|
||||
/* GP1 horizontal display range: bits 0..11 = X2, bits 12..23 = X1 */
|
||||
gp1_hrange_x1_shift = 12, gp1_hrange_x1_width = 12, gp1_hrange_x1_mask = 0xFFF,
|
||||
gp1_hrange_x2_shift = 0, gp1_hrange_x2_width = 12, gp1_hrange_x2_mask = 0xFFF,
|
||||
/* GP1 horizontal display range: bits 0..11 = X2, bits 12..23 = X1 */
|
||||
gp1_hrange_x1_pos = 12, gp1_hrange_x1_width = 12,
|
||||
gp1_hrange_x2_pos = 0, gp1_hrange_x2_width = 12,
|
||||
|
||||
/* GP1 vertical display range: bits 0..9 = Y2, bits 10..19 = Y1 */
|
||||
gp1_vrange_y1_shift = 10, gp1_vrange_y1_width = 10, gp1_vrange_y1_mask = 0x3FF,
|
||||
gp1_vrange_y2_shift = 0, gp1_vrange_y2_width = 10, gp1_vrange_y2_mask = 0x3FF,
|
||||
/* GP1 vertical display range: bits 0..9 = Y2, bits 10..19 = Y1 */
|
||||
gp1_vrange_y1_pos = 10, gp1_vrange_y1_width = 10,
|
||||
gp1_vrange_y2_pos = 0, gp1_vrange_y2_width = 10,
|
||||
|
||||
/* GP1 draw area (top-left or bottom-right): bits 0..9 = X, bits 10..19 = Y
|
||||
* (10-bit signed — caller pre-signs and masks with the named mask) */
|
||||
gp1_draw_x_shift = 0, gp1_draw_x_width = 10, gp1_draw_x_mask = 0x3FF,
|
||||
gp1_draw_y_shift = 10, gp1_draw_y_width = 10, gp1_draw_y_mask = 0x3FF,
|
||||
/* GP1 draw area (top-left or bottom-right): bits 0..9 = X, bits 10..19 = Y
|
||||
* (10-bit signed — caller pre-signs) */
|
||||
gp1_draw_x_pos = 0, gp1_draw_x_width = 10,
|
||||
gp1_draw_y_pos = 10, gp1_draw_y_width = 10,
|
||||
};
|
||||
|
||||
/* ---- Layer 1.5: GP1 per-field encoders ---- */
|
||||
#define enc_gp1_disp_hres(h) (((h) & gp1_disp_hres_mask) << gp1_disp_hres_shift)
|
||||
#define enc_gp1_disp_vres(v) (((v) & gp1_disp_vres_mask) << gp1_disp_vres_shift)
|
||||
#define enc_gp1_disp_color(c) (((c) & gp1_disp_color_mask) << gp1_disp_color_shift)
|
||||
#define enc_gp1_disp_interlace(i) (((i) & gp1_disp_interlace_mask << gp1_disp_interlace_shift)
|
||||
#define enc_gp1_disp_hres(h) ((h) << gp1_disp_hres_pos)
|
||||
#define enc_gp1_disp_vres(v) ((v) << gp1_disp_vres_pos)
|
||||
#define enc_gp1_disp_color(c) ((c) << gp1_disp_color_pos)
|
||||
#define enc_gp1_disp_interlace(i) ((i) << gp1_disp_interlace_pos)
|
||||
|
||||
#define enc_gp1_hrange_x1(x1) (((x1) & gp1_hrange_x1_mask) << gp1_hrange_x1_shift)
|
||||
#define enc_gp1_hrange_x2(x2) (((x2) & gp1_hrange_x2_mask) << gp1_hrange_x2_shift)
|
||||
#define enc_gp1_vrange_y1(y1) (((y1) & gp1_vrange_y1_mask) << gp1_vrange_y1_shift)
|
||||
#define enc_gp1_vrange_y2(y2) (((y2) & gp1_vrange_y2_mask) << gp1_vrange_y2_shift)
|
||||
#define enc_gp1_draw_x(x) (((x) & gp1_draw_x_mask) << gp1_draw_x_shift)
|
||||
#define enc_gp1_draw_y(y) (((y) & gp1_draw_y_mask) << gp1_draw_y_shift)
|
||||
#define enc_gp1_hrange_x1(x1) ((x1) << gp1_hrange_x1_pos)
|
||||
#define enc_gp1_hrange_x2(x2) ((x2) << gp1_hrange_x2_pos)
|
||||
#define enc_gp1_vrange_y1(y1) ((y1) << gp1_vrange_y1_pos)
|
||||
#define enc_gp1_vrange_y2(y2) ((y2) << gp1_vrange_y2_pos)
|
||||
#define enc_gp1_draw_x(x) ((x) << gp1_draw_x_pos)
|
||||
#define enc_gp1_draw_y(y) ((y) << gp1_draw_y_pos)
|
||||
|
||||
/* ---- Layer 2: GP1 composite encoders ---- */
|
||||
#define enc_gp1_disp_mode_word(h, v, c, i) (enc_gp0_cmd(gp1_cmd_DisplayMode) | enc_gp1_disp_hres(h) | enc_gp1_disp_vres(v) | enc_gp1_disp_color(c) | enc_gp1_disp_interlace(i))
|
||||
@@ -267,12 +250,16 @@ enum {
|
||||
#define enc_gp1_vrange_word(y1, y2) (enc_gp0_cmd(gp1_cmd_VerticalDisplayRange) | enc_gp1_vrange_y1(y1) | enc_gp1_vrange_y2(y2))
|
||||
|
||||
/* ---- Layer 2: GP0 state-setter composite encoders ----
|
||||
* GP0(0xE3) SetDrawArea top-left and GP0(0xE4) SetDrawArea bottom-right
|
||||
* both use the same X/Y 10-bit signed payload as GP1 DisplayRange. */
|
||||
* GP0(0xE3) SetDrawArea top-left and GP0(0xE4) SetDrawArea bottom-right both use the same X/Y 10-bit signed payload as GP1 DisplayRange. */
|
||||
#define enc_gp0_draw_area_tl_word(x, y) (enc_gp0_cmd(gp0_cmd_SetDrawArea_TopLeft) | enc_gp1_draw_x(x) | enc_gp1_draw_y(y))
|
||||
#define enc_gp0_draw_area_br_word(x, y) (enc_gp0_cmd(gp0_cmd_SetDrawArea_BotRight) | enc_gp1_draw_x(x) | enc_gp1_draw_y(y))
|
||||
|
||||
/* ---- Layer 3: GP1 semantic word builders ---- */
|
||||
#define gp1_word_Reset() enc_gp0_cmd_word(gp1_cmd_Reset)
|
||||
#define gp1_word_ResetCmdBuffer() enc_gp0_cmd_word(gp1_cmd_ResetCmdBuffer)
|
||||
#define gp1_word_AcknowledgeIRQ() enc_gp0_cmd_word(gp1_cmd_AcknowledgeIRQ)
|
||||
#define gp1_word_StartDisplayArea() enc_gp0_cmd_word(gp1_cmd_StartDisplayArea)
|
||||
|
||||
#define gp1_word_display_enable(on) (enc_gp0_cmd(gp1_cmd_DisplayEnable) | ((on) & 1))
|
||||
#define gp1_word_display_disable() gp1_word_display_enable(0)
|
||||
#define gp1_word_display_mode_320x240_15bit_ntsc enc_gp1_disp_mode_word(gp1_disp_HRes_320, gp1_disp_VRes_240, gp1_disp_Color15, 0)
|
||||
@@ -290,47 +277,43 @@ enum {
|
||||
/* ============================================================================
|
||||
* Pre-baked GPU state words
|
||||
* ============================================================================
|
||||
*
|
||||
* Common command words for boot-time GPU init and standard
|
||||
* display configurations. Each one is a pure compile-time integer
|
||||
* constant ready to drop into a `.word` directive.
|
||||
*
|
||||
* These are the equivalents of the `gp_HorizontalDisplayRange_3168_608`,
|
||||
* `gp_VerticalDisplayRange_264_24`, `gp_DisplayMode_320x240_15bit_NTSC`,
|
||||
* `gp_SetDrawMode_DrawAllowed`, `gp_DMA_*` `.equ`s from the pre-rewrite
|
||||
* gp.h / graphics_hello/gp.s, rebuilt using the layer-cake encoders so
|
||||
* no magic numbers appear in any body.
|
||||
* Common command words for boot-time GPU init and standard display configurations.
|
||||
* ============================================================================ */
|
||||
|
||||
/* ---- Display enable (1-bit payload on DisplayEnable cmd) ---- */
|
||||
#define gp1_word_display_enabled enc_gp0_cmd_word(gp1_cmd_DisplayEnable)
|
||||
#define gp1_word_display_disabled (enc_gp0_cmd_word(gp1_cmd_DisplayEnable) | 1)
|
||||
|
||||
#define gp1_word_DisplayOn() gp1_word_display_enable(0)
|
||||
#define gp1_word_DisplayOff() gp1_word_display_enable(1)
|
||||
|
||||
/* ---- DMA direction (2-bit payload on DMADirection cmd 0x04) ---- */
|
||||
enum {
|
||||
gp1_dma_dir_Off = 0,
|
||||
gp1_dma_dir_FIFO = 1,
|
||||
gp1_dma_dir_CPU_to_GPU = 2,
|
||||
gp1_dma_dir_GPUREAD_to_CPU = 3,
|
||||
gp1_dma_dir_Off = 0,
|
||||
gp1_dma_dir_FIFO = 1,
|
||||
gp1_dma_dir_CPU_to_GPU = 2,
|
||||
gp1_dma_dir_GPUREAD_to_CPU = 3,
|
||||
};
|
||||
#define gp1_word_dma_direction(dir) (enc_gp0_cmd(gp1_cmd_DMADirection) | ((dir) & 0x3))
|
||||
#define gp1_word_dma_to_gpu() gp1_word_dma_direction(gp1_dma_dir_CPU_to_GPU)
|
||||
#define gp1_word_dma_read_cpu() gp1_word_dma_direction(gp1_dma_dir_GPUREAD_to_CPU)
|
||||
|
||||
/* ---- Standard display ranges (NTSC + PAL pre-baked) ---- */
|
||||
/* Horizontal range values are in video clock units (8 units/pixel); vertical range values are scanline numbers. */
|
||||
enum {
|
||||
/* NTSC horizontal range: X1=608, X2=3168 */
|
||||
gp1_hrange_NTSC_x1 = 0x260,
|
||||
gp1_hrange_NTSC_x2 = 0xC60,
|
||||
/* PAL horizontal range (same as NTSC for most CRTs) */
|
||||
gp1_hrange_PAL_x1 = 0x260,
|
||||
gp1_hrange_PAL_x2 = 0xC60,
|
||||
/* NTSC horizontal range: X1=608, X2=3168 */
|
||||
gp1_hrange_NTSC_x1 = 0x260,
|
||||
gp1_hrange_NTSC_x2 = 0xC60,
|
||||
/* PAL horizontal range (same as NTSC for most CRTs) */
|
||||
gp1_hrange_PAL_x1 = 0x260,
|
||||
gp1_hrange_PAL_x2 = 0xC60,
|
||||
|
||||
/* NTSC vertical range: Y1=24, Y2=264 */
|
||||
gp1_vrange_NTSC_y1 = 24,
|
||||
gp1_vrange_NTSC_y2 = 264,
|
||||
/* PAL vertical range: Y1=24, Y2=504 */
|
||||
gp1_vrange_PAL_y1 = 24,
|
||||
gp1_vrange_PAL_y2 = 504,
|
||||
/* NTSC vertical range: Y1=24, Y2=264 */
|
||||
gp1_vrange_NTSC_y1 = 24,
|
||||
gp1_vrange_NTSC_y2 = 264,
|
||||
/* PAL vertical range: Y1=24, Y2=504 */
|
||||
gp1_vrange_PAL_y1 = 24,
|
||||
gp1_vrange_PAL_y2 = 504,
|
||||
};
|
||||
|
||||
#define gp1_word_horizontal_range_ntsc enc_gp1_hrange_word(gp1_hrange_NTSC_x1, gp1_hrange_NTSC_x2)
|
||||
@@ -339,16 +322,51 @@ enum {
|
||||
#define gp1_word_vertical_range_pal enc_gp1_vrange_word(gp1_vrange_PAL_y1, gp1_vrange_PAL_y2)
|
||||
|
||||
/* ---- Draw-mode setting (TPage / draw-area allowance) ---- */
|
||||
/* The pre-baked "drawing enabled" word is the standard post-init state. */
|
||||
/* The "drawing enabled" word is the standard post-init state. */
|
||||
enum {
|
||||
gp0_DrawMode_DrawToDispBit = 10,
|
||||
/* Per psx-spx, the standard 0xE1 layout has dfe at bit 10. But libpsyx's PutDrawEnv
|
||||
* uses bit 19 (in the "unused" 14-23 range) for dfe in the DR_ENV code[0] — and the
|
||||
* PSX hardware honors bit 19 in the DR_ENV context (not bit 10). So we need a
|
||||
* separate bit definition for the DR_ENV-specific DrawMode. */
|
||||
gp0_DrawMode_DrawToDispBit = 10, // standard psx-spx bit 10 (dfe)
|
||||
gp0_DrawMode_DR_ENV_DrawToDispBit = 19, // libpsyx DR_ENV code[0] (dfe in DR_ENV context)
|
||||
gp0_DrawMode_DR_ENV_isbgBit = 19, // libpsyx uses bit 19 for isbg too
|
||||
};
|
||||
#define gp0_word_draw_mode_drawing_allowed (enc_gp0_cmd(gp0_cmd_DrawModeSetting) | (1 << gp0_DrawMode_DrawToDispBit))
|
||||
|
||||
/* ---- DrawArea pre-baked at origin (0,0) and full screen (320x240) ---- */
|
||||
/* DR_ENV-specific DrawMode variants (libpsyx SetDrawEnv layout).
|
||||
* The DR_ENV is a 16-word packet emitted at boot by gp_screen_init's ac_put_draw_env_demo
|
||||
* atom component. Within the DR_ENV, the 0xE1 command is reused in three different bit
|
||||
* configurations:
|
||||
* code[0] = `gp0_word_draw_mode_drawing_allowed` (dfe=1; standard post-init state)
|
||||
* code[6] = `gp0_word_dr_env_bg_color_cmd(isbg, r, g, b)` (initial-bg-color path)
|
||||
* code[7] = `gp0_word_dr_env_draw_mode(isbg)` (isbg-flag path)
|
||||
* Bits 0-23 of the 0xE1 word are the payload; bits 24-31 are the cmd byte (0xE1). */
|
||||
#define gp0_word_dr_env_bg_color_cmd(isbg, r, g, b) (enc_gp0_cmd(gp0_cmd_DrawModeSetting) | (1 << gp0_DrawMode_DrawToDispBit) | ((isbg) ? gp0_dr_env_isbg_bit : 0) | enc_gp0_color_r(r) | enc_gp0_color_g(g) | enc_gp0_color_b(b))
|
||||
#define gp0_word_dr_env_draw_mode(isbg) (enc_gp0_cmd(gp0_cmd_DrawModeSetting) | (1 << gp0_DrawMode_DrawToDispBit) | ((isbg) ? gp0_dr_env_isbg_bit : 0))
|
||||
|
||||
/* State-setter bare-cmd words (no immediate payload; the GPU uses the current state machine already programmed). */
|
||||
#define gp0_word_set_texture_window() enc_gp0_cmd_word(gp0_cmd_SetTextureWindow)
|
||||
#define gp0_word_set_draw_offset() enc_gp0_cmd_word(gp0_cmd_SetDrawOffset)
|
||||
#define gp0_word_set_mask_bit() enc_gp0_cmd_word(gp0_cmd_SetMaskBit)
|
||||
|
||||
/* DR_ENV code[5] Mask (0xE6 cmd + isbg bit). The isbg bit is set so the GPU knows the auto-clear path is active (paired with code[6] + code[7]). */
|
||||
#define gp0_word_dr_env_mask() (gp0_word_set_mask_bit() | gp0_dr_env_isbg_bit)
|
||||
|
||||
/* DR_ENV pre-baked constants (libpsyx PutDrawEnv layout).
|
||||
* DR_ENV is a 16-word packet: tag = (length << 24) | addr, where length = 15 (15 code words follow) and addr = 0 (chain to nothing). */
|
||||
enum {
|
||||
PolyTag_len_bits = 8,
|
||||
PolyTag_addr_bits = 24,
|
||||
|
||||
gp0_dr_env_tag = (15 << 24) | 0x00FFFFFF,
|
||||
gp0_dr_env_isbg_bit = (1 << gp0_DrawMode_DR_ENV_isbgBit),
|
||||
};
|
||||
|
||||
/* ---- DrawArea at origin (0,0) and full screen (320x240) ---- */
|
||||
#define gp0_word_draw_area_top_left_origin enc_gp0_draw_area_tl_word(0, 0)
|
||||
#define gp0_word_draw_area_bottom_right_320x240 enc_gp0_draw_area_br_word(320, 240)
|
||||
#define gp0_word_draw_area_bottom_right_640x480 enc_gp0_draw_area_br_word(640, 480)
|
||||
#define gp0_word_draw_area_bottom_right_320x240 enc_gp0_draw_area_br_word(319, 239)
|
||||
#define gp0_word_draw_area_bottom_right_640x480 enc_gp0_draw_area_br_word(639, 479)
|
||||
|
||||
#pragma endregion GPU Ports & Commands
|
||||
|
||||
@@ -359,9 +377,9 @@ enum {
|
||||
* Read from HW_GP1; the lower bits are DMA-block-size (variable-width).
|
||||
* ============================================================================ */
|
||||
enum {
|
||||
gp1_Status_BitReady = 31,
|
||||
gp1_Status_BitSendingDMA = 25,
|
||||
gp1_Status_DMABlockSizeShift = 0,
|
||||
gp1_Status_BitReady = 31,
|
||||
gp1_Status_BitSendingDMA = 25,
|
||||
gp1_Status_DMABlockSizeShift = 0,
|
||||
};
|
||||
|
||||
#define gp1_status_is_ready() ((HW_GP1[0] >> gp1_Status_BitReady) & 1)
|
||||
@@ -372,17 +390,14 @@ enum {
|
||||
/* ============================================================================
|
||||
* Primitive structs (8 polygon variants + tag)
|
||||
* ============================================================================
|
||||
* Each struct follows the GPU-documented memory layout for the corresponding primitive command.
|
||||
* The PolyTag is the OT-link header; the rest of the struct is the primitive's body.
|
||||
*
|
||||
* Each struct follows the GPU-documented memory layout for the corresponding
|
||||
* primitive command. The PolyTag is the OT-link header; the rest of the
|
||||
* struct is the primitive's body.
|
||||
*
|
||||
* The current working layouts match the existing demo (floor_tri uses
|
||||
* Poly_F3; cube_tri uses Poly_G4). They are NOT necessarily byte-identical
|
||||
* to the PSX-SPX reference layout — the demo layout uses color+vertex
|
||||
* interleaving that doesn't match the standard PSX SDK file format. For
|
||||
* PSX-SDK file compatibility, the textured variants (FT*, GT*) would need
|
||||
* layout adjustments; out of scope for this track.
|
||||
* The current working layouts match the existing demo
|
||||
* (floor_tri uses Poly_F3; cube_tri uses Poly_G4).
|
||||
* They are NOT necessarily byte-identical to the PSX-SPX reference layout.
|
||||
* The demo layout uses color+vertex interleaving that doesn't match the standard PSX SDK file format.
|
||||
* For PSX-SDK file compatibility, the textured variants (FT*, GT*) would need layout adjustments.
|
||||
* ============================================================================ */
|
||||
|
||||
/* ---------- RGB8 (3-byte packed color) ---------- */
|
||||
@@ -390,13 +405,13 @@ typedef Struct_(RGB8) { B1 r; B1 g; B1 b; };
|
||||
#define rgb8(r,g,b) ((RGB8){r,g,b})
|
||||
|
||||
/* ---------- PolyTag (the OT-link header; 1 word) ---------- */
|
||||
enum {
|
||||
polytag_len_bits = 8,
|
||||
polytag_addr_bits = 24,
|
||||
};
|
||||
// enum {
|
||||
// PolyTag_len_bits = 8,
|
||||
// PolyTag_addr_bits = 24,
|
||||
// };
|
||||
typedef Struct_(PolyTag) {
|
||||
union {
|
||||
U4 bf_addr_len;
|
||||
U4 code;
|
||||
struct {
|
||||
U4 addr: 24;
|
||||
U4 len: 8;
|
||||
@@ -404,112 +419,108 @@ typedef Struct_(PolyTag) {
|
||||
};
|
||||
};
|
||||
|
||||
/* DSL cast convention: every cast uses `C_()`, every pointer qualifier
|
||||
* is `R_` (restrict) or `V_` (volatile). No raw C-style casts. RHS values
|
||||
* are assumed to be `U4` — caller passes a `U4` directly. */
|
||||
#define set_len(tag,v) (C_(PolyTag_R,tag)->len = u4_(v))
|
||||
#define set_addr(tag,v) (C_(PolyTag_R,tag)->addr = u4_(v))
|
||||
/* `set_code` is no longer in the new PolyTag design — the code byte lives
|
||||
* in the primitive body (e.g. `((Poly_F3*)(p))->code`), not in the tag.
|
||||
* Use the typed primitive structs (Poly_F3, Poly_G4, etc.) and the
|
||||
* `set_poly_*` setters, which set both the tag's length and the code. */
|
||||
/* `set_code` is no longer in the new PolyTag design
|
||||
* (e.g. `((Poly_F3*)(p))->code`), not in the tag.
|
||||
* Use the typed primitive structs (Poly_F3, Poly_G4, etc.) and the `set_poly_*` setters, which set both the tag's length and the code. */
|
||||
#define get_len(tag) C_(U4,C_(PolyTag_R,tag)->len)
|
||||
#define get_addr(tag) C_(U4,C_(PolyTag_R,tag)->addr)
|
||||
|
||||
/* ---------- Poly_F3 (Flat Triangle; 5 words) ---------- */
|
||||
typedef Struct_(Poly_F3) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
union {
|
||||
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; };
|
||||
A3_V2_S2 points;
|
||||
};
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
union {
|
||||
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; };
|
||||
A3_V2_S2 points;
|
||||
};
|
||||
};
|
||||
|
||||
/* ---------- Poly_F4 (Flat Quad; 6 words) ---------- */
|
||||
typedef Struct_(Poly_F4) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
union {
|
||||
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; V2_S2 p3; };
|
||||
A4_V2_S2 points;
|
||||
};
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
union {
|
||||
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; V2_S2 p3; };
|
||||
A4_V2_S2 points;
|
||||
};
|
||||
};
|
||||
|
||||
/* ---------- Poly_G3 (Gouraud Triangle; 6 words) ---------- */
|
||||
/* ---------- Poly_G3 (Gouraud Triangle; 7 words) ---------- */
|
||||
typedef Struct_(Poly_G3) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2;
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2;
|
||||
};
|
||||
|
||||
/* ---------- Poly_G4 (Gouraud Quad; 5 words in the demo's interleaved layout) ---------- */
|
||||
/* ---------- Poly_G4 (Gouraud Quad; 9 words) ---------- */
|
||||
typedef Struct_(Poly_G4) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2; RGB8 c3; B1 pad3;
|
||||
V2_S2 p3;
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2; RGB8 c3; B1 pad3;
|
||||
V2_S2 p3;
|
||||
};
|
||||
|
||||
/* ---------- Poly_FT3 (Flat Textured Triangle; placeholder layout) ---------- */
|
||||
/* TODO(Ed): verify the textured-variant layout against PSX-SPX when needed. */
|
||||
typedef Struct_(Poly_FT3) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 p0; U1 u0; U1 v0;
|
||||
V2_S2 p1; U1 u1; U1 v1;
|
||||
V2_S2 p2; U1 u2; U1 v2;
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 p0; U1 u0; U1 v0;
|
||||
V2_S2 p1; U1 u1; U1 v1;
|
||||
V2_S2 p2; U1 u2; U1 v2;
|
||||
};
|
||||
|
||||
/* ---------- Poly_FT4 (Flat Textured Quad; placeholder layout) ---------- */
|
||||
/* ---------- Poly_FT4 (Flat Textured Quad) ---------- */
|
||||
typedef Struct_(Poly_FT4) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 p0; U1 u0; U1 v0;
|
||||
V2_S2 p1; U1 u1; U1 v1;
|
||||
V2_S2 p2; U1 u2; U1 v2;
|
||||
V2_S2 p3; U1 u3; U1 v3;
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 p0; U1 u0; U1 v0;
|
||||
V2_S2 p1; U1 u1; U1 v1;
|
||||
V2_S2 p2; U1 u2; U1 v2;
|
||||
V2_S2 p3; U1 u3; U1 v3;
|
||||
};
|
||||
|
||||
/* ---------- Poly_GT3 (Gouraud Textured Triangle; placeholder layout) ---------- */
|
||||
/* ---------- Poly_GT3 (Gouraud Textured Triangle) ---------- */
|
||||
typedef Struct_(Poly_GT3) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 tp0; U1 u0; U1 v0;
|
||||
V2_S2 tp1; U1 u1; U1 v1;
|
||||
V2_S2 tp2; U1 u2; U1 v2;
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 tp0; U1 u0; U1 v0;
|
||||
V2_S2 tp1; U1 u1; U1 v1;
|
||||
V2_S2 tp2; U1 u2; U1 v2;
|
||||
};
|
||||
|
||||
/* ---------- Poly_GT4 (Gouraud Textured Quad; placeholder layout) ---------- */
|
||||
/* ---------- Poly_GT4 (Gouraud Textured Quad) ---------- */
|
||||
typedef Struct_(Poly_GT4) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2; RGB8 c3; B1 pad3;
|
||||
V2_S2 p3;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 tp0; U1 u0; U1 v0;
|
||||
V2_S2 tp1; U1 u1; U1 v1;
|
||||
V2_S2 tp2; U1 u2; U1 v2;
|
||||
V2_S2 tp3; U1 u3; U1 v3;
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2; RGB8 c3; B1 pad3;
|
||||
V2_S2 p3;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 tp0; U1 u0; U1 v0;
|
||||
V2_S2 tp1; U1 u1; U1 v1;
|
||||
V2_S2 tp2; U1 u2; U1 v2;
|
||||
V2_S2 tp3; U1 u3; U1 v3;
|
||||
};
|
||||
|
||||
/* ---------- Primitive setters (C-level, no emitted words) ----------
|
||||
/* ---------- Primitive setters (C-level) ----------
|
||||
* DSL cast convention: every cast via C_(), every pointer via R_/V_. */
|
||||
#define set_poly_f3(p) set_len(p, 4), C_(Poly_F3_R, p)->code = gp0_cmd_poly_f3
|
||||
#define set_poly_ft3(p) set_len(p, 7), C_(Poly_FT3_R,p)->code = gp0_cmd_poly_ft3
|
||||
@@ -530,7 +541,6 @@ typedef Struct_(Poly_GT4) {
|
||||
/* ============================================================================
|
||||
* Texture Page (TPage) bit layout
|
||||
* ============================================================================
|
||||
*
|
||||
* The TPage data word sent via GP0(0x2X) has:
|
||||
* bits 0..3 = texture page X (4 bits, 64-px units, 0..16)
|
||||
* bit 4 = texture page Y (1 bit, 64-px units, 0/1)
|
||||
@@ -542,59 +552,61 @@ typedef Struct_(Poly_GT4) {
|
||||
* bits 12..31 = reserved (zero)
|
||||
* ============================================================================ */
|
||||
enum {
|
||||
/* ---- Layer 1: TPage bitfield shifts / widths / masks ---- */
|
||||
gp0_tpage_x_shift = 0, gp0_tpage_x_width = 4, gp0_tpage_x_mask = 0xF,
|
||||
gp0_tpage_y_shift = 4, gp0_tpage_y_width = 1, gp0_tpage_y_mask = 0x1,
|
||||
gp0_tpage_semi_trans_shift = 5, gp0_tpage_semi_trans_width = 2, gp0_tpage_semi_trans_mask = 0x3,
|
||||
gp0_tpage_color_depth_shift = 7, gp0_tpage_color_depth_width = 2, gp0_tpage_color_depth_mask = 0x3,
|
||||
gp0_tpage_dither_shift = 9, gp0_tpage_dither_width = 1, gp0_tpage_dither_mask = 0x1,
|
||||
gp0_tpage_draw_to_disp_shift = 10, gp0_tpage_draw_to_disp_width = 1, gp0_tpage_draw_to_disp_mask = 0x1,
|
||||
gp0_tpage_tex_disable_shift = 11, gp0_tpage_tex_disable_width = 1, gp0_tpage_tex_disable_mask = 0x1,
|
||||
/* ---- Layer 1: TPage bitfield shifts / widths ---- */
|
||||
gp0_tpage_x_pos = 0, gp0_tpage_x_width = 4,
|
||||
gp0_tpage_y_pos = 4, gp0_tpage_y_width = 1,
|
||||
gp0_tpage_semi_trans_pos = 5, gp0_tpage_semi_trans_width = 2,
|
||||
gp0_tpage_color_depth_pos = 7, gp0_tpage_color_depth_width = 2,
|
||||
gp0_tpage_dither_pos = 9, gp0_tpage_dither_width = 1,
|
||||
gp0_tpage_draw_to_disp_pos = 10, gp0_tpage_draw_to_disp_width = 1,
|
||||
gp0_tpage_tex_disable_pos = 11, gp0_tpage_tex_disable_width = 1,
|
||||
|
||||
/* TPage color-depth payload values (NOT bit positions — these go in
|
||||
* the 2-bit field at gp0_tpage_color_depth_shift). */
|
||||
gp0_tpage_color_4bpp = 0x0,
|
||||
gp0_tpage_color_8bpp = 0x1,
|
||||
gp0_tpage_color_16bpp = 0x2,
|
||||
/* TPage color-depth payload values (NOT bit positions — these go in
|
||||
* the 2-bit field at gp0_tpage_color_depth_pos). */
|
||||
gp0_tpage_color_4bpp = 0x0,
|
||||
gp0_tpage_color_8bpp = 0x1,
|
||||
gp0_tpage_color_16bpp = 0x2,
|
||||
|
||||
/* TPage semi-transparency mode payload values (NOT bit positions). */
|
||||
gp0_tpage_semi_trans_none = 0x0,
|
||||
gp0_tpage_semi_trans_alpha = 0x1,
|
||||
gp0_tpage_semi_trans_add = 0x2,
|
||||
gp0_tpage_semi_trans_sub = 0x3,
|
||||
/* Default TPage value libpsyx's SetDefDrawEnv writes (matches the `li v1, 10; sh v1, 20(v0)` sequence at C11_only.elf:0x8001273C). */
|
||||
gp0_tpage_default = 10,
|
||||
|
||||
/* TPage semi-transparency mode payload values. */
|
||||
gp0_tpage_semi_trans_none = 0x0,
|
||||
gp0_tpage_semi_trans_alpha = 0x1,
|
||||
gp0_tpage_semi_trans_add = 0x2,
|
||||
gp0_tpage_semi_trans_sub = 0x3,
|
||||
};
|
||||
|
||||
/* ---- Layer 1.5: TPage per-field encoders. Mirrors enc_gte_sf/mx/v in gte.h. ---- */
|
||||
#define enc_gp0_tpage_x(x) (((x) & gp0_tpage_x_mask) << gp0_tpage_x_shift)
|
||||
#define enc_gp0_tpage_y(y) (((y) & gp0_tpage_y_mask) << gp0_tpage_y_shift)
|
||||
#define enc_gp0_tpage_semi_trans(s) (((s) & gp0_tpage_semi_trans_mask) << gp0_tpage_semi_trans_shift)
|
||||
#define enc_gp0_tpage_color_depth(c) (((c) & gp0_tpage_color_depth_mask) << gp0_tpage_color_depth_shift)
|
||||
#define enc_gp0_tpage_dither(d) (((d) & gp0_tpage_dither_mask) << gp0_tpage_dither_shift)
|
||||
#define enc_gp0_tpage_draw_to_disp(d) (((d) & gp0_tpage_draw_to_disp_mask) << gp0_tpage_draw_to_disp_shift)
|
||||
#define enc_gp0_tpage_tex_disable(t) (((t) & gp0_tpage_tex_disable_mask) << gp0_tpage_tex_disable_shift)
|
||||
#define enc_gp0_tpage_x(x) ((x) << gp0_tpage_x_pos)
|
||||
#define enc_gp0_tpage_y(y) ((y) << gp0_tpage_y_pos)
|
||||
#define enc_gp0_tpage_semi_trans(s) ((s) << gp0_tpage_semi_trans_pos)
|
||||
#define enc_gp0_tpage_color_depth(c) ((c) << gp0_tpage_color_depth_pos)
|
||||
#define enc_gp0_tpage_dither(d) ((d) << gp0_tpage_dither_pos)
|
||||
#define enc_gp0_tpage_draw_to_disp(d) ((d) << gp0_tpage_draw_to_disp_pos)
|
||||
#define enc_gp0_tpage_tex_disable(t) ((t) << gp0_tpage_tex_disable_pos)
|
||||
|
||||
/* ---- Layer 2: TPage composite encoder. Mirrors enc_gte_cmdw in gte.h ---- */
|
||||
#define enc_gp0_tpage_word(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
|
||||
(enc_gp0_tpage_x(x) \
|
||||
| enc_gp0_tpage_y(y) \
|
||||
| enc_gp0_tpage_semi_trans(semi_trans) \
|
||||
| enc_gp0_tpage_color_depth(color_depth) \
|
||||
| enc_gp0_tpage_dither(dither) \
|
||||
| enc_gp0_tpage_draw_to_disp(draw_to_disp) \
|
||||
| enc_gp0_tpage_tex_disable(tex_disable))
|
||||
(enc_gp0_tpage_x(x) \
|
||||
| enc_gp0_tpage_y(y) \
|
||||
| enc_gp0_tpage_semi_trans(semi_trans) \
|
||||
| enc_gp0_tpage_color_depth(color_depth) \
|
||||
| enc_gp0_tpage_dither(dither) \
|
||||
| enc_gp0_tpage_draw_to_disp(draw_to_disp) \
|
||||
| enc_gp0_tpage_tex_disable(tex_disable))
|
||||
|
||||
typedef Struct_(TexturePage) { U4 raw; };
|
||||
|
||||
/* ---- Layer 3: TPage semantic word builder ---- */
|
||||
#define gp0_word_tpage(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
|
||||
enc_gp0_tpage_word((x), (y), (semi_trans), (color_depth), (dither), (draw_to_disp), (tex_disable))
|
||||
enc_gp0_tpage_word((x), (y), (semi_trans), (color_depth), (dither), (draw_to_disp), (tex_disable))
|
||||
#pragma endregion TPage
|
||||
|
||||
#pragma region CLUT
|
||||
/* ============================================================================
|
||||
* CLUT (Color Look-Up Table) semantics
|
||||
* ============================================================================
|
||||
*
|
||||
* CLUT is loaded into VRAM by sending a GP0 command whose payload is:
|
||||
* bits 0..5 = Y in 16-px units (palette row)
|
||||
* bits 6..14 = X in 16-px units (palette column)
|
||||
@@ -602,17 +614,17 @@ typedef Struct_(TexturePage) { U4 raw; };
|
||||
* bits 24..31 = command byte — 0x20 (4bpp load) or 0x25 (8bpp load)
|
||||
* ============================================================================ */
|
||||
enum {
|
||||
/* ---- Layer 1: CLUT bitfield shifts / widths / masks ---- */
|
||||
gp0_clut_y_shift = 0, gp0_clut_y_width = 6, gp0_clut_y_mask = 0x3F,
|
||||
gp0_clut_x_shift = 6, gp0_clut_x_width = 9, gp0_clut_x_mask = 0x1FF,
|
||||
/* CLUT-load cmd-byte variants — the upper byte of the GP0 word. */
|
||||
gp0_clut_cmd_Load4bpp = 0x20,
|
||||
gp0_clut_cmd_Load8bpp = 0x25,
|
||||
/* ---- Layer 1: CLUT bitfield shifts / widths ---- */
|
||||
gp0_clut_y_pos = 0, gp0_clut_y_width = 6,
|
||||
gp0_clut_x_pos = 6, gp0_clut_x_width = 9,
|
||||
/* CLUT-load cmd-byte variants — the upper byte of the GP0 word. */
|
||||
gp0_clut_cmd_Load4bpp = 0x20,
|
||||
gp0_clut_cmd_Load8bpp = 0x25,
|
||||
};
|
||||
|
||||
/* ---- Layer 1.5: CLUT per-field encoders ---- */
|
||||
#define enc_gp0_clut_x(x) (((x) & gp0_clut_x_mask) << gp0_clut_x_shift)
|
||||
#define enc_gp0_clut_y(y) (((y) & gp0_clut_y_mask) << gp0_clut_y_shift)
|
||||
#define enc_gp0_clut_x(x) ((x) << gp0_clut_x_pos)
|
||||
#define enc_gp0_clut_y(y) ((y) << gp0_clut_y_pos)
|
||||
|
||||
/* ---- Layer 2: CLUT composite encoder ---- */
|
||||
#define enc_gp0_clut_word(cmd, x, y) (enc_gp0_cmd(cmd) | enc_gp0_clut_x(x) | enc_gp0_clut_y(y))
|
||||
@@ -627,7 +639,6 @@ enum {
|
||||
/* ============================================================================
|
||||
* TIM file format constants and headers
|
||||
* ============================================================================
|
||||
*
|
||||
* TIM (Sony .TIM texture image) file structure:
|
||||
* +0x00 U4 file_id (always 0x10 = TIM magic)
|
||||
* +0x04 U4 version (always 0x00 for v1)
|
||||
@@ -645,31 +656,30 @@ enum {
|
||||
* +0x06 U2 px_height
|
||||
* +0x08 ... pixel data
|
||||
*
|
||||
* Future?: add `tim_load_to_vram(tim_ptr, vram_addr)` that
|
||||
* emits the necessary GP0 commands. Stoppped for now at the
|
||||
* struct + enum level for this track.
|
||||
* Future?: add `tim_load_to_vram(tim_ptr, vram_addr)` that emits the necessary GP0 commands.
|
||||
* Stoppped for now at the struct + enum level.
|
||||
* ============================================================================ */
|
||||
enum {
|
||||
tim_file_id_magic = 0x10,
|
||||
tim_type_4bpp = 0x00,
|
||||
tim_type_8bpp = 0x01,
|
||||
tim_type_16bpp = 0x02,
|
||||
tim_type_32bpp = 0x03,
|
||||
tim_type_mixed = 0x04,
|
||||
tim_flag_has_clut = 0x08,
|
||||
tim_file_id_magic = 0x10,
|
||||
tim_type_4bpp = 0x00,
|
||||
tim_type_8bpp = 0x01,
|
||||
tim_type_16bpp = 0x02,
|
||||
tim_type_32bpp = 0x03,
|
||||
tim_type_mixed = 0x04,
|
||||
tim_flag_has_clut = 0x08,
|
||||
};
|
||||
|
||||
typedef Struct_(TIM_Header) {
|
||||
U4 file_id; /* always 0x10 = "TIM" magic */
|
||||
U4 version; /* ignored; always 0 */
|
||||
U4 flags; /* bits 0..2 = type, bit 3 = has_clut */
|
||||
U4 file_id; /* always 0x10 = "TIM" magic */
|
||||
U4 version; /* ignored; always 0 */
|
||||
U4 flags; /* bits 0..2 = type, bit 3 = has_clut */
|
||||
};
|
||||
typedef Struct_(TIM_SectionHeader) {
|
||||
U4 section_length; /* bytes in this section including this header */
|
||||
U2 org_x; /* origin in VRAM */
|
||||
U2 org_y;
|
||||
U2 width; /* width in pixels */
|
||||
U2 height; /* height in pixels */
|
||||
U4 section_length; /* bytes in this section including this header */
|
||||
U2 org_x; /* origin in VRAM */
|
||||
U2 org_y;
|
||||
U2 width; /* width in pixels */
|
||||
U2 height; /* height in pixels */
|
||||
};
|
||||
#pragma endregion TIM File Format
|
||||
|
||||
@@ -678,35 +688,24 @@ typedef Struct_(TIM_SectionHeader) {
|
||||
* Tape-side GPU operations (NOT in this header)
|
||||
* ============================================================================
|
||||
*
|
||||
* No `mac_gp0_send` or related macros live in gp.h. Rationale: the
|
||||
* Lottes tape model uses OT-DMA for primitive submission, so atom bodies
|
||||
* write to main RAM (the OT/primitive buffer) and to GTE state — never
|
||||
* directly to the GPU ports at 0x1F801810 / 0x1F801814. See
|
||||
* `mac_format_f3_color`, `mac_insert_ot_tag`, `mac_gte_store_f3` in
|
||||
* lottes_tape.h for the patterns atom bodies actually use.
|
||||
* No `mac_gp0_send` or related macros live in gp.h.
|
||||
* Rationale: the Lottes tape model uses OT-DMA for primitive submission, so atom bodies write to main RAM (the OT/primitive buffer)
|
||||
* and to GTE state — never directly to the GPU ports at 0x1F801810 / 0x1F801814.
|
||||
* See `mac_format_f3_color`, `mac_insert_ot_tag`, `mac_gte_store_f3` in lottes_tape.h for the patterns atom bodies actually use.
|
||||
*
|
||||
* If a feature need arises requires tape-side GPU port writes (e.g. DMA-kick to
|
||||
* start GPU consumption of the OT, VBlank sync via GP1 status poll),
|
||||
* the right home is `lottes_tape.h` alongside the rest of the `mac_*`
|
||||
* family — the encoder infrastructure is already in place:
|
||||
* If a feature need arises requires tape-side GPU port writes
|
||||
* (e.g. DMA-kick to start GPU consumption of the OT, VBlank sync via GP1 status poll),
|
||||
* the right home is `lottes_tape.h` alongside the rest of the `mac_*` family:
|
||||
* 1. The caller pins a register to hold the IO base, e.g. register U4 r_io rgcc(R_T4) = IO_BASE_ADDR;
|
||||
* The compiler emits `lui R_T4, IO_BASE_ADDR_HI16` outside the atom body (in the C prologue before tape_run).
|
||||
* 2. The atom body uses `store_word(R_data, R_T4, GPIO_PORT0_OFFSET)` to write to GP0, and `store_word(R_data, R_T4, GPIO_PORT1_OFFSET)`
|
||||
* to write to GP1. Both are preprocessor-encodable because R_T4 is a fixed register and the GPIO_PORT*_OFFSET constants
|
||||
* fit in the `sw`'s 16-bit signed offset field. No placeholder-pun, no asm constraints, no hidden register choice.
|
||||
* Same pattern as the old graphics_hello/hello_gp_routines.s `reg_io_offset`/`gcmd_push` convention.
|
||||
*
|
||||
* 1. The caller pins a register to hold the IO base, e.g.
|
||||
* register U4 r_io rgcc(R_T4) = IO_BASE_ADDR;
|
||||
* The compiler emits `lui R_T4, IO_BASE_ADDR_HI16` outside the
|
||||
* atom body (in the C prologue before tape_run).
|
||||
*
|
||||
* 2. The atom body uses `store_word(R_data, R_T4, GPIO_PORT0_OFFSET)`
|
||||
* to write to GP0, and `store_word(R_data, R_T4, GPIO_PORT1_OFFSET)`
|
||||
* to write to GP1. Both are preprocessor-encodable because R_T4 is
|
||||
* a fixed register and the GPIO_PORT*_OFFSET constants fit in the
|
||||
* `sw`'s 16-bit signed offset field. No placeholder-pun, no asm
|
||||
* constraints, no hidden register choice. Same pattern as the
|
||||
* old graphics_hello/hello_gp_routines.s `reg_io_offset`/`gcmd_push`
|
||||
* convention.
|
||||
*
|
||||
* This mirrors the existing tape-side wave-context discipline: the
|
||||
* caller binds the IO-base register via `rgcc()`, the macro assumes
|
||||
* the binding is in effect, and the encoding falls out at preprocessor
|
||||
* time. No additional GPU-domain macro layer required.
|
||||
* This mirrors the existing tape-side wave-context discipline:
|
||||
* the caller binds the IO-base register via `rgcc()`, the macro assumes the binding is in effect,
|
||||
* and the encoding falls out at preprocessor time.
|
||||
* No additional GPU-domain macro layer required.
|
||||
* ============================================================================ */
|
||||
#pragma endregion Tape-Side Macros
|
||||
|
||||
@@ -2,10 +2,8 @@
|
||||
* duffle DSL — GPU Vendor Mnemonics (opt-in)
|
||||
* ============================================================================
|
||||
*
|
||||
* Provides the PSYQ-style CamelCase aliases for the canonical duffle GPU
|
||||
* primitive setters and OT operations. The duffle snake_case names are
|
||||
* primary; this header is for users who prefer the PSYQ SDK function
|
||||
* names from the legacy C API.
|
||||
* Provides the PSYQ-style CamelCase aliases for the duffle GPU primitive setters and OT operations.
|
||||
* The duffle snake_case names are primary; this header is for users who prefer the PSYQ SDK function names from the legacy C API.
|
||||
*
|
||||
* USAGE: #include "duffle/gp_vendor_sym.h" // after gp.h
|
||||
*
|
||||
@@ -23,15 +21,11 @@
|
||||
* OT operations:
|
||||
* AddPrim(ot, p) -> orderingtbl_add_primitive(ot, p)
|
||||
*
|
||||
* The gp0_cmd_* / gp1_cmd_* byte constants are already short and
|
||||
* descriptive; no vendor alias is provided for them.
|
||||
*
|
||||
* The vendor mnemonics are NOT registered with the duffle word-count
|
||||
* metadata (tape_atom.metadata.h). They expand to the duffle canonical
|
||||
* macros which DO have word-count entries (the ones emitted by
|
||||
* mac_format_f3_color / mac_gte_store_f3 / etc.). Verification: V13
|
||||
* (objdump byte-identical) holds.
|
||||
* The gp0_cmd_* / gp1_cmd_* byte constants are already short and descriptive; no vendor alias is provided for them.
|
||||
*
|
||||
* The vendor mnemonics are NOT registered with the duffle word-count metadata (word_counts.metadata.h).
|
||||
* They expand to the duffle macros which DO have word-count entries
|
||||
* (the ones emitted by mac_format_f3_color / mac_gte_store_f3 / etc.). Verification: V13 (objdump byte-identical) holds.
|
||||
* ============================================================================ */
|
||||
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
|
||||
@@ -0,0 +1,394 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# include "gen/macs.h"
|
||||
# include "gen/offsets.h"
|
||||
# include "gte.h"
|
||||
# include "gp.h"
|
||||
# include "lottes_tape.h"
|
||||
#endif
|
||||
|
||||
ATOM_FILE_DEBUGGER_LINE_MARKER(gte_atom_c);
|
||||
|
||||
#pragma region MACs (Mips Atom Components)
|
||||
|
||||
/* Words: 3; Loads 3 S2 indices from the face array */
|
||||
FI_ Slice_MipsCode ac_load_tri_indices(AtomBuilder_R ab, U4 r_face_cusor, U4 r_i0, U4 r_i1, U4 r_i2)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)),
|
||||
load_half_u(r_i1, r_face_cusor, 1 * S_(S2)),
|
||||
load_half_u(r_i2, r_face_cusor, 2 * S_(S2)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_gte_mv_to_cr_diag_v3s4(AtomBuilder_R ab, Reg_(V3_S4) v) MipsAtomComp_Proc_(ab, {
|
||||
gte_mv_to_ctrl_r(v.y, gte_cr_RT13),
|
||||
gte_mv_to_ctrl_r(v.z, gte_cr_RT22),
|
||||
gte_mv_to_ctrl_r(v.x, gte_cr_RT11),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_gte_ld_ir123_v3s4(AtomBuilder_R ab, Reg_(V3_S4) v) MipsAtomComp_Proc_(ab, {
|
||||
gte_mv_to_data_r(v.x, C2_IR1),
|
||||
gte_mv_to_data_r(v.y, C2_IR2),
|
||||
gte_mv_to_data_r(v.z, C2_IR3),
|
||||
})
|
||||
|
||||
/* ─── GTE OP cross product (a × b → a) ───
|
||||
* Sets up RT diagonal from a.xyz, IR1/2/3 from b.xyz, fires OP,
|
||||
* reads MAC1/2/3, shifts right 12 (S12.20 → S12.0 OuterProduct12), writes back to a.xyz.
|
||||
* Composes the three sub-primitives (RT-load, IR-load, OP, MAC-read, shift)
|
||||
* into one component for use by atoms that need the cross product inline.
|
||||
*
|
||||
* Output gpr (a) aliases source-A gpr; MAC read clobbers source-A's load targets,
|
||||
* but by that point the RT load is complete and source A is dead.
|
||||
* Pipeline: clobbers IR1..3, MAC1..3, RT11..33.
|
||||
*
|
||||
* The CPU→COP2 transfer chains (3 ctc2, 3 mtc2) require a 2-slot retirement gap,
|
||||
* and the MFC2→GPR chain (3 mfc2) requires a 1-slot retirement gap, before the GPR can be read.
|
||||
* The hazard nops are inlined below — same convention as ac_gte_gpf_scale — so any atom body inlining this component inherits them.
|
||||
*
|
||||
* Words: 18 (3 ctc2 + 2 nop + 3 mtc2 + 2 nop + 1 op + 3 mfc2 + 1 nop + 3 sra).
|
||||
*/
|
||||
FI_ Slice_MipsCode ac_gte_op_cross_v3s4(AtomBuilder_R ab, Reg_(V3_S4) a, Reg_(V3_S4) b) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
mac_gte_mv_to_cr_diag_v3s4(a), GteDelay_ /* RT diagonal: D1 = a.x, D2 = a.y, D3 = a.z */
|
||||
mac_gte_ld_ir123_v3s4(b), GteDelay_ /* IR: second operand (b.xyz) */
|
||||
gte_cmdw_cross, /* OP: MAC1/2/3 = a × b (S12.20) */
|
||||
mac_gte_mv_from_mac123_v3s4(a), GteDelay_ /* Read MAC1/2/3 → a.xyz (overwrites source-A's load targets) */
|
||||
mac_shift_aright_v3s4_self(a, 12), /* Right-shift MAC by 12 (S12.20 → S12.0 OuterProduct12) */
|
||||
})
|
||||
|
||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
|
||||
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
|
||||
FI_ Slice_MipsCode ac_gte_store_f3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)),
|
||||
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_F3,p1)),
|
||||
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_F3,p2)),
|
||||
})
|
||||
|
||||
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
|
||||
I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, Reg vbase, Reg v0, Reg v1, Reg v2) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
shift_lleft(R_AT, v0, v3s2_byteoff), add_u_self(R_AT, vbase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
shift_lleft(R_AT, v1, v3s2_byteoff), add_u_self(R_AT, vbase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
||||
shift_lleft(R_AT, v2, v3s2_byteoff), add_u_self(R_AT, vbase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
||||
})
|
||||
|
||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
|
||||
* G4 triangle portion to p0/p1/p2.
|
||||
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
|
||||
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
|
||||
* (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
|
||||
FI_ Slice_MipsCode ac_gte_store_g4_p012(AtomBuilder_R ab, Reg r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)),
|
||||
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)),
|
||||
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2)),
|
||||
})
|
||||
|
||||
/* Words: 1; Stores the V3 screen coord to the G4's p3 slot.
|
||||
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2;
|
||||
* SXY0 still holds v0.screen from the earlier RTPT.
|
||||
*/
|
||||
FI_ Slice_MipsCode ac_gte_store_g4_p3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) })
|
||||
|
||||
/* ─── STAGE 1 of normalize: SQR + mfc2 MAC1/2/3 ───
|
||||
* Emits squared magnitude per component (in MAC1/2/3) into caller-provided scratch regs. */
|
||||
FI_ Slice_MipsCode ac_gte_sqr_v3(AtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_sq_x, U4 r_sq_y, U4 r_sq_z) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop),
|
||||
gte_mv_from_data_r(r_sq_x, C2_MAC1),
|
||||
gte_mv_from_data_r(r_sq_y, C2_MAC2),
|
||||
gte_mv_from_data_r(r_sq_z, C2_MAC3),
|
||||
})
|
||||
|
||||
/* ─── SQR FIRE — mtc2 3 GPRs into IR1/IR2/IR3, then fire SQR. ─── */
|
||||
FI_ Slice_MipsCode ac_gte_sqr_v3s4(AtomBuilder_R ab, Reg r_sx, Reg r_sy, Reg r_sz, MipsCode delay_slot)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
gte_mv_to_data_r(r_sx, C2_IR1),
|
||||
gte_mv_to_data_r(r_sy, C2_IR2),
|
||||
gte_mv_to_data_r(r_sz, C2_IR3),
|
||||
delay_slot, gte_cmdw_sqr,
|
||||
})
|
||||
|
||||
/* ─── STAGE 4 of normalize: mtc2 IR0..3 + GPF + mfc2 MAC + srav finalize ───
|
||||
* Reusable standalone — given an IR0 = 1/|v| estimate (typically from a sqrtbl lookup) and a shift count
|
||||
* (typically (31 - LZCR)/2), multiplies IR0*IR[i] via GPF and shifts right to produce the normalized output.
|
||||
* Used standalone for "scale vector by scalar".
|
||||
* Words: 11. Clobbers: IR0..3, MAC1..3. Uses gte_cmdw_gpf (sf=0, lm=0). */
|
||||
FI_ Slice_MipsCode ac_gte_gpf_scale(AtomBuilder_R ab,
|
||||
U4 r_sx, U4 r_sy, U4 r_sz,
|
||||
U4 r_recip_est, U4 r_shift,
|
||||
U4 r_dx, U4 r_dy, U4 r_dz)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
gte_mv_to_data_r(r_recip_est, C2_IR0),
|
||||
gte_mv_to_data_r(r_sx, C2_IR1),
|
||||
gte_mv_to_data_r(r_sy, C2_IR2),
|
||||
gte_mv_to_data_r(r_sz, C2_IR3),
|
||||
GteDelay_ nop2, /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */
|
||||
gte_cmdw_gpf,
|
||||
gte_mv_from_data_r(r_dx, C2_MAC1),
|
||||
gte_mv_from_data_r(r_dy, C2_MAC2),
|
||||
gte_mv_from_data_r(r_dz, C2_MAC3),
|
||||
shift_aright_var(r_dx, r_dx, r_shift),
|
||||
shift_aright_var(r_dy, r_dy, r_shift),
|
||||
shift_aright_var(r_dz, r_dz, r_shift),
|
||||
})
|
||||
|
||||
/* ─── TRANS MATRIX (libgte TransMatrix port) ───
|
||||
* Atom component — auto-generates mac_trans_matrix Mac composer macro.
|
||||
* m->t = v (struct copy; libgte's TransMatrix at 0x8001a540 is just 3 store_words, no GTE, no add).
|
||||
* Uses 1 GPR (r_t1 = off value) per axis; per-axis load-delay-slot pattern.
|
||||
* Words: 9. Clobbers: r_t1. */
|
||||
FI_ Slice_MipsCode ac_trans_mt3s3s4(AtomBuilder_R ab
|
||||
, U4 r_mtx, U4 r_off
|
||||
, U4 r_t0, U4 r_t1, U4 r_t2
|
||||
) MipsAtomComp_Proc_(ab, {
|
||||
load_word( r_t0, r_off, O_(V3_S4,x)),
|
||||
load_word( r_t1, r_off, O_(V3_S4,y)),
|
||||
load_word( r_t2, r_off, O_(V3_S4,z)),
|
||||
store_word(r_t0, r_mtx, O_(MT3_S2S4,t[0])),
|
||||
store_word(r_t1, r_mtx, O_(MT3_S2S4,t[1])),
|
||||
store_word(r_t2, r_mtx, O_(MT3_S2S4,t[2])),
|
||||
})
|
||||
|
||||
/* ─── LZCR ROUND EVEN + HALF-SHIFT ───
|
||||
* Takes the raw LZCR leading-zero/ones count (from mfc2 C2_LZCR, range 1..32 per PSX-SPX cop2r31) and the |v|² sum (in r_mag_sq from the MAC1+MAC2+MAC3 add).
|
||||
* Produces:
|
||||
* r_shift ← LZCR rounded down to even (clear bit 0)
|
||||
* r_mag_sq_copy ← |v|² sum (moved out of r_mag_sq before it's overwritten)
|
||||
* r_mag_sq ← (31 - even_LZCR) / 2 = the final srav/GPF shift amount
|
||||
*
|
||||
* Rounding to even ensures (31 - LZCR) is always odd, so the >> 1 division is consistent — no 0.5 loss.
|
||||
* The caller branches on LZCR < 24 to decide left-shift vs right-shift of r_mag_sq_copy, then saves the shift count.
|
||||
*
|
||||
* Note: C2_LZCR (cop2r31) is a fixed read-only C2 data register — the caller must read it via mfc2 from C2_LZCR;
|
||||
* there is no register choice at the hardware level. Only the GPR that holds the result is caller-determined. */
|
||||
FI_ Slice_MipsCode ac_lzcr_round_even_half_shift(AtomBuilder_R ab,
|
||||
U4 r_shift,
|
||||
U4 r_mag_sq,
|
||||
U4 r_mag_sq_copy)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
and_i(r_shift, r_shift, gte_lzcr_even_mask),
|
||||
or_u(r_mag_sq_copy, r_mag_sq, 0),
|
||||
li_s( r_mag_sq, 31),
|
||||
sub_s( r_mag_sq, r_mag_sq, r_shift),
|
||||
shift_aright(r_mag_sq, r_mag_sq, 1),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_gte_general_purpose_interopolation(AtomBuilder_R ab
|
||||
, Reg to_ir0, Reg to_ir1, Reg to_ir2, Reg to_ir3
|
||||
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3
|
||||
, MipsCode nop_slot1, MipsCode nop_slot2)
|
||||
MipsAtomComp_Proc_(ab, {
|
||||
gte_mv_to_data_r(to_ir0, C2_IR0),
|
||||
gte_mv_to_data_r(to_ir1, C2_IR1), /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */
|
||||
gte_mv_to_data_r(to_ir2, C2_IR2),
|
||||
gte_mv_to_data_r(to_ir3, C2_IR3), /* IR3 = src.z (reloaded) */
|
||||
GteDelay_ nop_slot1,
|
||||
GteDelay_ nop_slot2,
|
||||
gte_cmdw_gpf,
|
||||
gte_mv_from_data_r(fr_mac1, C2_MAC1),
|
||||
gte_mv_from_data_r(fr_mac2, C2_MAC2),
|
||||
gte_mv_from_data_r(fr_mac3, C2_MAC3),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_gte_mv_from_data_r_mac123(AtomBuilder_R ab
|
||||
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3)
|
||||
MipsAtomComp_Proc_(ab, {
|
||||
gte_mv_from_data_r(fr_mac1, C2_MAC1),
|
||||
gte_mv_from_data_r(fr_mac2, C2_MAC2),
|
||||
gte_mv_from_data_r(fr_mac3, C2_MAC3),
|
||||
})
|
||||
|
||||
|
||||
FI_ Slice_MipsCode ac_gte_mv_from_mac123_v3s4(AtomBuilder_R ab, Reg_(V3_S4) v) MipsAtomComp_ProcMap_(ab, mac_gte_mv_from_data_r_mac123(v.x, v.y, v.z))
|
||||
|
||||
#pragma endregion MACs (Mips Atom Components)
|
||||
|
||||
#pragma region Atom Procs
|
||||
|
||||
/* ─── Local copy of PSYQ's sqrtbl (1/sqrt lookup table for VectorNormal). ───
|
||||
* Source: PSYQ 4.7 libgte sqrtbl at 0x800185B4 in hello_camera.elf.
|
||||
* objdump -s --start-address=0x800185B4 --stop-address=0x800185F4 hello_camera.elf → 192 entries × 16-bit signed, in 1.12 fixed-point (max value 0x1000 = 1.0).
|
||||
*
|
||||
* Data is identical to the libgte original (byte-for-byte verified).
|
||||
*
|
||||
* ─── Per-entry semantics (decoded from libgte msc02 VectorNormal) ───
|
||||
* Each entry is `1/sqrt(x)` in 1.12 fixed point (value / 4096).
|
||||
* The 192 entries span 4 octaves of the input magnitude, with 48 entries per octave:
|
||||
* Octave 0 (entries 0- 47): mantissa in [0x8000, 0x10000) output ~[1.000, 0.707]
|
||||
* Octave 1 (entries 48- 95): mantissa in [0x10000, 0x20000) output ~[0.707, 0.500]
|
||||
* Octave 2 (entries 96-143): mantissa in [0x20000, 0x40000) output ~[0.500, 0.354]
|
||||
* Octave 3 (entries144-191): mantissa in [0x40000, 0x80000) output ~[0.354, 0.251]
|
||||
* Within each octave, 8 sub-entries interpolate over the 8 fractional bits of the mantissa
|
||||
* (the byte `(0x80 | (i mod 8))` for the lower-byte of the aligned value).
|
||||
* Sampling the first value of each octave:
|
||||
* [0] 0x1000 = 1.0000 ; 1 / sqrt(1.0000)
|
||||
* [48] 0x0e4f = 0.8940 ; 1 / sqrt(1.2500)
|
||||
* [96] 0x0d10 = 0.8164 ; 1 / sqrt(1.5000)
|
||||
* [144] 0x0c0a = 0.7520 ; 1 / sqrt(1.7500)
|
||||
* And representative sub-entries within octave 0 (mantissa in [0x8000, 0x8100)):
|
||||
* [0] 0x1000 = 1.0000 ; 1 / sqrt(0x8000)
|
||||
* [1] 0x0fe0 = 0.9922 ; 1 / sqrt(0x8100)
|
||||
* [2] 0x0fc1 = 0.9846 ; 1 / sqrt(0x8200)
|
||||
* [3] 0x0fa3 = 0.9773 ; 1 / sqrt(0x8300)
|
||||
* [4] 0x0f85 = 0.9700 ; 1 / sqrt(0x8400)
|
||||
* [5] 0x0f68 = 0.9629 ; 1 / sqrt(0x8500)
|
||||
* [6] 0x0f4c = 0.9561 ; 1 / sqrt(0x8600)
|
||||
* [7] 0x0f30 = 0.9492 ; 1 / sqrt(0x8700)
|
||||
*
|
||||
* The algorithm's `addi -64 / sll 1 / lh` selects the entry at `(aligned - 64) * 2` for the case where `aligned` has its top bit at bit 24.
|
||||
* After the sllv/srav pair, `aligned` always lands in `[0x80, 0x100)`
|
||||
* (with top bit at bit 24 → after `sub $aligned - 64`, the index sits in `[0x40, 0x80) * 2 = [0x80, 0x100)` bytes = entries [64, 128) within the sqrtbl).
|
||||
* The earlier 64 entries (octave 0) are reached when the magnitude after shifting puts the top bit below bit 24 (the `sllv` branch),
|
||||
* and the load upper_halves of the table bracket the input range.
|
||||
* The later 64 entries (octaves 2-3) are the `srav` branch when the magnitude's top bit is well above bit 24.
|
||||
*
|
||||
* Reproduced verbatim from libgte (verified against libpsn00b/psxgte/vector.s:100-123 — 24 rows × 8 halfwords, last entry 0x0804).
|
||||
* */
|
||||
internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
|
||||
0x1000, 0x0fe0, 0x0fc1, 0x0fa3, 0x0f85, 0x0f68, 0x0f4c, 0x0f30,
|
||||
0x0f15, 0x0efb, 0x0ee1, 0x0ec7, 0x0eae, 0x0e96, 0x0e7e, 0x0e66,
|
||||
0x0e4f, 0x0e38, 0x0e22, 0x0e0c, 0x0df7, 0x0de2, 0x0dcd, 0x0db9,
|
||||
0x0da5, 0x0d91, 0x0d7e, 0x0d6b, 0x0d58, 0x0d45, 0x0d33, 0x0d21,
|
||||
0x0d10, 0x0cff, 0x0cee, 0x0cdd, 0x0ccc, 0x0cbc, 0x0cac, 0x0c9c,
|
||||
0x0c8d, 0x0c7d, 0x0c6e, 0x0c5f, 0x0c51, 0x0c42, 0x0c34, 0x0c26,
|
||||
0x0c18, 0x0c0a, 0x0bfd, 0x0bef, 0x0be2, 0x0bd5, 0x0bc8, 0x0bbb,
|
||||
0x0baf, 0x0ba2, 0x0b96, 0x0b8a, 0x0b7e, 0x0b72, 0x0b67, 0x0b5b,
|
||||
0x0b50, 0x0b45, 0x0b39, 0x0b2e, 0x0b24, 0x0b19, 0x0b0e, 0x0b04,
|
||||
0x0af9, 0x0aef, 0x0ae5, 0x0adb, 0x0ad1, 0x0ac7, 0x0abd, 0x0ab4,
|
||||
0x0aaa, 0x0aa1, 0x0a97, 0x0a8e, 0x0a85, 0x0a7c, 0x0a73, 0x0a6a,
|
||||
0x0a61, 0x0a59, 0x0a50, 0x0a47, 0x0a3f, 0x0a37, 0x0a2e, 0x0a26,
|
||||
0x0a1e, 0x0a16, 0x0a0e, 0x0a06, 0x09fe, 0x09f6, 0x09ef, 0x09e7,
|
||||
0x09e0, 0x09d8, 0x09d1, 0x09c9, 0x09c2, 0x09bb, 0x09b4, 0x09ad,
|
||||
0x09a5, 0x099e, 0x0998, 0x0991, 0x098a, 0x0983, 0x097c, 0x0976,
|
||||
0x096f, 0x0969, 0x0962, 0x095c, 0x0955, 0x094f, 0x0949, 0x0943,
|
||||
0x093c, 0x0936, 0x0930, 0x092a, 0x0924, 0x091e, 0x0918, 0x0912,
|
||||
0x090d, 0x0907, 0x0901, 0x08fb, 0x08f6, 0x08f0, 0x08eb, 0x08e5,
|
||||
0x08e0, 0x08da, 0x08d5, 0x08cf, 0x08ca, 0x08c5, 0x08bf, 0x08ba,
|
||||
0x08b5, 0x08b0, 0x08ab, 0x08a6, 0x08a1, 0x089c, 0x0897, 0x0892,
|
||||
0x088d, 0x0888, 0x0883, 0x087e, 0x087a, 0x0875, 0x0870, 0x086b,
|
||||
0x0867, 0x0862, 0x085e, 0x0859, 0x0855, 0x0850, 0x084c, 0x0847,
|
||||
0x0843, 0x083e, 0x083a, 0x0836, 0x0831, 0x082d, 0x0829, 0x0824,
|
||||
0x0820, 0x081c, 0x0818, 0x0814, 0x0810, 0x080c, 0x0808, 0x0804,
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_normalize_v3s4) {
|
||||
U2 src_offset; /* offset of src V3_S4 within the BIOS scratchpad */
|
||||
U2 dst_offset; /* offset of dst V3_S4 within the BIOS scratchpad */
|
||||
};
|
||||
typedef Struct_(RegUse_normalize_v3s4) {
|
||||
union { Reg_(V3_S4) res, src; };
|
||||
union { Reg r0, src_ptr, mac2; };
|
||||
union { Reg r1, dst_ptr; };
|
||||
union { Reg r2, dst_offset, mac1, v_sqr_aligned; };
|
||||
union { Reg r3, src_offset, btarget, shift_count, sqrtbl_index; };
|
||||
union { Reg r4, mac3, v_sqr_sum, scale_exp, srav_shift; };
|
||||
union { Reg r5, lzcr, inv_len; };
|
||||
};
|
||||
/* ─── Full normalize (all 4 stages inline) ───
|
||||
* Generic 4-stage GTE normalize (SQR → sum+LZCR → align+sqrtbl → GPF+srav). */
|
||||
internal MipsAtom* normalize_v3s4(AtomArena_R aa, RegUse_normalize_v3s4 r)
|
||||
MipsAtom_Proc_(aa, {
|
||||
load_half(r.src_offset, R_TapePtr, O_(Binds_normalize_v3s4, src_offset)),
|
||||
load_half(r.dst_offset, R_TapePtr, O_(Binds_normalize_v3s4, dst_offset)),
|
||||
LdSlot_ add_u(r.src_ptr, R_ScratchBase, r.src_offset),
|
||||
LdSlot_ add_u(r.dst_ptr, R_ScratchBase, r.dst_offset),
|
||||
LdSlot_ add_ui_self(R_TapePtr, S_(Binds_normalize_v3s4)),
|
||||
|
||||
mac_load_v3s4(r.src, r.src_ptr, 0),
|
||||
|
||||
/* Stage 1: mtc2 src → IR1/2/3, SQR fires. */
|
||||
LdSlot_ mac_gte_sqr_v3s4(r.src.x, r.src.y, r.src.z, LdSlot_ nop),
|
||||
|
||||
/* Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS. src_ptr is dead; reuse as mac2. */
|
||||
mac_gte_mv_from_data_r_mac123(r.mac1, r.mac2, r.mac3), LdSlot_ nop,
|
||||
add_u_self( r.v_sqr_sum, r.mac1),
|
||||
add_u_self( r.v_sqr_sum, r.mac2),
|
||||
gte_mv_to_data_r( r.v_sqr_sum, C2_LZCS), GteDelay_ nop2,
|
||||
gte_mv_from_data_r(r.lzcr, C2_LZCR), GteDelay_ nop,
|
||||
|
||||
/* Stage 3: even(LZCR), half-shift, align |v|² to bit 24. */
|
||||
mac_lzcr_round_even_half_shift(r.lzcr, r.v_sqr_sum, r.v_sqr_aligned),
|
||||
add_si( r.btarget, r.lzcr, -24),
|
||||
branch_lt_zero(r.btarget, atom_offset(aligned_done, srav_path)), BdSlot_ nop, /* bltz → srav_path (LZCR < 24 path) */
|
||||
jump_rel(atom_offset(srav_path, aligned_done)), /* b → aligned_done (LZCR >= 24 path) */
|
||||
BdSlot_ shift_lleft_var(r.v_sqr_aligned, r.v_sqr_aligned, r.btarget),
|
||||
atom_label(srav_path)
|
||||
li_s( r.shift_count, 24),
|
||||
sub_s(r.shift_count, r.shift_count, r.lzcr),
|
||||
shift_aright_var(r.v_sqr_aligned, r.v_sqr_aligned, r.shift_count),
|
||||
atom_label(aligned_done)
|
||||
add_si( r.v_sqr_aligned, r.v_sqr_aligned, -64),
|
||||
shift_lleft(r.v_sqr_aligned, r.v_sqr_aligned, 1),
|
||||
mac_load_word_imm(r.sqrtbl_index, & gte_normalize_sqr_tbl), add_u_self(r.sqrtbl_index, r.v_sqr_aligned),
|
||||
load_half(r.inv_len, r.sqrtbl_index, 0),
|
||||
LdSlot_ nop,
|
||||
|
||||
mac_gte_general_purpose_interopolation(r.inv_len,
|
||||
r.src.x, r.src.y, r.src.z,
|
||||
r.res.x, r.res.y, r.res.z,
|
||||
GteDelay_ load_word(R_AtomJmp, R_TapePtr, 0), LdSlot_ // ac_yield: word 1
|
||||
GteDelay_ add_ui_self( R_TapePtr, S_(MipsCode)) // ac_yield: word 2
|
||||
),
|
||||
mac_shift_aright_var_v3s4_self(r.res, r.srav_shift),
|
||||
mac_store_v3s4(r.res, r.dst_ptr, 0),
|
||||
|
||||
jump_reg(R_AtomJmp), BdSlot_ nop // ac_yield: word 3-4
|
||||
})
|
||||
|
||||
|
||||
/* ─── GTE OP cross product (a × b → out) ───
|
||||
* Generalized V3_S4 cross product via GTE OP (OuterProduct12 libpsyx convention).
|
||||
* The >> 12 shift converts S12.20 → S12.0 OuterProduct12. */
|
||||
typedef Struct_(Binds_gte_cross_v3s4) { V3_S4* src_a; V3_S4* src_b; V3_S4* out; };
|
||||
typedef Struct_(RegUse_gte_cross_v3s4) {
|
||||
Reg_(V3_S4) a;
|
||||
Reg_(V3_S4) b;
|
||||
union { Reg out, t0; } x;
|
||||
union { Reg src_a, t1, rt11; } y;
|
||||
union { Reg src_b, t2, rt22; } z;
|
||||
};
|
||||
internal MipsAtom* gte_cross_v3s4(AtomArena_R aa, RegUse_gte_cross_v3s4 r)
|
||||
atom_info(atom_bind(Binds_gte_cross_v3s4)) MipsAtom_Proc_(aa, {
|
||||
load_word(r.y.src_a, R_TapePtr, O_(Binds_gte_cross_v3s4,src_a)),
|
||||
load_word(r.z.src_b, R_TapePtr, O_(Binds_gte_cross_v3s4,src_b)),
|
||||
load_word(r.x.out, R_TapePtr, O_(Binds_gte_cross_v3s4,out)),
|
||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_gte_cross_v3s4)),
|
||||
|
||||
mac_load_v3s4(r.a, r.y.src_a, 0), LdSlot_
|
||||
mac_load_v3s4(r.b, r.z.src_b, 0), LdSlot_
|
||||
mac_gte_op_cross_v3s4(r.a, r.b), /* RT diagonal + IR + OP + MAC read + shift */
|
||||
mac_store_v3s4(r.a, r.x.out, 0),
|
||||
|
||||
mac_yield()
|
||||
})
|
||||
#pragma endregion Atom Procs
|
||||
|
||||
#pragma region Baked Atoms
|
||||
|
||||
typedef Struct_(Binds_SetGteMT3S2S4) {
|
||||
MT3_S2S4* transform;
|
||||
};
|
||||
internal MipsAtom_(set_gte_mt3s2s4) atom_info(
|
||||
atom_bind(Binds_SetGteMT3S2S4)
|
||||
, atom_reads(R_TapePtr)
|
||||
){
|
||||
/* Pop matrix address from tape into R_T3 ($11) */
|
||||
load_word(R_T3, R_TapePtr, O_(Binds_SetGteMT3S2S4,transform)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_SetGteMT3S2S4)),
|
||||
/* Load 3x3 Rotation + 3x1 Translation from R_T3 into GTE CONTROL Regs (ctc2) */
|
||||
load_word(R_T0, R_T3, 0),
|
||||
load_word(R_T1, R_T3, 4),
|
||||
gte_mv_to_ctrl_r(R_T0, gte_cr_RT11),
|
||||
gte_mv_to_ctrl_r(R_T1, gte_cr_RT12),
|
||||
load_word(R_T0, R_T3, 8),
|
||||
load_word(R_T1, R_T3, 12),
|
||||
load_word(R_T2, R_T3, 16),
|
||||
gte_mv_to_ctrl_r(R_T0, gte_cr_RT13),
|
||||
gte_mv_to_ctrl_r(R_T1, gte_cr_RT21),
|
||||
gte_mv_to_ctrl_r(R_T2, gte_cr_RT22),
|
||||
load_word(R_T0, R_T3, 20),
|
||||
load_word(R_T1, R_T3, 24),
|
||||
load_word(R_T2, R_T3, 28),
|
||||
gte_mv_to_ctrl_r(R_T0, gte_cr_TRX),
|
||||
gte_mv_to_ctrl_r(R_T1, gte_cr_TRY),
|
||||
gte_mv_to_ctrl_r(R_T2, gte_cr_TRZ),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
#pragma endregion Baked Atoms
|
||||
+246
-247
@@ -16,10 +16,6 @@
|
||||
* gte_mv_to_data_r (gte + mv + to + data + register)
|
||||
* gte_lw_v0_xy(base) (gte + lw + v0 + xy)
|
||||
* load_upper_i (load-upper + immediate, unique verb)
|
||||
*
|
||||
* Vendor mnemonics (gte_mtc2, gte_mfc2, gte_lwc2, gte_swc2, etc.) are
|
||||
* NOT in this header. They live in the opt-in `gte_vendor_sym.h` for
|
||||
* users who prefer the textbook MIPS assembly mnemonics.
|
||||
* ============================================================================ */
|
||||
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
@@ -34,76 +30,27 @@
|
||||
* gte.h — Geometry Transformation Engine (COP2) for the PS1
|
||||
* ============================================================================
|
||||
*
|
||||
* Hand-rolled DSL for emitting GTE/MIPS instruction words as raw `.word`
|
||||
* constants from C. No GCC inline-assembly string syntax in the code body.
|
||||
*
|
||||
* PHILOSOPHY
|
||||
* ----------
|
||||
* 1. A 32-bit instruction word is composed from per-field encoders. Each
|
||||
* encoder knows only its own bit range; the composite ORs them together.
|
||||
* No magic numbers inside any encoder body. Every shift and mask is a
|
||||
* named constant from the bitfield-layout enum below.
|
||||
*
|
||||
* 2. Pure (compile-time) instructions. Every GTE *command* (RTPS, RTPT,
|
||||
* NCLIP, MVMVA, …) and every COP2 *transfer* (ctc2/cfc2) with a constant
|
||||
* rs/rt/rd — are emitted as a single integer constant via
|
||||
* `asm_inline(...)` from gcc_asm.h. The C compiler constant-folds
|
||||
* these into `.word` directives in .rodata.
|
||||
*
|
||||
* 3. Runtime-base-register instructions (lwc2, swc2, lw, sw, …) cannot be
|
||||
* a pure compile-time word because the `rs` field is chosen by the
|
||||
* compiler at codegen. For these we use a "placeholder-pun" pattern:
|
||||
* a fixed register number (R_T4 = $12) is baked into the rs field of
|
||||
* the `.word` constant, and the macro declares a `"r"(arg)` input
|
||||
* constraint plus a clobber on the same register. The compiler is
|
||||
* therefore *forced* to bind `arg` to that exact register, and the
|
||||
* constant is correct.
|
||||
*
|
||||
* USAGE
|
||||
* -----
|
||||
* // Pure command sequence — all bits compile-time:
|
||||
* asm volatile(
|
||||
* asm_inline( gte_cmd_rtpt , gte_cmd_nclip , gte_cmd_avsz3 )
|
||||
* asm_clobber( clbr_volatile_gprs )
|
||||
* );
|
||||
*
|
||||
* // Runtime-base-register load — caller picks the base GPR:
|
||||
* register V3_S2* p_in_12 __asm__("$12") = verts[0].ptr;
|
||||
* gte_load_v0(p_in_12, R_T4); // R_T4 = 12 = $t4 = $12
|
||||
*
|
||||
* // Three independent bases for an RTPT pipeline:
|
||||
* register V3_S2* p0 gcc_reg(R_T4) = verts[0].ptr;
|
||||
* register V3_S2* p1 gcc_reg(R_T5) = verts[1].ptr;
|
||||
* register V3_S2* p2 gcc_reg(R_T6) = verts[2].ptr;
|
||||
* gte_load_v0(p0, R_T4);
|
||||
* gte_load_v1(p1, R_T5);
|
||||
* gte_load_v2(p2, R_T6);
|
||||
* gte_rtpt();
|
||||
* Hand-rolled DSL for emitting GTE/MIPS instruction words from C.
|
||||
* No GCC inline-assembly string syntax in the code body.
|
||||
*
|
||||
* STYLE NOTES
|
||||
* -----------
|
||||
* - Per-field encoders are named `enc_gte_<field>(value)` and each one
|
||||
* self-masks its argument before shifting. Mirrors the `enc_op / enc_rs
|
||||
* / enc_rt / ...` family in mips.h.
|
||||
* - The composite `enc_gte_cmdw(sf, mx, v, cv, lm, cmd)` is a flat OR of
|
||||
* the per-field encoders, plus the COP2/CO base.
|
||||
* - Pre-baked shortcuts (`gte_cmd_rtpt`, `gte_cmd_rtps`, …) are defined
|
||||
* for the common cases so call sites read like assembly source.
|
||||
* - All register/field values are enums (not `#define`s) so they show up
|
||||
* in debugger symbol tables and IDE autocomplete.
|
||||
* - Per-field encoders are named `enc_gte_<field>(value)` and each one self-masks its argument before shifting.
|
||||
* Mirrors the `enc_op / enc_rs / enc_rt / ...` family in mips.h.
|
||||
* - The composite `enc_gte_cmdw(sf, mx, v, cv, lm, cmd)` is a flat OR of the per-field encoders, plus the COP2/CO base.
|
||||
* - Pre-baked shortcuts (`gte_cmd_rtpt`, `gte_cmd_rtps`, …) are defined for the common cases so call sites read like assembly source.
|
||||
* - All register/field values are enums (not `#define`s) so they show up in debugger symbol tables and IDE autocomplete.
|
||||
*
|
||||
* SEE ALSO
|
||||
* --------
|
||||
* - gcc_asm.h: the `.word` emitter (`asm_inline`, `asm_clobber`, clobbers)
|
||||
* - mips.h: the MIPS encoder layer this builds on
|
||||
* - mips.h: The MIPS encoder layer this builds on.
|
||||
*/
|
||||
|
||||
/* C2 data registers */
|
||||
|
||||
/* --- GTE Data Registers (Coprocessor 2) ---
|
||||
* Preprocessor-visible integer ids for the COP2 data register file.
|
||||
* Each enum value is bound to a parallel `_Code` `#define` so the
|
||||
* preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
||||
* Each enum value is bound to a parallel `_Code` `#define` so the preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
||||
* Same pattern as the GPR `_Code` set in mips.h. */
|
||||
#define C2_VXY0_Code 0
|
||||
#define C2_VZ0_Code 1
|
||||
@@ -151,20 +98,20 @@ enum {
|
||||
|
||||
/* Semantic Aliases for GTE Data Registers */
|
||||
enum {
|
||||
gte_in_v0_xy = C2_VXY0, /* Input Vector 0 (X, Y) */
|
||||
gte_in_v0_z = C2_VZ0, /* Input Vector 0 (Z) */
|
||||
gte_in_v1_xy = C2_VXY1, /* Input Vector 1 (X, Y) */
|
||||
gte_in_v1_z = C2_VZ1, /* Input Vector 1 (Z) */
|
||||
gte_in_v2_xy = C2_VXY2, /* Input Vector 2 (X, Y) */
|
||||
gte_in_v2_z = C2_VZ2, /* Input Vector 2 (Z) */
|
||||
gte_in_rgb = C2_RGB, /* Input Color (R, G, B, MipsCode) */
|
||||
gte_out_scr_xy0 = C2_SXY0, /* Output Screen Coord 0 (X, Y) */
|
||||
gte_out_scr_xy1 = C2_SXY1, /* Output Screen Coord 1 (X, Y) */
|
||||
gte_out_scr_xy2 = C2_SXY2, /* Output Screen Coord 2 (X, Y) */
|
||||
gte_out_depth = C2_OTZ, /* Output Ordering Table Z (Depth) */
|
||||
gte_math_accum0 = C2_MAC0, /* Math Accumulator 0 */
|
||||
gte_math_accum1 = C2_MAC1, /* Math Accumulator 1 */
|
||||
gte_math_accum2 = C2_MAC2, /* Math Accumulator 2 */
|
||||
C2_InV0_XY = C2_VXY0, /* Input Vector 0 (X, Y) */
|
||||
C2_InV0_Z = C2_VZ0, /* Input Vector 0 (Z) */
|
||||
C2_InV1_XY = C2_VXY1, /* Input Vector 1 (X, Y) */
|
||||
C2_InV1_Z = C2_VZ1, /* Input Vector 1 (Z) */
|
||||
C2_InV2_XY = C2_VXY2, /* Input Vector 2 (X, Y) */
|
||||
C2_InV2_Z = C2_VZ2, /* Input Vector 2 (Z) */
|
||||
C2_In_RGB = C2_RGB, /* Input Color (R, G, B, MipsCode) */
|
||||
C2_OutSrc_XY0 = C2_SXY0, /* Output Screen Coord 0 (X, Y) */
|
||||
C2_OutSrc_XY1 = C2_SXY1, /* Output Screen Coord 1 (X, Y) */
|
||||
C2_OutSrc_XY2 = C2_SXY2, /* Output Screen Coord 2 (X, Y) */
|
||||
C2_OutDepth = C2_OTZ, /* Output Ordering Table Z (Depth) */
|
||||
C2_MathAccu0 = C2_MAC0, /* Math Accumulator 0 */
|
||||
C2_MathAccu1 = C2_MAC1, /* Math Accumulator 1 */
|
||||
C2_MathAccu2 = C2_MAC2, /* Math Accumulator 2 */
|
||||
};
|
||||
|
||||
/* --- GTE Command Semantics (The Bitfield Meanings) ---
|
||||
@@ -211,6 +158,8 @@ enum {
|
||||
gte_cmd_nclip = 0x06, /* Normal Clipping (Backface culling) */
|
||||
gte_cmd_op = 0x0C, /* Outer Product */
|
||||
gte_cmd_mvmva = 0x12, /* Matrix Vector Multiply & Add (Custom math) */
|
||||
gte_cmd_sqr = 0x28, /* Square vector — MAC[i] = IR[i]²; IR[i] ← MAC[i] saturated */
|
||||
gte_cmd_gpf = 0x3D, /* General-purpose Interpolation — MAC[i] = IR0 * IR[i] */
|
||||
|
||||
/* --- GTE Command Bit-Field Layout ---
|
||||
* A GTE command word (sent to COP2 with RS=1) is laid out as:
|
||||
@@ -221,25 +170,46 @@ enum {
|
||||
* +------------+--+-----+------+------+------+------+---+--------+----------+
|
||||
* \_____ GTE_PAYLOAD _____/ \__ GTE_CMD __/
|
||||
*
|
||||
* Shifts/masks below are the *bit positions* and *bit widths* of each
|
||||
* configurable field, used by the ENC_GTE_CMD encoder. Mirrors the
|
||||
* OPCODE_SHIFT / RS_SHIFT convention used in mips.h.
|
||||
* Offset position & masks below are the *bit positions* and *bit widths* of each configurable field, used by the ENC_GTE_CMD encoder.
|
||||
* Mirrors the OPCODE_POS / RS_POS convention used in mips.h.
|
||||
*/
|
||||
|
||||
gte_shift_sf = 19, gte_width_sf = 1, gte_mask_sf = 0x1,
|
||||
gte_shift_mx = 17, gte_width_mx = 2, gte_mask_mx = 0x3,
|
||||
gte_shift_v = 15, gte_width_v = 2, gte_mask_v = 0x3,
|
||||
gte_shift_cv = 13, gte_width_cv = 2, gte_mask_cv = 0x3,
|
||||
gte_shift_lm = 10, gte_width_lm = 1, gte_mask_lm = 0x1,
|
||||
gte_shift_cmd = 0, gte_width_cmd = 6, gte_mask_cmd = 0x3F,
|
||||
gte_pos_sf = 19, gte_width_sf = 1,
|
||||
gte_pos_mx = 17, gte_width_mx = 2,
|
||||
gte_pos_v = 15, gte_width_v = 2,
|
||||
gte_pos_cv = 13, gte_width_cv = 2,
|
||||
gte_pos_lm = 10, gte_width_lm = 1,
|
||||
gte_pos_cmd = 0, gte_width_cmd = 6,
|
||||
|
||||
/* Fake command number (bits 24-20) — IGNORED by the GTE hardware per PSX-SPX `geometrytransformationenginegte.md` line 48.
|
||||
* libgte's compiler emits non-zero values in this field as a disassembly signature. */
|
||||
gte_pos_fake_cmd = 20,
|
||||
gte_width_fake_cmd = 5,
|
||||
};
|
||||
|
||||
/* --- GTE Control Register Aliases (Pitfall 1) ---
|
||||
* Three pairs of aliases map to the C2 control-register slot:
|
||||
* C2[24] = gte_cr_RBK (background R) | gte_cr_OFX (screen offset X)
|
||||
* C2[25] = gte_cr_GBK (background G) | gte_cr_OFY (screen offset Y)
|
||||
* C2[26] = gte_cr_BBK (background B) | gte_cr_H (projection plane distance H)
|
||||
* Cross-alias writes inside one atom body, or across the wave-context boundary, silently clobber each other.
|
||||
* The metaprogram's check_gte_cr_alias_writes (CHECK_RULES row) warns about each pair per source.
|
||||
* See psx-spx docs/gte_reference.md §"Control-register alias table" for the silicon rationale and the libgte outer-product convention.
|
||||
*/
|
||||
|
||||
/* --- RT-matrix packed-slot convention (Pitfall 4) ---
|
||||
* The silicon packs two 16-bit RT elements per 32-bit C2 slot:
|
||||
* C2[2] = (RT22 << 16) | RT13 (gte_cr_RT13 writes the low half, gte_cr_RT22 writes the high half)
|
||||
* C2[4] = (RT33 << 16) | RT22 (gte_cr_RT22 writes the low half — clobbers prior RT22 value if RT13 was also written)
|
||||
* OP and MVMVA read D1/D2/D3 from these packed slots.
|
||||
* The libgte outer-product convention (see ac_apply_matrix_lv at gte.atom.c:108-122) writes C2[2] then C2[4] in sequence;
|
||||
* the SECOND write's low half is RT22, not RT13.
|
||||
*/
|
||||
|
||||
/* --- GTE Control Register Indices (for ctc2/cfc2) ---
|
||||
* Preprocessor-visible integer ids for the COP2 control register file.
|
||||
* Each enum value is bound to a parallel `_Code` `#define` so the
|
||||
* preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
||||
* Same pattern as the GPR `_Code` set in mips.h. Note: indices 21-23
|
||||
* are reserved/unused on real hardware, so there's a gap. */
|
||||
* Each enum value is bound to a parallel `_Code` `#define` so the preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
||||
* Same pattern as the GPR `_Code` set in mips.h. Note: indices 21-23 are reserved/unused on real hardware, so there's a gap. */
|
||||
#define gte_cr_RT11_Code 0
|
||||
#define gte_cr_RT12_Code 1 /* packed with RT13 in bits 16..31 */
|
||||
#define gte_cr_RT13_Code 2 /* packed with RT22 in bits 16..31 */
|
||||
@@ -267,8 +237,9 @@ enum {
|
||||
#define gte_cr_RFC_Code 27
|
||||
#define gte_cr_GFC_Code 28
|
||||
#define gte_cr_BFC_Code 29
|
||||
#define gte_cr_OFX_Code 30
|
||||
#define gte_cr_OFY_Code 31
|
||||
#define gte_cr_OFX_Code 24
|
||||
#define gte_cr_OFY_Code 25
|
||||
#define gte_cr_H_Code 26
|
||||
|
||||
enum {
|
||||
gte_cr_RT11 = gte_cr_RT11_Code, gte_cr_RT12 = gte_cr_RT12_Code, gte_cr_RT13 = gte_cr_RT13_Code,
|
||||
@@ -290,21 +261,16 @@ enum { _C2_OPS_ = 0
|
||||
|
||||
/* COP2 transfer sub-opcodes (5-bit field in the `rs` slot of enc_gte_tx).
|
||||
*
|
||||
* Spans the 2x2 {From, To} × {Data, Control} register classes that the
|
||||
* GTE exposes:
|
||||
*
|
||||
* Spans the 2x2 {From, To} × {Data, Control} register classes that the GTE exposes:
|
||||
* bit 1 (0x02): register class — 0 = data, 1 = control
|
||||
* bit 2 (0x04): direction — 0 = read, 1 = write
|
||||
*
|
||||
* The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit
|
||||
* numbers as the general MIPS `cop_mf` / `cop_mt` defined in mips.h
|
||||
* (which target the data register file on any coprocessor). They are
|
||||
* re-aliased here so the four-way table reads like the spec mnemonics
|
||||
* (MFC2 / CFC2 / MTC2 / CTC2) and so the encoding lives next to its
|
||||
* only consumer (this header).
|
||||
* The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit numbers as general MIPS `cop_mf` / `cop_mt` defined in mips.h
|
||||
* (which target the data register file on any coprocessor).
|
||||
* They are re-aliased here so the four-way table reads like the spec mnemonics (MFC2 / CFC2 / MTC2 / CTC2)
|
||||
* and so the encoding is next to its only consumer (this header).
|
||||
*
|
||||
* Vendor mnemonic aliases (gte_mfc2 / gte_mtc2 / gte_cfc2 / gte_ctc2)
|
||||
* live in gte_vendor_sym.h. */
|
||||
* Vendor mnemonic aliases (gte_mfc2 / gte_mtc2 / gte_cfc2 / gte_ctc2) live in gte_vendor_sym.h. */
|
||||
enum { _C2_TX_SUBS_ = 0
|
||||
, sub_mfc2 = 0x00 /* MFC2: Move From Coprocessor 2 data reg */
|
||||
, sub_cfc2 = 0x02 /* CFC2: Copy From Coprocessor 2 ctrl reg */
|
||||
@@ -314,11 +280,11 @@ enum { _C2_TX_SUBS_ = 0
|
||||
|
||||
/* COP2 (GTE) Transfer Format: mfc2 / cfc2 / mtc2 / ctc2 rt, rd
|
||||
* Layout: [op_cop2:6][sub:5][rt:5][rd:5][0:11]
|
||||
* - sub: one of sub_mfc2 / sub_cfc2 / sub_mtc2 / sub_ctc2
|
||||
* - rt: GPR source/dest
|
||||
* - rd: COP2 register index (0..31):
|
||||
* data class → C2_VXY0_Code..C2_LZCR_Code (gte_in_v0_xy..gte_math_accum2 aliases)
|
||||
* ctrl class → gte_cr_RT11_Code..gte_cr_OFY_Code */
|
||||
* - sub: one of sub_mfc2 / sub_cfc2 / sub_mtc2 / sub_ctc2
|
||||
* - rt: GPR source/dest
|
||||
* - rd: COP2 register index (0..31):
|
||||
* data class → C2_VXY0_Code..C2_LZCR_Code (gte_in_v0_xy..gte_math_accum2 aliases)
|
||||
* ctrl class → gte_cr_RT11_Code..gte_cr_OFY_Code */
|
||||
#define enc_gte_tx(sub, rt, rd) (enc_op(op_cop2) | enc_rs(sub) | enc_rt(rt) | enc_rd(rd))
|
||||
|
||||
|
||||
@@ -326,8 +292,7 @@ enum { _C2_TX_SUBS_ = 0
|
||||
// #define gte_mv_from_data_r(rt, rd) enc_gte_tx(cop_mf, (rt), (rd)) /* Move GTE Control Register (rd) to GPR (rt) */
|
||||
|
||||
/* GTE Data vs Control Register Transfers
|
||||
*
|
||||
* Each macro emits a single .word constant for one of MFC2/CFC2/MTC2/CTC2.
|
||||
* Each macro emits a single instruction for one of MFC2/CFC2/MTC2/CTC2.
|
||||
*
|
||||
* `rd` is the C2 register index in the file the sub-opcode names:
|
||||
* gte_mv_from_data_r / gte_mv_to_data_r → C2 data register file
|
||||
@@ -342,14 +307,14 @@ enum { _C2_TX_SUBS_ = 0
|
||||
#define gte_mv_from_ctrl_r(rt, rd) enc_gte_tx(sub_cfc2, (rt), (rd)) /* Copy From ctrl reg */
|
||||
#define gte_mv_to_data_r(rt, rd) enc_gte_tx(sub_mtc2, (rt), (rd)) /* Move To data reg */
|
||||
#define gte_mv_to_ctrl_r(rt, rd) enc_gte_tx(sub_ctc2, (rt), (rd)) /* Copy To ctrl reg */
|
||||
#define GteDelay_ // Annotate an instruction as filling a CPU <-> GTE DMA delay slot/s
|
||||
|
||||
/* COP2 Data Load (lwc2): `lwc2 rt, off(rs)`
|
||||
* Layout: [op_lwc2:6][rs:5][rt:5][imm:16]
|
||||
* - rs: GPR base address
|
||||
* - rt: COP2 data register index (0..31)
|
||||
* - imm: signed 16-bit offset
|
||||
* NOTE: When `rs` is a runtime register, the encoding cannot be pre-baked
|
||||
* into a .word — use the string-style `gte_load_v0` macro below instead. */
|
||||
* NOTE: When `rs` is a runtime register, the encoding cannot be pre-baked into a .word — use the string-style `gte_load_v0` macro below instead. */
|
||||
#define enc_gte_lw(rt, base, off) enc_i(op_lwc2, (base), (rt), (off))
|
||||
/* Store Word */
|
||||
#define enc_gte_sw(rt, base, off) enc_i(op_swc2, (base), (rt), (off))
|
||||
@@ -358,31 +323,30 @@ enum { _C2_TX_SUBS_ = 0
|
||||
* `swc2` is redundant when we're already inside the `gte_` namespace.
|
||||
* gte_lw rt, base, off → lwc2 rt, off(base)
|
||||
* gte_sw rt, base, off → swc2 rt, off(base)
|
||||
* For the typical user-facing vector-level load (xy + z as two
|
||||
* instructions), use the higher-level `gte_load_vN` macros below. */
|
||||
* For the typical user-facing vector-level load (xy + z as two instructions), use the higher-level `gte_load_vN` macros below. */
|
||||
#define gte_lw(rt, base, off) enc_gte_lw(rt, base, off)
|
||||
#define gte_sw(rt, base, off) enc_gte_sw(rt, base, off)
|
||||
|
||||
/* GTE Command Format
|
||||
* Opcode is always MIPS_OP_COP2, RS is always 1 (CO).
|
||||
* The lower 25 bits are the GTE-specific command payload.
|
||||
* Lower 25 bits are GTE-specific command payload.
|
||||
*
|
||||
* The granular `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs`
|
||||
* pattern in mips.h: each one self-masks and shifts its own field, so a
|
||||
* caller can build up a GTE command piece by piece (handy for state-driven
|
||||
* MVMVA emitters that vary one field at a time).
|
||||
* The `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs` pattern in mips.h:
|
||||
* Each one self-masks and shifts its own field, so a caller can build up a GTE command piece by piece
|
||||
* (handy for state-driven MVMVA emitters that vary one field at a time).
|
||||
*
|
||||
* `ENC_GTE_CMD` is the all-in-one convenience for emitting a full command
|
||||
* word in one go. It just ORs the per-field encoders together. */
|
||||
* `ENC_GTE_CMD` is an all-in-one convenience for emitting a full command word.
|
||||
* It just ORs the per-field encoders together. */
|
||||
#define gte_cmd_base (enc_op(op_cop2) | (1 << 25))
|
||||
|
||||
/* Per-field encoders. Each one does (value & mask) << shift on its own. */
|
||||
#define enc_gte_sf(sf) (((sf) & gte_mask_sf ) << gte_shift_sf )
|
||||
#define enc_gte_mx(mx) (((mx) & gte_mask_mx ) << gte_shift_mx )
|
||||
#define enc_gte_v(v) (((v) & gte_mask_v ) << gte_shift_v )
|
||||
#define enc_gte_cv(cv) (((cv) & gte_mask_cv ) << gte_shift_cv )
|
||||
#define enc_gte_lm(lm) (((lm) & gte_mask_lm ) << gte_shift_lm )
|
||||
#define enc_gte_cmd(cmd) (((cmd) & gte_mask_cmd) << gte_shift_cmd)
|
||||
#define enc_gte_sf(sf) ((sf) << gte_pos_sf )
|
||||
#define enc_gte_mx(mx) ((mx) << gte_pos_mx )
|
||||
#define enc_gte_v(v) ((v) << gte_pos_v )
|
||||
#define enc_gte_cv(cv) ((cv) << gte_pos_cv )
|
||||
#define enc_gte_lm(lm) ((lm) << gte_pos_lm )
|
||||
#define enc_gte_cmd(cmd) ((cmd) << gte_pos_cmd )
|
||||
#define enc_gte_fake_cmd(x) ((x) << gte_pos_fake_cmd)
|
||||
|
||||
/* Composite: all six GTE fields + the COP2/CO base. */
|
||||
#define enc_gte_cmdw(sf, mx, v, cv, lm, cmd) ( \
|
||||
@@ -397,41 +361,35 @@ enum { _C2_TX_SUBS_ = 0
|
||||
|
||||
/* GTE command words for the common cases.
|
||||
*
|
||||
* These are pure compile-time integer constants — the C compiler
|
||||
* constant-folds them into `.word` directives in .rodata. Use them
|
||||
* inside `asm_inline(...)` blocks (see `gte_rtpt` below for the
|
||||
* canonical idiom).
|
||||
* These are pure compile-time integer constants — the C compiler constant-folds them into `.word` directives in .rodata.
|
||||
* Use them inside `asm_inline(...)` blocks (see `gte_rtpt` below for the idiom).
|
||||
*
|
||||
* Decomposition (per the `enc_gte_<field>` definitions above):
|
||||
* gte_cmdw_<name> = gte_cmd_base | enc_gte_cmd(<cmd>)
|
||||
|
||||
* The SF/MX/V/CV/LM fields are all zero in the common cases (standard
|
||||
* rotation-matrix, no scaling factor, V0 vector, translation vector,
|
||||
* no clamp), so the only varying bits are the `cmd` field.
|
||||
* The SF / MX / V / CV / LM fields are all zero in the common cases
|
||||
* (standard rotation-matrix, no scaling factor, V0 vector, translation vector, no clamp),
|
||||
* so the only varying bits are the `cmd` field.
|
||||
*
|
||||
* Naming follows the file's convention: `gte_cmd_*` is the raw
|
||||
* 6-bit `cmd` field id, `gte_cmdw_*` is the fully-encoded 32-bit
|
||||
* instruction word ready to drop into a `.word` directive.
|
||||
* Naming convention:
|
||||
* - `gte_cmd_*` : Raw 6-bit `cmd` field id
|
||||
* - `gte_cmdw_* : 32-bit instruction word ready to drop into a `.word` directive.
|
||||
*
|
||||
* --------------------------------------------------------------------------
|
||||
* PsyQ-compatibility note (RTPS/RTPT):
|
||||
* The original Sony PsyQ `inline_n.h` ships RTPT as `cop2 0x0280030` and
|
||||
* RTPS as `cop2 0x0180001`. Both have `0x20` set in the upper-reserved
|
||||
* region (bit 21) AND `sf=1` (bit 19) — i.e. the "no division" flag.
|
||||
* Per psx-spec these bits are reserved/must-be-zero, but the real GTE
|
||||
* hardware and PCSX-Redux's GTE model both IGNORE them on these two
|
||||
* commands (the perspective divide happens regardless of `sf`).
|
||||
* The original Sony PsyQ `inline_n.h` ships RTPT as `cop2 0x0280030` and RTPS as `cop2 0x0180001`.
|
||||
* Both have `0x20` set in the upper-reserved region (bit 21) AND `sf=1` (bit 19) — i.e. the "no division" flag.
|
||||
* Per psx-spec these bits are reserved/must-be-zero,
|
||||
* but the real GTE hardware and PCSX-Redux's GTE model both IGNORE them on these two commands
|
||||
* (the perspective divide happens regardless of `sf`).
|
||||
*
|
||||
* If we emit a strictly-spec-compliant word (`sf=0`, reserved bits
|
||||
* clear), PCSX-Redux's GTE checks those bits more strictly than the
|
||||
* silicon does and RTPT silently no-ops — the floor's screen
|
||||
* coordinates come out as raw projection-of-rotation (Z never
|
||||
* divided), `nclip` ends up wrong, and the triangle is culled.
|
||||
* If we emit a strictly-spec-compliant word (`sf=0`, reserved bits clear),
|
||||
* PCSX-Redux's GTE checks those bits more strictly than the silicon does and RTPT silently no-ops.
|
||||
* The floor's screen coordinates come out as raw projection-of-rotation (Z never divided),
|
||||
* `nclip` ends up wrong, and the triangle is culled.
|
||||
*
|
||||
* So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to
|
||||
* match the working bit pattern everyone has shipped for 25 years.
|
||||
* NCLIP/OP/MVMVA stay spec-clean — their reserved bits really are
|
||||
* zero in the original PsyQ source.
|
||||
* So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern.
|
||||
* NCLIP / OP / MVMVA stay spec-clean — their reserved bits really are zero in the original PsyQ source.
|
||||
* --------------------------------------------------------------------------
|
||||
*/
|
||||
#define gte_cmdw_psyq_compat (1u << 21 | enc_gte_sf(gte_sf_integer))
|
||||
@@ -440,10 +398,91 @@ enum { _C2_TX_SUBS_ = 0
|
||||
#define gte_cmdw_rtpt (gte_cmd_base | enc_gte_cmd(gte_cmd_rtpt ) | gte_cmdw_psyq_compat)
|
||||
#define gte_cmdw_nclip (gte_cmd_base | enc_gte_cmd(gte_cmd_nclip))
|
||||
#define gte_cmdw_op (gte_cmd_base | enc_gte_cmd(gte_cmd_op ))
|
||||
#define gte_cmdw_outer_product gte_cmdw_op /* "outer product" -- PSY-Q terminology */
|
||||
#define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology. */
|
||||
#define gte_cmdw_cross gte_cmdw_op /* "cross product" -- geometric-algebra terminology.
|
||||
* RGA(Lengyel): The GTE OP is a 3D signed-16-bit D x IR cross, not a generic RGA exterior product.
|
||||
* The wedge alias is a 3D complement interpretation of the same 3 scalars (MAC1..MAC3). */
|
||||
#define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva))
|
||||
|
||||
/* MVMVA with sf=0 (no shift, full-integer), cv=3 (no translation), v=3 (IR vector input).
|
||||
* Reads input from IR1/2/3 (loaded via mtc2 rt, C2_IRx). MAC1/2/3 = RT row · IR (full product, no >>12).
|
||||
* Per PSX-SPX: SAR (sf*12) with sf=0 = SAR 0 = no shift. */
|
||||
#define gte_cmdw_mvmva_sf0_ir (gte_cmd_base | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||
|
||||
/* MVMVA with sf=1 (>>12 shift, 4.12 fixed-point), cv=3 (no translation), v=3 (IR): for ApplyMatrixLV.
|
||||
* Reads input from IR1/2/3 (loaded via mtc2 rt, C2_IRx). MAC1/2/3 = (RT row · IR) >> 12.
|
||||
* Per PSX-SPX: SAR (sf*12) with sf=1 = SAR 12 = arithmetic right-shift by 12.
|
||||
* This matches the libgte C-side ApplyMatrixLV output (R*pos >> 12). */
|
||||
#define gte_cmdw_mvmva_ir (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||
|
||||
/* MVMVA: sf=0, mx=3 (Light matrix), v=3 (IR), cv=3 (no TR).
|
||||
* For pass1 of the C11 two-pass decomposition. Reads L matrix.
|
||||
* Since L matrix is typically zero, pass1 contributes 0 to the combine. */
|
||||
#define gte_cmdw_mvmva_sf0_mx3_v3_cv3 (gte_cmd_base | enc_gte_sf(0) | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||
|
||||
/* MVMVA: sf=1 (>>12), mx=3 (Light matrix), v=2 (V0), cv=0 (with TR).
|
||||
* Matches the C11 ApplyMatrixLV pass 2 command word (0x49E012) exactly.
|
||||
* The combine is (pass1 << 3) + pass2. */
|
||||
#define gte_cmdw_mvmva_pass2_c11 (gte_cmd_base | enc_gte_sf(1) | enc_gte_v(2) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||
|
||||
/* MVMVA: sf=0, mx=3, v=2, cv=0. Matches the C11 pass 1 command. */
|
||||
#define gte_cmdw_mvmva_pass1_c11 (gte_cmd_base | enc_gte_v(2) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||
#define gte_cmdw_mvmva_no_tr gte_cmdw_mvmva_ir
|
||||
|
||||
/* MVMVA pass 2 — C11 ApplyMatrixLV command.
|
||||
* Decoded: op_cop2 | CO | fake_cmd=4 | sf=1 (>>12) | mx=0 (RT matrix) | v=3 (IR) | cv=3 (no translation) | lm=0 | cmd=MVMVA.
|
||||
* Reads (RT row · IR) >> 12 into MAC1/2/3. Per-field composition (no opaque literal)
|
||||
* keeps the bit layout visible at the call site + matches the libgte C-side byte-exact. */
|
||||
#define gte_cmdw_mvmva_c11_pass2 (gte_cmd_base | enc_gte_fake_cmd(4) | enc_gte_sf(1) | enc_gte_v(3) | enc_gte_mx(0) | enc_gte_cv(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||
|
||||
/* MVMVA: sf=1 (>>12), mx=0 (RT matrix), v=0 (V0), cv=3 (no TR). */
|
||||
#define gte_cmdw_mvmva_sf1_mx0_v0_cv3 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_v(0) | enc_gte_mx(0) | enc_gte_cmd(gte_cmd_mvmva))
|
||||
|
||||
/* RTPS with sf=1 (12-bit shift, no translation): matches the output of libgte's ApplyMatrixLV when the GTE pipeline expects R*pos >> 12.
|
||||
* The shift produces values like (-270, 710, 1713) which match the C11 reference path. */
|
||||
#define gte_cmdw_rtps_sf1 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_cmd(gte_cmd_rtps))
|
||||
|
||||
/* SQR / GPF cosmetic-bits compat helpers.
|
||||
* Each command's `_compat` macro ORs in the `fake_cmd` field value libgte happens to emit.
|
||||
* The hardware ignores these bits (per PSX-SPX line 48). */
|
||||
#define gte_cmdw_sqr_fake_sig enc_gte_fake_cmd(0x0A)
|
||||
#define gte_cmdw_gpf_fake_sig enc_gte_fake_cmd(0x19)
|
||||
|
||||
/* SQR — Square Vector.
|
||||
* PSX-SPX `geometrytransformationenginegte.md` §"SQR":
|
||||
* [MAC1,MAC2,MAC3] = [IR1*IR1, IR2*IR2, IR3*IR3] SHR (sf*12)
|
||||
* [IR1,IR2,IR3] = [MAC1,MAC2,MAC3] (saturated to 0x7FFF when lm=1)
|
||||
* Sourced verbatim from libgte msc02 VectorNormal disassembly at 0x800160b0:
|
||||
* 0x4AA00428 = gte_cmd_base | gte_cmdw_sqr_compat | enc_gte_lm(1) | enc_gte_cmd(0x28)
|
||||
* bit 19 sf=0
|
||||
* bit 10 lm=1
|
||||
* bits 5-0 cmd=0x28=SQR
|
||||
* bits 24-20 = 0x0A (libgte "nonsense SDK command number" signature) */
|
||||
#define gte_cmdw_sqr (gte_cmd_base | enc_gte_cmd(gte_cmd_sqr) | enc_gte_lm(1) | gte_cmdw_sqr_fake_sig)
|
||||
|
||||
/* GPF — General-purpose Interpolation.
|
||||
* PSX-SPX `geometrytransformationenginegte.md` §"GPF":
|
||||
* [MAC1,MAC2,MAC3] = (([IR1,IR2,IR3] * IR0) + [MAC1,MAC2,MAC3]) SAR (sf * 12)
|
||||
* [IR1,IR2,IR3] = [MAC1,MAC2,MAC3]
|
||||
* Sourced verbatim from libgte msc02 VectorNormal disassembly at 0x8001613c:
|
||||
* 0x4B90003D = gte_cmd_base | gte_cmdw_gpf_compat | enc_gte_cmd(0x3D)
|
||||
* bit 19 sf = 0
|
||||
* bit 10 lm = 0
|
||||
* bits 5-0 cmd = 0x3D = GPF
|
||||
* bits 24-20 = 0x19 (libgte "nonsense SDK command number" signature) */
|
||||
#define gte_cmdw_gpf (gte_cmd_base | enc_gte_cmd(gte_cmd_gpf) | gte_cmdw_gpf_fake_sig)
|
||||
|
||||
/* Mask to round LZCR (leading-zero/ones count, range 1..32 per PSX-SPX cop2r31) down to even.
|
||||
* The normalize_v3s4 half-shift logic computes (31 - LZCR) >> 1; clearing bit 0 ensures the subtraction result is always odd, so the >> 1 division is consistent (no 0.5 loss). */
|
||||
enum {
|
||||
gte_lzcr_even_mask = 0xFFFE, /* all bits except bit 0 */
|
||||
};
|
||||
|
||||
#define gte_cmdw_rotate_translate_perspective_single gte_cmdw_rtps
|
||||
#define gte_cmdw_rotate_translate_perspective_triple gte_cmdw_rtpt
|
||||
/* RGA(Lengyel): RTPS/RTPT consume the matrix expansion of a rigid transformation (rotation matrix + translation vector) loaded into the RT/TR control registers.
|
||||
* For unitized points the same result equals the motor antiproduct; the GTE executes the LA form, not a symbolic antiproduct. */
|
||||
|
||||
/* PsyQ compatibility bits for AVSZ3 (Bits 20, 22, 24 must be set) */
|
||||
#define gte_cmdw_psyq_avsz3_compat (0x15 << 20)
|
||||
@@ -462,20 +501,16 @@ enum { _C2_TX_SUBS_ = 0
|
||||
|
||||
/**
|
||||
* @brief Loads a single SVECTOR to GTE vector register V0
|
||||
*
|
||||
* @details Loads values from an SVECTOR struct to GTE data registers C2_VXY0
|
||||
* (XY at offset 0) and C2_VZ0 (Z at offset 4) using `lwc2`.
|
||||
*
|
||||
* Uses string-style GCC inline asm with `%0` substitution because the
|
||||
* base register `r0` is a runtime GPR chosen by the compiler — it cannot
|
||||
* be encoded into a static `.word` constant.
|
||||
* Uses string-style GCC inline asm with `%0` substitution because the base register `r0` is a runtime GPR chosen by the compiler.
|
||||
* It cannot be encoded into a static `.word` constant.
|
||||
*
|
||||
* Usage:
|
||||
* asm_gte_load_v0(svector_ptr);
|
||||
* Usage: asm_gte_load_v0(svector_ptr);
|
||||
*/
|
||||
|
||||
/* lwc2 encoding helpers parameterized on the base GPR.
|
||||
*
|
||||
* gte_lw_v0_xy(base) → lwc2 $0, 0(base) ; C2_VXY0
|
||||
* gte_lw_v0_z(base) → lwc2 $1, 4(base) ; C2_VZ0
|
||||
* gte_lw_v1_xy(base) → lwc2 $2, 0(base) ; C2_VXY1
|
||||
@@ -484,8 +519,7 @@ enum { _C2_TX_SUBS_ = 0
|
||||
* gte_lw_v2_z(base) → lwc2 $5, 4(base) ; C2_VZ2
|
||||
*
|
||||
* `base` is the GPR number to bake into the .word constant's `rs` field.
|
||||
* These are pure compile-time integers; the C compiler constant-folds
|
||||
* them into .word directives. */
|
||||
* These are pure compile-time integers; the C compiler constant-folds them into .word directives. */
|
||||
|
||||
enum {
|
||||
GTE_Z_Offset = 4
|
||||
@@ -499,27 +533,21 @@ enum {
|
||||
#define gte_lw_v2_z(base) enc_gte_lw(gte_in_v2_z, (base), GTE_Z_Offset)
|
||||
|
||||
/* gte_load_vN(r_ptr, base) — placeholder-punned lwc2 loaders
|
||||
*
|
||||
* Emits `.word` constants encoding `lwc2 $N, off(<base>)` for the chosen
|
||||
* GTE vector register, where `<base>` is the GPR number you pass in
|
||||
* Emits `.word` constants encoding `lwc2 $N, off(<base>)` for the chosen GTE vector register, where `<base>` is the GPR number you pass in
|
||||
* (typically one of R_T4..R_T9 for the standard "3-pointer" pattern).
|
||||
*
|
||||
* The caller MUST bind `r_ptr` to that same GPR via a register variable:
|
||||
* register V3_S2* p_in_12 __asm__("$12") = my_ptr;
|
||||
* gte_load_v0(p_in_12, R_T4); // R_T4 = 12, base is $12
|
||||
*
|
||||
* Then `"r"(r_ptr)` inside the asm binds to $12 (the only register
|
||||
* `p_in_12` can live in), which is exactly the register the .word
|
||||
* constants expect. A `"$12"` clobber would conflict with the
|
||||
* register-variable binding ("asm specifier for variable conflicts
|
||||
* with asm clobber list"), so we omit it. The other ABI-clobbers
|
||||
* ($2/$8/$9/$31) stay because the GTE instructions don't touch
|
||||
* caller-saved GPRs but the kernel does treat them as volatile.
|
||||
* Then `"r"(r_ptr)` inside the asm binds to $12 (the only register `p_in_12` can live in),
|
||||
* which is exactly the register the .word constants expect.
|
||||
* A `"$12"` clobber would conflict with the register-variable binding ("asm specifier for variable conflicts with asm clobber list"), so we omit it.
|
||||
* The other ABI-clobbers ($2/$8/$9/$31) stay because the GTE instructions don't touch caller-saved GPRs but the kernel does treat them as volatile.
|
||||
*
|
||||
* WHICH REGISTER TO PICK
|
||||
* ----------------------
|
||||
* Any caller-saved GPR is safe. Recommended default for an RTPT-style
|
||||
* 3-pointer pipeline:
|
||||
* Any caller-saved GPR is safe. Recommended default for an RTPT-style 3-pointer pipeline:
|
||||
* gte_load_v0(p0, R_T4); // $12
|
||||
* gte_load_v1(p1, R_T5); // $13
|
||||
* gte_load_v2(p2, R_T6); // $14
|
||||
@@ -532,8 +560,7 @@ enum {
|
||||
* clobbers section : "$2", "$8", ..., "memory" (from asm_clobber)
|
||||
* 3 colons total, GCC-legal. No string-syntax mnemonics in the .word body.
|
||||
*
|
||||
* The `asm_clobber(...)` helper from gcc_asm.h prepends the colon that
|
||||
* starts the clobbers section. */
|
||||
* The `asm_clobber(...)` helper from gcc_asm.h prepends the colon that starts the clobbers section. */
|
||||
#define gte_load_v0(r_ptr, base) asm volatile( \
|
||||
asm_words( gte_lw_v0_xy(base), gte_lw_v0_z(base) ) \
|
||||
asm_rpins, r_use(r_ptr) \
|
||||
@@ -552,12 +579,10 @@ enum {
|
||||
asm_clobber: rlit(R_V0), rlit(R_T0), rlit(R_T1), rlit(R_RA), clb_mem_drain \
|
||||
)
|
||||
|
||||
/* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — the canonical prelude to gte_cmd_rtpt.
|
||||
*
|
||||
* Loads all three GTE input vectors (6 words) from three separate pointers,
|
||||
* one per GTE vector register, each loaded from its own base GPR. Caller
|
||||
* must bind each `pN` to `bN` via a register variable.
|
||||
/* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — prelude to gte_cmd_rtpt.
|
||||
*
|
||||
* Loads all three GTE input vectors (6 words) from three separate pointers, one per GTE vector register, each loaded from its own base GPR.
|
||||
* Caller must bind each `pN` to `bN` via a register variable.
|
||||
* register V3_S2* p0 rgcc(R_T4) = verts[0].ptr; // → __asm__("$12")
|
||||
* register V3_S2* p1 rgcc(R_T5) = verts[1].ptr; // → __asm__("$13")
|
||||
* register V3_S2* p2 rgcc(R_T6) = verts[2].ptr; // → __asm__("$14")
|
||||
@@ -576,29 +601,20 @@ enum {
|
||||
|
||||
/**
|
||||
* @brief Rotate, Translate and Perspective Triple (23 cycles)
|
||||
*
|
||||
* @details Performs rotation, translation and perspective calculation of three
|
||||
* vertices at once. The equation performed is the same as gte_rtps() only
|
||||
* repeated three times for each vertex. The result of the first vertex is
|
||||
* stored in GTE data register C2_SXY0, the second vector in C2_SXY1 then
|
||||
* C2_SXY2.
|
||||
* @details Performs rotation, translation and perspective calculation of three vertices at once.
|
||||
* The equation performed is the same as gte_rtps() only repeated three times for each vertex.
|
||||
* The result of the first vertex is stored in GTE data register C2_SXY0, the second vector in C2_SXY1 then C2_SXY2.
|
||||
*
|
||||
* Encoder-style emission (no inline-asm strings in the code body):
|
||||
* 1. Two `nop` words fill the COP2 pipeline latency — the GTE
|
||||
* takes ~8 cycles per perspective divide, and the nops let any
|
||||
* preceding lwc2/swc2 retire before RTPT starts reading its
|
||||
* inputs from V0/V1/V2.
|
||||
* 2. The RTPT command word itself is `gte_cmdw_rtpt` (see the
|
||||
* pre-baked encoders above) — `0x0280030` decoded as
|
||||
* `op_cop2` | CO(1) | cmd=RTPT, with all SF/MX/V/CV/LM fields
|
||||
* zero (standard rotation, no scaling, V0 vector, translation
|
||||
* vector, no clamp).
|
||||
* 1. Two `nop` words fill the COP2 pipeline latency — the GTE takes ~8 cycles per perspective divide,
|
||||
* and the nops let any preceding lwc2/swc2 retire before RTPT starts reading its inputs from V0/V1/V2.
|
||||
* 2. The RTPT command word itself is `gte_cmdw_rtpt` (see the pre-baked encoders above) —
|
||||
* `0x0280030` decoded as `op_cop2` | CO(1) | cmd=RTPT, with all SF/MX/V/CV/LM fields zero
|
||||
* (standard rotation, no scaling, V0 vector, translation vector, no clamp).
|
||||
*
|
||||
* Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel
|
||||
* ABI) plus the standard "memory" barrier. Does not clobber any COP2
|
||||
* data/control register — those have to be saved by the caller if
|
||||
* they need to survive across the call (RTPT writes SXY0..2, SZ0..3,
|
||||
* OTZ, MAC0..3, IR0..3, etc.).
|
||||
* Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel ABI)
|
||||
* plus the standard "memory" barrier. Does not clobber any COP2 data/control register —
|
||||
* those have to be saved by the caller if they need to survive across the call (RTPT writes SXY0..2, SZ0..3, OTZ, MAC0..3, IR0..3, etc.).
|
||||
*/
|
||||
#define gte_rtpt() \
|
||||
asm volatile( \
|
||||
@@ -614,32 +630,24 @@ enum {
|
||||
|
||||
/**
|
||||
* @brief Normal clipping (8 cycles)
|
||||
*
|
||||
* @details Computes the sign of three screen coordinates (C2_SXY0-2) used for
|
||||
* backface culling. If the value of C2_MAC0 is negative, the coordinates are
|
||||
* inverted and thus the triangle is back facing.
|
||||
* @details Computes the sign of three screen coordinates (C2_SXY0-2) used for backface culling.
|
||||
* If the value of C2_MAC0 is negative, the coordinates are inverted and thus the triangle is back facing.
|
||||
*
|
||||
* The following equation is performed when executing this GTE command:
|
||||
*
|
||||
* MAC0 = SX0*SY1 + SX1*SY2 + SX2*SY0 - SX0*SY2 - SX1*SY0 - SX2*SY1
|
||||
*
|
||||
* Encoder-style emission (no inline-asm strings in the code body):
|
||||
* 1. Two `nop` words fill the COP2 pipeline latency - the GTE
|
||||
* pipeline takes a few cycles per op, and the nops let any
|
||||
* preceding lwc2/swc2/RTPT retire before NCLIP starts reading
|
||||
* its inputs from SXY0/SXY1/SXY2.
|
||||
* 2. The NCLIP command word itself is `gte_cmdw_nclip` (see the
|
||||
* pre-baked encoders above) - `0x01400006` decoded as
|
||||
* `op_cop2` | CO(1) | cmd=NCLIP, with all SF/MX/V/CV/LM fields
|
||||
* zero. NCLIP is spec-clean in the original PsyQ source
|
||||
* (unlike RTPS/RTPT which carry the `gte_cmdw_psyq_compat`
|
||||
* quirk), so `gte_cmdw_nclip` does NOT OR in any reserved bits.
|
||||
* 1. Two `nop` words fill the COP2 pipeline latency
|
||||
* - the GTE pipeline takes a few cycles per op, and the nops let any preceding
|
||||
* lwc2/swc2/RTPT retire before NCLIP starts reading its inputs from SXY0/SXY1/SXY2.
|
||||
* 2. The NCLIP command word itself is `gte_cmdw_nclip` (see the pre-baked encoders above)
|
||||
* - `0x01400006` decoded as `op_cop2` | CO(1) | cmd=NCLIP, with all SF/MX/V/CV/LM fields zero.
|
||||
* NCLIP is spec-clean in the original PsyQ source (unlike RTPS/RTPT which carry the `gte_cmdw_psyq_compat` quirk),
|
||||
* so `gte_cmdw_nclip` does NOT OR in any reserved bits.
|
||||
*
|
||||
* Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel
|
||||
* ABI) plus the standard "memory" barrier. Does not clobber any COP2
|
||||
* data/control register - those have to be saved by the caller if
|
||||
* they need to survive across the call (NCLIP writes MAC0 only; it
|
||||
* is purely a sign-of-double-product computation on SXY0..2).
|
||||
* Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel ABI) plus the standard "memory" barrier.
|
||||
* Does not clobber any COP2 data/control register.
|
||||
* Those have to be saved by the caller if they need to survive across the call (NCLIP writes MAC0 only;
|
||||
* it is purely a sign-of-double-product computation on SXY0..2).
|
||||
*/
|
||||
#define gte_nclip() \
|
||||
asm volatile( \
|
||||
@@ -665,14 +673,10 @@ enum {
|
||||
"cop2 0x0158002D;")
|
||||
|
||||
/* asm_gte_matrix_set_rotation(r0)
|
||||
* Loads the 3x3 rotation matrix at `r0` into the GTE's rotation-matrix control registers (RT11..RT22, indices 0..4) via ctc2.
|
||||
*
|
||||
* Loads the 3x3 rotation matrix at `r0` into the GTE's rotation-matrix
|
||||
* control registers (RT11..RT22, indices 0..4) via ctc2.
|
||||
*
|
||||
* Memory layout at r0: five contiguous 32-bit words (offsets 0..16),
|
||||
* each holding two packed 16-bit matrix elements. The first 1.5 rows
|
||||
* of a standard PSX SDK MATRIX struct (where each row is laid out as
|
||||
* [RT_xx, RT_xy] | [RT_xz, pad] | ...).
|
||||
* Memory layout at r0: five contiguous 32-bit words (offsets 0..16), each holding two packed 16-bit matrix elements.
|
||||
* The first 1.5 rows of a standard PSX SDK MATRIX struct (where each row is laid out as [RT_xx, RT_xy] | [RT_xz, pad] | ...).
|
||||
*
|
||||
* Generated MIPS (mirrors the source macro):
|
||||
* lw $12, 0( %0 ) ; word 0
|
||||
@@ -686,41 +690,36 @@ enum {
|
||||
* ctc2 $13, $3 ; → C2_RT21
|
||||
* ctc2 $14, $4 ; → C2_RT22
|
||||
*
|
||||
* Same contract as gte_load_v0: caller MUST bind `r0` to $12 via a
|
||||
* register variable (`rgcc(R_T4)`) for the `lw $12, off(...)`
|
||||
* instructions to read from the right base. The `"r"(r0)` constraint
|
||||
* alone doesn't force a specific GPR — it just lets GCC pick one.
|
||||
* The .word constants here bake R_T4/R_T5/R_T6 into the `rs` field
|
||||
* of each lw, so the lw instructions will only do the right thing
|
||||
* if $12/$13/$14 hold the matrix base at runtime.
|
||||
* Same contract as gte_load_v0: caller MUST bind `r0` to $12 via a register variable (`rgcc(R_T4)`) for the `lw $12, off(...)`
|
||||
* instructions to read from the right base. The `"r"(r0)` constraint alone doesn't force a specific GPR — it just lets GCC pick one.
|
||||
* The .word constants here bake R_T4/R_T5/R_T6 into the `rs` field of each lw, so the lw instructions will
|
||||
* only do the right thing if $12 / $13 / $14 hold the matrix base at runtime.
|
||||
*
|
||||
* M3_S2* m = ...;
|
||||
* register M3_S2* m_in_12 rgcc(R_T4) = m;
|
||||
* asm_gte_matrix_set_rotation(m_in_12);
|
||||
*
|
||||
* We clobber $12/$13/$14 (the ones we use as scratch inside the
|
||||
* inline asm) plus the system clobbers; we don't clobber `r0` because
|
||||
* the `rgcc` binding already says "this variable lives in $12".
|
||||
* We clobber $12/$13/$14 (the ones we use as scratch inside the inline asm)
|
||||
* plus the system clobbers; we don't clobber `r0` because the `rgcc` binding already says "this variable lives in $12".
|
||||
*
|
||||
* WARNING: Incomplete by design. The source macro only writes RT11..RT22
|
||||
* (5 of 9 rotation elements); RT23 and the entire RT3x row are left
|
||||
* untouched. Real libpsn00b SetRotMatrix writes all 9. Use only when the
|
||||
* GTE's remaining rotation entries are already correct, or you will
|
||||
* get stale-RT2x/RT3x artifacts in RTPS/RTPT/MVMVA output.
|
||||
* WARNING: Incomplete by design. The source macro only writes RT11..RT22 (5 of 9 rotation elements);
|
||||
* RT23 and the entire RT3x row are left untouched.
|
||||
* Real libpsn00b SetRotMatrix writes all 9. Use only when the GTE's remaining rotation entries are already correct,
|
||||
* or you will get stale-RT2x/RT3x artifacts in RTPS/RTPT/MVMVA output.
|
||||
*/
|
||||
#define asm_gte_matrix_set_rotation(r0) \
|
||||
asm volatile( \
|
||||
asm_words( \
|
||||
load_word(R_T5, R_T4, 0) \
|
||||
, load_word(R_T6, R_T4, 4) \
|
||||
, gte_mt( R_T5, 0) \
|
||||
, gte_mt( R_T6, 1) \
|
||||
, gte_mv_to_data_r( R_T5, 0) \
|
||||
, gte_mv_to_data_r( R_T6, 1) \
|
||||
, load_word(R_T5, R_T4, 8) \
|
||||
, load_word(R_T6, R_T4, 12) \
|
||||
, load_word(R_T4, R_T4, 16) \
|
||||
, gte_mt( R_T5, 2) \
|
||||
, gte_mt( R_T6, 3) \
|
||||
, gte_mt( R_T4, 4) \
|
||||
, gte_mv_to_data_r( R_T5, 2) \
|
||||
, gte_mv_to_data_r( R_T6, 3) \
|
||||
, gte_mv_to_data_r( R_T4, 4) \
|
||||
) \
|
||||
, r_use(r0) \
|
||||
asm_clobber: clbr_volatile_gprs, rlit(R_T4), rlit(R_T5), rlit(R_T6) \
|
||||
|
||||
@@ -2,10 +2,8 @@
|
||||
* duffle DSL — GTE Vendor Mnemonics (opt-in)
|
||||
* ============================================================================
|
||||
*
|
||||
* Provides the textbook MIPS assembly mnemonics for the GTE/COP2
|
||||
* instructions as thin aliases to the canonical duffle macros in gte.h.
|
||||
* The duffle names are primary; this header is for users who prefer
|
||||
* the textbook mnemonics.
|
||||
* Provides the textbook MIPS assembly mnemonics for the GTE/COP2 instructions as thin aliases to the duffle macros in gte.h.
|
||||
* The duffle names are primary; this header is for users who prefer the textbook mnemonics.
|
||||
*
|
||||
* USAGE: #include "duffle/gte_vendor_sym.h" // after gte.h
|
||||
*
|
||||
@@ -21,12 +19,6 @@
|
||||
* gte_swc2(rt, base, off) -> gte_sw(rt, base, off)
|
||||
* (the lower-level vector variants gte_lw_v0_xy etc. don't have
|
||||
* vendor mnemonics; they're already gte_-prefixed and short)
|
||||
*
|
||||
* The vendor mnemonics are NOT registered with the duffle word-count
|
||||
* metadata (tape_atom.metadata.h). They expand to the duffle canonical
|
||||
* macros which DO have word-count entries. Verification: V3 (objdump
|
||||
* byte-identical) holds.
|
||||
*
|
||||
* ============================================================================ */
|
||||
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
|
||||
+392
-304
@@ -1,107 +1,267 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "gen/macs.h"
|
||||
# include "gen/offsets.h"
|
||||
|
||||
# include "dsl.h"
|
||||
# include "gcc_asm.h"
|
||||
# include "mips.h"
|
||||
# include "gte.h"
|
||||
# include "memory.h"
|
||||
# include "atom_dsl.h"
|
||||
# include "gen/duffle.macs.h"
|
||||
# include "gen/duffle.offsets.h"
|
||||
# include "dsl.atom.h"
|
||||
#endif
|
||||
|
||||
typedef U4 const MipsCode;
|
||||
#pragma region Tape Drive
|
||||
/* -----------------------------------------------------------------------------------------------------------
|
||||
* TAPE DRIVE ABI
|
||||
* -----------------------------------------------------------------------------------------------------------
|
||||
* Note(Ed): One of the main purposes of this codebase is to help me learn this,
|
||||
* as such the information below may not* be entirely realized or finalized conceptually.
|
||||
* -----------------------------------------------------------------------------------------------------------
|
||||
* This ABI and its associated legos were directly inspired by researching the work of
|
||||
* Timothy Lottes and Onat Türkçüoğlu; along with many others. It's the simplest bootstrap of a
|
||||
* directly executed chain of assemby arrays (Atoms) that terminate with a yield sequence to the next atom.
|
||||
* These eventually lead to a terminal atom for the tape which is defined below as "tape_exit".
|
||||
*
|
||||
* It behaves as one of the simplest runtime harnesses ontop of a host-enviornment's execution engine
|
||||
* to author and compose programs with. From here various conventions can be further applied.
|
||||
* To make things easier to understand it may be better to focus on what this ABI does not have.
|
||||
* It does not have have any branching within the tape but relative branches within atoms or between atoms.
|
||||
* Branching nearly is always downstream. Automatic stack usage is non-existent.
|
||||
* Push/Pop, FIFO, or Arena/Bump data structures are used by atoms explicitly.
|
||||
* In it's current form with the C11 macro DSL, the user also has fullfill manual register allocation per atom.
|
||||
*
|
||||
* One of the remarkable things about utilizing this ABI is its essentially interopable with CPUs, GPUs, FPGA,
|
||||
* or, basically anything from the 5th generation consoles and onward.
|
||||
* The ABI directly reflects how all computational hardware must be architected in order to execute
|
||||
* digital logic effectively on current era tech.
|
||||
* On the PS1 we don't have access to a few features like multi-threading, speculative execution, or L3 cache;
|
||||
* but, we can set the foundation for legoing whats required for eventually expanding this ABI's paradigm
|
||||
* and core atoms to take those newer hardware features into account. For example, you can easily expand
|
||||
* this to support wave-based execution model on a PS2 or PS3. Not having a stack or
|
||||
* automatic register allocation means the user cannot ignore excessive argument shuffle across workload or
|
||||
* waves and thier phases. Crossing ABI boundaries to other runtimes that do has obviouss penalties.
|
||||
*
|
||||
* Learning data-oriented code becomes a natural progression. Your not fighting a stack-based procedural
|
||||
* paradigm that wants to argument shuffle. There is no ambiguity due to the lack of constraints, for example,
|
||||
* on how the user may "call" a procedure in traditional random dispatch runtimes. The user does have to
|
||||
* hammer down "rules" or patterns for massaging the compiler to dissolve those call frames; just to get
|
||||
* the asesmbly into its desired form. The form is obvious, and once the user gets to author these compoonents
|
||||
* it becomes a game of tetris.
|
||||
*
|
||||
* Another feature is this ABI is very compatible with bootstrapping and developing simple toolchains built off
|
||||
* of bit-packed annotated command streams the user can directly author, maintatain, and immediately execute.
|
||||
* That being like a color forth, or maybe something more familar like an immediate mode library
|
||||
* for various systems such as GUIs. This can make the tetris less of a chore with some helpful policy
|
||||
* generation for allocation of registers, helping to choose resuable components, designing DSL on the fly, etc.
|
||||
* -----------------------------------------------------------------------------------------------------------
|
||||
* TODO(Ed): We need pretty ascii diagrams and proper guides, articles, etc.
|
||||
* -----------------------------------------------------------------------------------------------------------
|
||||
* For now this ideation has just started functioning. I'm abusing C11 & a lua metaprogram to help establish
|
||||
* a hybrid toolchain to ideate on a traditional text-based authoring UX for this paradigm.
|
||||
* If pcsx-redux provides viable hot-reload and persistent data storage beyond save-states
|
||||
* (just copying ram to filesystem), I can author a color forth to mess around with.
|
||||
* With either an editor in-emulator or on the actual machine itself. Assembly is tedius,
|
||||
* but I think this codebase most likely has a pretty ergonomic flavor worst case...
|
||||
* */
|
||||
/* Register Allocation Info */
|
||||
enum {
|
||||
R_ScratchBase = R_SP atom_reg, /* Scratchpad base address (host frame top) */
|
||||
R_AtomJmp = R_FP atom_reg, /* Next atom target (yield handshake scratch) */
|
||||
R_TapePtr = R_RA atom_reg, /* The Instruction Stream Pointer */
|
||||
/* Stringification codes for the GCC inline assembler clobber lists. */
|
||||
#define R_ScratchBase_Code R_SP_Code
|
||||
#define R_AtomJmp_Code R_FP_Code
|
||||
#define R_TapePtr_Code R_RA_Code
|
||||
|
||||
// R_InCursor = R_T4,
|
||||
// #define R_InCursor_Code R_T4_Code
|
||||
|
||||
// Reserved Registers (Callee-saved across the host ABI transition):
|
||||
// - R_SP: Holds the scratchpad base while tape code executes.
|
||||
// - R_FP: Holds the next atom target.
|
||||
// - R_RA: Holds the tape cursor.
|
||||
// All atom-body allocations must stay out of these.
|
||||
// Atom bodies may freely use R2-R25.
|
||||
|
||||
// All allocatable registers for atom bodies (R2-R25, 24 registers):
|
||||
|
||||
R_PsuedoVolatile = R_AT, // Assembler temporary; never allocate.
|
||||
|
||||
// Atom Allocation Pool
|
||||
R_Atom0 = R_T0,
|
||||
R_Atom1 = R_T1,
|
||||
R_Atom2 = R_T2,
|
||||
R_Atom3 = R_T3,
|
||||
R_Atom4 = R_T4,
|
||||
R_Atom5 = R_T5,
|
||||
R_Atom6 = R_T6,
|
||||
R_Atom7 = R_T7,
|
||||
R_Atom8 = R_T8,
|
||||
R_Atom9 = R_T9,
|
||||
R_Atom10 = R_V0, // Tend to be used with gte moves
|
||||
R_Atom11 = R_V1, // Tend to be used with gte moves
|
||||
R_Atom12 = R_A0,
|
||||
R_Atom13 = R_A1,
|
||||
R_Atom14 = R_A2,
|
||||
R_Atom15 = R_A3,
|
||||
R_Atom16 = R_S0,
|
||||
R_Atom17 = R_S1,
|
||||
R_Atom18 = R_S2,
|
||||
R_Atom19 = R_S3,
|
||||
R_Atom20 = R_S4,
|
||||
R_Atom21 = R_S5,
|
||||
R_Atom22 = R_S6,
|
||||
R_Atom23 = R_S7,
|
||||
};
|
||||
|
||||
typedef U2 Reg; // Register parameter used with atom or atom component procedures
|
||||
#define Reg_(type) tmpl(Reg,type) // Just a way to template register allocations of C-struct types.
|
||||
|
||||
typedef U4 const MipsCode; // Underlying type to mips asm words.
|
||||
typedef Slice_(MipsCode);
|
||||
typedef Slice_MipsCode MipsAtom;
|
||||
|
||||
#define MipsAtom_(sym) MipsCode tmpl(code,sym) [] align_(4) =
|
||||
typedef U4 const MipsAtom; // Underlying type to a mips atom defnition
|
||||
typedef Slice_(MipsAtom);
|
||||
|
||||
// Bare form: file-scope declaration with hardcoded body.
|
||||
// Used for components with no args (e.g., ac_load_tri_indices) or
|
||||
// identifier-args (hardcoded register names).
|
||||
// Sometimes a user will define a bundle of atoms that represent a procedure of work as:
|
||||
// MipsAtom* <identifier>[...];
|
||||
// Unfortuantely if using slice_from_array it will make the slice's pointer: MipsAtom** so this enforce its defined as MipsAtom*
|
||||
// TODO(Ed): Alternatively we can make the MipsAtom an opaque pointer to the atom... so that the proc returns 'MipsAtom'.
|
||||
#define atombundle_from_array(array) (Slice_MipsAtom){.ptr=array[0],.len=Array_len(array)}
|
||||
|
||||
// Underlying type to an ptr to an array of mips asm words that must terminate with an ac_yield.
|
||||
#define MipsAtom_(sym) MipsCode sym [] align_(4) =
|
||||
|
||||
// Used for atoms with value-args
|
||||
// internal MipsAtom* X_proc(AtomArena_R aa, args) MipsAtom_Proc_(X, aa, { body })
|
||||
// expands to:
|
||||
// internal MipsAtom* X_proc(AtomArena_R aa, args) { MipsCode atom_comp_code[] align_(4) = { body }; return atomarena_push(aa, slice_from_array(MipsCode, atom_comp_code)); }
|
||||
// The atom name is derived by the Lua metaprogram from the preceding
|
||||
// `MipsAtom* X_proc(...)` declaration (backward walk from the macro site,
|
||||
// strips the `_proc` suffix).
|
||||
#define MipsAtom_Proc_(aa, ...) { MipsCode atom_comp_code[] align_(4) = __VA_ARGS__; return atomarena_push(aa, slice_from_array(MipsCode, atom_comp_code)); }
|
||||
|
||||
// Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names).
|
||||
// MipsAtomComp_(ac_X) { body }
|
||||
// expands to:
|
||||
// MipsCode ac_X[] align_(4) = { body };
|
||||
#define MipsAtomComp_(sym) MipsCode sym [] align_(4) =
|
||||
|
||||
// Function form: function-body block that returns a MipsAtom slice.
|
||||
// Used for components with value-args (e.g., ac_format_f3_color).
|
||||
// FI_ MipsAtom ac_X(args) MipsAtomComp_Proc_(ac_X, { body })
|
||||
// Used for components with value-args (mandatory `ab` (atom-builder) arg).
|
||||
// FI_ void ac_X(MipsAtomBuilder_R ab, args) MipsAtomComp_Proc_(ab, { body })
|
||||
// expands to:
|
||||
// FI_ MipsAtom ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return slice_from_array(MipsCode, ac_X); }
|
||||
#define MipsAtomComp_Proc_(sym, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; return slice_from_array(MipsCode, sym); }
|
||||
// FI_ void ac_X(MipsAtomBuilder_R ab, args) {
|
||||
// MipsCode atom_comp_code[] align_(4) = { body };
|
||||
// atombuilder_push(ab, slice_from_array(MipsCode, atom_comp_code));
|
||||
// }
|
||||
// The body must NOT include mac_yield() (the parent atom yields).
|
||||
// The component name is derived by the Lua metaprogram from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration (backward walk from the macro site).
|
||||
// Inline-only callers (the generated `mac_<name>` aliases) skip the `ab` arg via metaprogram filtering; escape callers (ac_<name> invoked as a function) pass a long-lived builder.
|
||||
#define MipsAtomComp_Proc_(ab, ...) { MipsCode atom_comp_code[] align_(4) = __VA_ARGS__; atombuilder_push(ab, slice_from_array(MipsCode, atom_comp_code)); }
|
||||
|
||||
// Auto-generated component macros (<module>/gen/<dir>/<dir>.macs.h)
|
||||
// are included manually by the unity build. The metaprogram puts them
|
||||
// Used for trivial mappings from one atom component proc to the command of a more baser (meant for type-mapping)
|
||||
#define MipsAtomComp_ProcMap_(ab, base_command) atom_dbg_skip MipsAtomComp_Proc_(ab, {base_command })
|
||||
|
||||
/* Register aliases (moved up from the Tape Drive region below so that
|
||||
* mac_yield's body and the Mips Atom Builder functions can reference
|
||||
* them. The C compiler processes the file top-to-bottom, so the enum
|
||||
* must be visible before any use.) */
|
||||
enum {
|
||||
R_AtomJmp = R_T9,
|
||||
R_TapePtr = R_T8, /* The Instruction Stream Pointer */
|
||||
R_InCursor = R_T4, /* Input data cursor */
|
||||
// WIP: Atoms Assocated closely with each other to form a tape procedure. (Maybe also a phase in a procedure/pipeline?)
|
||||
|
||||
R_PrimCursor = R_T7, /* VRAM output cursor (primitive buffer) */
|
||||
R_FaceCursor = R_T4, /* Input data cursor (indices/faces) */
|
||||
R_VertBase = R_T5, /* Base address of the vertex array */
|
||||
R_OtBase = R_T6, /* Base address of the Ordering Table */
|
||||
#define AtomBundle_(name) Struct_(tmpl(AtomBundle,name))
|
||||
#define AtomBundle_Len(name) S_(tmpl(AtomBundle,name))/S_(MipsAtom*)
|
||||
#define AtomBundleEntry_(bundle,entry) tmpl(bundle,entry)
|
||||
|
||||
/* Stringification codes for the GCC inline assembler clobber lists */
|
||||
#define R_TapePtr_Code R_T8_Code
|
||||
#define R_InCursor_Code R_T4_Code
|
||||
/* Line-table anchor: gcc only adds a file to the .debug_line file table when the contains line-numbered content.
|
||||
Files containing only atoms and atom components.
|
||||
Place `ATOM_FILE_LINE_MARKER();` once at file scope in any `.atom.c` that defines atoms.
|
||||
Macro expands to a file-scope `internal U4 const` declaration keeps the file in the line table.
|
||||
The constant is in `.rodata` so the linker may eliminate it. */
|
||||
#define ATOM_FILE_DEBUGGER_LINE_MARKER(file_name) internal U4 const tmpl(atom_file_debugger_line_marker,file_name) = 0
|
||||
|
||||
#define R_PrimCursor_Code R_T7_Code
|
||||
#define R_FaceCursor_Code R_T4_Code
|
||||
#define R_VertBase_Code R_T5_Code
|
||||
#define R_OtBase_Code R_T6_Code
|
||||
typedef Slice_MipsAtom Tape;
|
||||
|
||||
typedef Struct_(TapeHostFrame) {
|
||||
U4 s0;
|
||||
U4 s1;
|
||||
U4 s2;
|
||||
U4 s3;
|
||||
U4 s4;
|
||||
U4 s5;
|
||||
U4 s6;
|
||||
U4 s7;
|
||||
U4 fp;
|
||||
U4 sp;
|
||||
U4 ra;
|
||||
};
|
||||
|
||||
#pragma region Tape Drive
|
||||
/* ---------------------------------------------------------------------------
|
||||
* TAPE DRIVE ABI & REGISTER ALIASES (the enum moved earlier; see below)
|
||||
* ---------------------------------------------------------------------------*/
|
||||
enum {
|
||||
TapeHostFrame_Loc = Scratchpad_End - S_(TapeHostFrame),
|
||||
TapeScratch_Len = TapeHostFrame_Loc - Scratchpad_Loc,
|
||||
};
|
||||
static_assert(S_(TapeHostFrame) == 11 * S_(U4));
|
||||
static_assert(TapeHostFrame_Loc == 0x1F8003D4);
|
||||
|
||||
/* The 'Exit' Atom */
|
||||
MipsAtom_(tape_exit) { jump_reg(rret_addr), nop };
|
||||
atom_dbg_skip MipsAtom_(tape_enter) {
|
||||
mac_load_word_imm(R_V0, u4_(TapeHostFrame_Loc)),
|
||||
store_word(R_S0, R_V0, O_(TapeHostFrame,s0)),
|
||||
store_word(R_S1, R_V0, O_(TapeHostFrame,s1)),
|
||||
store_word(R_S2, R_V0, O_(TapeHostFrame,s2)),
|
||||
store_word(R_S3, R_V0, O_(TapeHostFrame,s3)),
|
||||
store_word(R_S4, R_V0, O_(TapeHostFrame,s4)),
|
||||
store_word(R_S5, R_V0, O_(TapeHostFrame,s5)),
|
||||
store_word(R_S6, R_V0, O_(TapeHostFrame,s6)),
|
||||
store_word(R_S7, R_V0, O_(TapeHostFrame,s7)),
|
||||
store_word(R_FP, R_V0, O_(TapeHostFrame,fp)),
|
||||
store_word(R_SP, R_V0, O_(TapeHostFrame,sp)),
|
||||
store_word(R_RA, R_V0, O_(TapeHostFrame,ra)),
|
||||
add_ui(R_TapePtr, R_A0, 0),
|
||||
load_upper_i(R_ScratchBase, u4_hi(Scratchpad_Loc)),
|
||||
load_word(R_AtomJmp, R_TapePtr, 0),
|
||||
add_ui_self( R_TapePtr, S_(MipsAtom)),
|
||||
jump_reg(R_AtomJmp), BdSlot_ nop,
|
||||
};
|
||||
|
||||
/* Generalized Tape Engine Runner */
|
||||
FI_ void tape_run(Slice_U4 tape) { register U4* tp rgcc(R_TapePtr) = tape.ptr; asm volatile(
|
||||
asm_words(
|
||||
add_ui( R_SP, R_SP, -MipsStackAlignment) /* Allocate stack space */
|
||||
, store_word( R_RA, R_SP, 0) /* Safely backup $ra to the stack */
|
||||
, load_word( R_AtomJmp, R_TapePtr, 0) /* Bootstrap the first jump */
|
||||
, add_ui_self(R_TapePtr, S_(MipsCode)) /* Advance tape */
|
||||
, call_reg( R_AtomJmp) /* jalr $t9 */
|
||||
, nop /* Branch delay slot */
|
||||
, load_word( R_RA, R_SP, 0) /* Restore $ra from stack */
|
||||
, add_ui_self(R_SP, MipsStackAlignment) /* Deallocate stack space */
|
||||
)
|
||||
asm_rpins, r_use(tp)
|
||||
asm_clobber:
|
||||
rlit(R_AT)
|
||||
, rlit(R_V0), rlit(R_V1)
|
||||
, rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3)
|
||||
/* Tell GCC the tape engine owns and destroys the workspace registers */
|
||||
, rlit(R_PrimCursor), rlit(R_FaceCursor), rlit(R_VertBase), rlit(R_OtBase)
|
||||
, rlit(R_T9)
|
||||
, clb_mem_drain
|
||||
); }
|
||||
atom_dbg_skip MipsAtom_(tape_exit) {
|
||||
mac_load_word_imm(R_V0, u4_(TapeHostFrame_Loc)),
|
||||
load_word(R_S0, R_V0, O_(TapeHostFrame,s0)),
|
||||
load_word(R_S1, R_V0, O_(TapeHostFrame,s1)),
|
||||
load_word(R_S2, R_V0, O_(TapeHostFrame,s2)),
|
||||
load_word(R_S3, R_V0, O_(TapeHostFrame,s3)),
|
||||
load_word(R_S4, R_V0, O_(TapeHostFrame,s4)),
|
||||
load_word(R_S5, R_V0, O_(TapeHostFrame,s5)),
|
||||
load_word(R_S6, R_V0, O_(TapeHostFrame,s6)),
|
||||
load_word(R_S7, R_V0, O_(TapeHostFrame,s7)),
|
||||
load_word(R_RA, R_V0, O_(TapeHostFrame,ra)),
|
||||
load_word(R_FP, R_V0, O_(TapeHostFrame,fp)),
|
||||
load_word(R_SP, R_V0, O_(TapeHostFrame,sp)),
|
||||
jump_reg(R_RA), BdSlot_ nop,
|
||||
};
|
||||
|
||||
typedef void Proc_(TapeEntryFn)(MipsAtom* tape_ptr);
|
||||
|
||||
FI_ void tape_run(Tape tape) { C_(TapeEntryFn*, tape_enter)(tape.ptr); }
|
||||
|
||||
// Procedural authoring of tapes:
|
||||
typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; };
|
||||
FI_ void tb_init(TapeBuilder* tb, FArena* arena) { tb->ptr = arena->start; tb->used = 0; }
|
||||
FI_ TapeBuilder tb_make_old( FArena* arena) { return (TapeBuilder){ arena->start, 0 }; }
|
||||
FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ mem.ptr, mem.len, 0 }; }
|
||||
FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ u4_(mem.ptr), mem.len, 0 }; } /* capacity in elements (matches used units) */
|
||||
|
||||
#define tb_emit_(tb, atom) tb_emit(tb, tmpl(code,atom))
|
||||
FI_ void tb_emit(TapeBuilder* tb, MipsCode* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
|
||||
FI_ void tb_emit(TapeBuilder* tb, MipsAtom* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
|
||||
FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; }
|
||||
#define tb_emit_(atom) tb_emit(& tb, atom)
|
||||
|
||||
FI_ Slice_U4 tb_end (TapeBuilder* tb) { tb_emit(tb,code_tape_exit); return (Slice_U4){ C_(U4*,tb->ptr), tb->used }; }
|
||||
FI_ Slice_U4 tb_slice(TapeBuilder tb) { return (Slice_U4){ C_(U4*,tb.ptr), tb.used }; }
|
||||
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,code_tape_exit))
|
||||
FI_ void tb_bind(TapeBuilder* tb, Slice data) { mem_copy(tb->ptr + tb->used * S_(MipsCode), u4_(data.ptr), data.len); tb->used += data.len / S_(MipsCode); }
|
||||
#define tb_bind_(tb,type,...) tb_bind(tb, (Slice){ (B1*)(& (type){__VA_ARGS__}), S_(type) }); static_assert(S_(type) % S_(MipsCode) == 0)
|
||||
|
||||
// NOTE(Ed): Wip still ideating convention. Possibly will never use a composite.
|
||||
#define tb_emit_wbind_(tb,atom,...) tb_emit(tb,atom); tb_bind_(tb,tmpl(Binds,atom),__VA_ARGS__)
|
||||
#define tb_emit_wbind2_(tb,atom,type,...) tb_emit(tb,atom); tb_bind_(tb,type,__VA_ARGS__)
|
||||
|
||||
FI_ Tape tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Tape){ C_(U4*,tb->ptr), tb->used }; }
|
||||
FI_ Tape tb_slice(TapeBuilder tb) { return (Tape){ C_(U4*,tb.ptr), tb.used }; }
|
||||
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit))
|
||||
|
||||
FI_ void tb_scope_run_end(TapeBuilder* tb) { tb_emit(tb,tape_exit); tape_run(tb_slice(tb[0])); }
|
||||
#define tb_scope_run(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_scope_run_end(tb))
|
||||
#pragma endregion Tape Drive
|
||||
|
||||
#pragma region Macro Mips Atom Components
|
||||
@@ -111,258 +271,186 @@ FI_ Slice_U4 tb_slice(TapeBuilder tb) { return (Sli
|
||||
* ---------------------------------------------------------------------------*/
|
||||
|
||||
// The 'Yield' sequence for Tape Atoms (mac_yield).
|
||||
MipsAtomComp_(ac_yield) {
|
||||
load_word(R_AtomJmp, R_TapePtr, 0),
|
||||
|
||||
atom_dbg_skip MipsAtomComp_(ac_yield) {
|
||||
load_word(R_AtomJmp, R_TapePtr, 0), LdSlot_
|
||||
add_ui_self( R_TapePtr, S_(MipsCode)),
|
||||
jump_reg( R_AtomJmp),
|
||||
nop,
|
||||
jump_reg( R_AtomJmp), BdSlot_ nop,
|
||||
};
|
||||
|
||||
/* Words: 3; Loads 3 S2 indices from the face array */
|
||||
MipsAtomComp_(ac_load_tri_indices) {
|
||||
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
||||
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
||||
atom_dbg_skip MipsAtomComp_(ac_yield_load) {
|
||||
load_word(R_AtomJmp, R_TapePtr, 0),
|
||||
};
|
||||
|
||||
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
|
||||
MipsAtomComp_(ac_load_tri_verts) {
|
||||
shift_lleft(R_AT, R_T0, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
shift_lleft(R_AT, R_T1, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
||||
shift_lleft(R_AT, R_T2, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
||||
atom_dbg_skip MipsAtomComp_(ac_yield_tail) {
|
||||
add_ui_self(R_TapePtr, S_(MipsCode)),
|
||||
jump_reg( R_AtomJmp), BdSlot_ nop,
|
||||
};
|
||||
|
||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
|
||||
* Hardcoded for Poly_F3 (5 words). For Poly_G4, use ac_insert_ot_tag_g4. */
|
||||
MipsAtomComp_(ac_insert_ot_tag_f3) {
|
||||
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
|
||||
add_u_self( R_T1, R_OtBase), // T1 = & OrderingTable[OTZ]
|
||||
load_word( R_AT, R_T1, O_(PolyTag,bf_addr_len)), // AT = old_ot_head
|
||||
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << polytag_len_bits), // V0 = (5 - 1) << 24 = 4 << 24
|
||||
mask_upper( R_AT, R_AT, S_(polytag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
|
||||
or_u( R_AT, R_AT, R_V0), // Merge length
|
||||
store_word( R_AT, R_PrimCursor, O_(PolyTag,bf_addr_len)), // prim->tag = packed(prim_length, old_addr)
|
||||
shift_lleft( R_AT, R_PrimCursor, S_(polytag_len_bits)), // AT = (prim_length << 24) | old_addr
|
||||
shift_lright(R_AT, R_AT, S_(polytag_len_bits)),
|
||||
store_word( R_AT, R_T1, O_(PolyTag,bf_addr_len)), // OrderingTable[OTZ] = PrimCursor
|
||||
};
|
||||
|
||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
|
||||
* Hardcoded for Poly_G4 (9 words). For Poly_F3, use ac_insert_ot_tag_f3. */
|
||||
MipsAtomComp_(ac_insert_ot_tag_g4) {
|
||||
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
|
||||
add_u_self( R_T1, R_OtBase), // T1 = & OrderingTable[OTZ]
|
||||
load_word( R_AT, R_T1, O_(PolyTag,bf_addr_len)), // AT = old_ot_head
|
||||
load_upper_i(R_V0, (S_(Poly_G4)/S_(U4) - S_(PolyTag)/S_(U4)) << polytag_len_bits), // V0 = (9 - 1) << 24 = 8 << 24
|
||||
mask_upper( R_AT, R_AT, S_(polytag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
|
||||
or_u( R_AT, R_AT, R_V0), // Merge length
|
||||
store_word( R_AT, R_PrimCursor, O_(PolyTag,bf_addr_len)), // prim->tag = packed(prim_length, old_addr)
|
||||
shift_lleft( R_AT, R_PrimCursor, S_(polytag_len_bits)), // AT = (prim_length << 24) | old_addr
|
||||
shift_lright(R_AT, R_AT, S_(polytag_len_bits)),
|
||||
store_word( R_AT, R_T1, O_(PolyTag,bf_addr_len)), // OrderingTable[OTZ] = PrimCursor
|
||||
};
|
||||
|
||||
/* Words: 3; Emits one (cmd|color) word to R_PrimCursor at the given
|
||||
* byte offset. Internal helper used by the *_format_*_color macros.
|
||||
* Args: off = U4 byte offset, code = GP0 cmd byte (0 for c1/c2/c3 of
|
||||
* a Poly_G4), r/g/b = 8-bit RGB byte values. */
|
||||
FI_ MipsAtom ac_pack_color_word(U4 off, U4 code, U1 r, U1 g, U1 b)
|
||||
MipsAtomComp_Proc_(ac_pack_color_word, {
|
||||
load_upper_i(R_AT, (code) << 8 | (b)),
|
||||
or_i_self( R_AT, ((g) << 8) | (r)),
|
||||
store_word( R_AT, R_PrimCursor, (off)),
|
||||
})
|
||||
|
||||
/* Words: 3; Emits the F3 command+color word (cmd byte | BLUE | GREEN | RED)
|
||||
* Args: _r, _g, _b are 8-bit RGB byte values (not raw 16-bit fields).
|
||||
* Migrated from hello_gte_tape.c; takes RGB form per the Phase 3 convention. */
|
||||
FI_ MipsAtom ac_format_f3_color(U1 r, U1 g, U1 b)
|
||||
MipsAtomComp_Proc_(ac_format_f3_color, { mac_pack_color_word(O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
|
||||
|
||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
|
||||
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
|
||||
MipsAtomComp_(ac_gte_store_f3_post_rtpt) {
|
||||
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_F3,p0)),
|
||||
gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_F3,p1)),
|
||||
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_F3,p2)),
|
||||
};
|
||||
|
||||
/* Words: 12; Emits the four (code|color) words of a Poly_G4.
|
||||
* Args: rN,gN,bN are 8-bit RGB byte values for each of the 4 vertices. */
|
||||
FI_ MipsAtom ac_format_g4_color(
|
||||
U1 r0, U1 g0, U1 b0,
|
||||
U1 r1, U1 g1, U1 b1,
|
||||
U1 r2, U1 g2, U1 b2,
|
||||
U1 r3, U1 g3, U1 b3)
|
||||
MipsAtomComp_Proc_(ac_format_g4_color, {
|
||||
mac_pack_color_word(O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0),
|
||||
mac_pack_color_word(O_(Poly_G4,c1), 0, r1,g1,b1),
|
||||
mac_pack_color_word(O_(Poly_G4,c2), 0, r2,g2,b2),
|
||||
mac_pack_color_word(O_(Poly_G4,c3), 0, r3,g3,b3),
|
||||
})
|
||||
|
||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
|
||||
* G4 triangle portion to p0/p1/p2.
|
||||
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen,
|
||||
* SXY2=v2.screen). MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2
|
||||
* get overwritten with v3 (RTPS writes only to SXY2, but to keep the
|
||||
* three registers aligned with v0/v1/v2 you must store before RTPS).
|
||||
* The macro name declares the pipeline position; check #6 (GTE state-
|
||||
* machine validation) verifies the call site matches the declaration. */
|
||||
MipsAtomComp_(ac_gte_store_g4_p012_post_rtpt_pre_rtps) {
|
||||
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_G4,p0)),
|
||||
gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_G4,p1)),
|
||||
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2)),
|
||||
};
|
||||
|
||||
/* Words: 1; Stores the V3 screen coord to the G4's p3 slot.
|
||||
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its
|
||||
* single-vertex result to SXY2; SXY0 still holds v0.screen from the
|
||||
* earlier RTPT — DO NOT read SXY0 here, that's the bug this name
|
||||
* prevents).
|
||||
*/
|
||||
MipsAtomComp_(ac_gte_store_g4_p3_post_rtps) { gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p3)) };
|
||||
|
||||
#pragma endregion Macro Atom Components
|
||||
|
||||
#pragma region Mips Atom Builder
|
||||
// This allows for runtime procedural authoring of mips atoms.
|
||||
#pragma region Atom Builder
|
||||
// This helps with runtime procedural authoring of mips atoms.
|
||||
typedef Relative_(FArena) Struct_(AtomBuilder) { U4 start; U4 capacity; U4 used; };
|
||||
|
||||
typedef Struct_(FMipsAtom512) { U4 data[512]; U4 used; };
|
||||
// Usual way to resolve an atom after the bulder is done.
|
||||
#define atom_from_atombuilder(ab) C_(MipsAtom*, (ab).start)
|
||||
|
||||
// FArena Related
|
||||
typedef Relative_(FArena) Struct_(MipsAtomBuilder) { U4 start; U4 capacity; U4 used; };
|
||||
// Whatever the builder is writting to should most likely coresspond
|
||||
// to something that can fit within instruction cache?
|
||||
|
||||
FI_ void atombuilder_unroll(MipsAtomBuilder_R ab, Slice_MipsCode_R code) {
|
||||
assert(ab->capacity - ab->used - code->len);
|
||||
mem_copy(ab->start, u4_(code->ptr), code->len);
|
||||
mem_bump(ab->start, ab->capacity, & ab->used, code->len);
|
||||
FI_ void atombuilder_push(AtomBuilder_R ab, Slice_MipsCode code) {
|
||||
assert(ab->capacity - ab->used - code.len);
|
||||
U4 dest = ab->start + ab->used * S_(MipsCode); U4 size = S_slice(code);
|
||||
mem_copy(dest, u4_(code.ptr), size); ab->used += size;
|
||||
}
|
||||
#define atombuilder_unroll_mac(ab, mac) atombuilder_unroll(ab, slice_arg_from_array(Slice_MipsCode, mac))
|
||||
#define atombuilder_push_mac(ab, mac) atombuilder_push(ab, slice_arg_from_array(Slice_MipsCode, mac))
|
||||
|
||||
// When done authoring, utilize this to cap-off the atom
|
||||
FI_ void atombuilder_end(MipsAtomBuilder_R ab) {
|
||||
LP_ MipsAtom_(yield) { mac_yield() };
|
||||
mem_copy(ab->start, u4_(code_yield), S_(code_yield));
|
||||
mem_bump(ab->start, ab->capacity, & ab->used, S_(code_yield));
|
||||
}
|
||||
|
||||
#define mipsatom_from_builder(ab) (MipsAtom){ab.start, ab.used}
|
||||
// When done authoring, utilize this to cap-off the atom (if not utilizing a MipsAtom_Proc).
|
||||
FI_ void atombuilder_end(AtomBuilder_R ab) { atombuilder_push(ab, slice_from_array(MipsCode, ac_yield)); }
|
||||
|
||||
FI_ void tb_emit_atombuilder(TapeBuilder_R tb, AtomBuilder_R ab) { tb_emit(tb, atom_from_atombuilder(ab[0])); }
|
||||
#pragma endregion Mips Atom Builder
|
||||
|
||||
#pragma region Atom Arena
|
||||
// Just a dedicated FArena that is meant to mem_copy and return atom definitions made with MipsAtom_Proc_
|
||||
typedef Relative_(FArena) Struct_(AtomArena) { U4 start; U4 capacity; U4 used; };
|
||||
|
||||
#define atomarena_unused_start(ab) ((ab).start + (ab).used)
|
||||
FI_ void atomarena_init(AtomArena_R arena, Slice mem) { assert(arena != nullptr);
|
||||
arena->start = u4_(mem.ptr);
|
||||
arena->capacity = mem.len;
|
||||
arena->used = 0;
|
||||
}
|
||||
FI_ AtomArena atomarena_make(Slice mem) { AtomArena a; atomarena_init(& a, mem); return a; }
|
||||
FI_ MipsAtom* atomarena_push(AtomArena_R aa, Slice_MipsCode code) {
|
||||
assert(aa->capacity - aa->used - code.len);
|
||||
U4 dest = atomarena_unused_start(aa[0]); U4 size = S_slice(code);
|
||||
mem_copy(dest, u4_(code.ptr), size); aa->used += size;
|
||||
return C_(MipsAtom*, dest);
|
||||
}
|
||||
FI_ void atomarena_reset(AtomArena_R aa) { aa->used = 0; }
|
||||
#pragma endregion Atom Arena
|
||||
|
||||
#pragma region RegFile (Register File Allocator)
|
||||
// A specialized allocator utilized to help the user track which registers are bound to values
|
||||
// that must be preserved for the arena's bounds.
|
||||
// TODO(Ed): Technically we can do this at comp-time with the metaprogram, but we may have namespace conflicts.
|
||||
// Unless we follow a convention for #define <Scope_Prefix> or something per register allocation boundary.
|
||||
|
||||
/* ABI reserves that are never handed out by alloc.
|
||||
* R_AT is the assembler temporary (per the MIPS O32 ABI).
|
||||
* R_K0/K1 are kernel reserves.
|
||||
* R_GP stays the host global pointer.
|
||||
* R_SP/R_FP/R_RA are tape runtime carriers between tape_enter and tape_exit. */
|
||||
U4 const regfile_abi_mask =
|
||||
(1u << R_0) | (1u << R_AT) |
|
||||
(1u << R_K0) | (1u << R_K1) |
|
||||
(1u << R_GP) | (1u << R_SP) |
|
||||
(1u << R_FP) | (1u << R_RA);
|
||||
|
||||
internal Reg const regfile_alloc_order[] = {
|
||||
R_V0, R_V1,
|
||||
R_A0, R_A1, R_A2, R_A3,
|
||||
R_T0, R_T1, R_T2, R_T3, R_T4, R_T5, R_T6, R_T7,
|
||||
R_S0, R_S1, R_S2, R_S3, R_S4, R_S5, R_S6, R_S7,
|
||||
R_T8, R_T9,
|
||||
};
|
||||
|
||||
typedef Struct_(RegFile) {
|
||||
A2_U2 GPR;
|
||||
A2_U2 GTE;
|
||||
};
|
||||
#define regfile(pin_mask) {.GPR={u4_lo(pin_mask), u4_hi(pin_mask)} }
|
||||
FI_ void regfile_init(RegFile_R rf) {
|
||||
/* pack the 32-bit ABI mask into the two U2s */
|
||||
rf->GPR[0] = u4_lo(regfile_abi_mask);
|
||||
rf->GPR[1] = u4_hi(regfile_abi_mask);
|
||||
rf->GTE[0] = rf->GTE[1] = 0;
|
||||
}
|
||||
FI_ RegFile regfile_make(void) { RegFile rf; regfile_init(& rf); return rf; }
|
||||
|
||||
typedef Struct_(RegFile_RInfo) {
|
||||
U2_R section;
|
||||
U2 mask;
|
||||
B2 occupied;
|
||||
};
|
||||
FI_ RegFile_RInfo regfile_rinfo(A2_U2 file, Reg r_id) {
|
||||
U2 s_id = r_id >> 4;
|
||||
U2_R section = & file[s_id];
|
||||
U2 mask = u2_(1u << (r_id & 15));
|
||||
B2 occupied = (section[0] & mask) != 0;
|
||||
return (RegFile_RInfo){section, mask, occupied};
|
||||
}
|
||||
FI_ Reg regfile__alloc_helper(A2_U2 file, Reg r_id) {
|
||||
Reg result = 0; RegFile_RInfo info = regfile_rinfo(file, r_id);
|
||||
if (info.occupied == false) {
|
||||
info.section[0] |= info.mask;
|
||||
result = r_id;
|
||||
}
|
||||
return result;
|
||||
}
|
||||
I_ Reg regfile_alloc(RegFile_R rf) {
|
||||
Reg allocated = 0;
|
||||
for index_iter(U4, r_id, R_V0, <, R_T9) {
|
||||
allocated = regfile__alloc_helper(rf->GPR, r_id);
|
||||
Jmp_nZero_(allocated,resolved);
|
||||
}
|
||||
assert(allocated != 0);
|
||||
resolved: return allocated;
|
||||
}
|
||||
FI_ Reg regfile_pin(RegFile_R rf, Reg r_id) {
|
||||
RegFile_RInfo info = regfile_rinfo(rf->GPR, r_id);
|
||||
assert(info.occupied == false);
|
||||
info.section[0] |= info.mask;
|
||||
return r_id;
|
||||
}
|
||||
FI_ void regfile_pin_mask(RegFile_R rf, U4 mask) {
|
||||
B4 occupied = u4_r(rf->GPR)[0] & mask;
|
||||
assert(occupied == false);
|
||||
u4_r(rf->GPR)[0] |= mask;
|
||||
}
|
||||
FI_ void regfile_free_mask(RegFile_R rf, U4 mask) {
|
||||
if (regfile_abi_mask & mask) return;
|
||||
u4_r(rf->GPR)[0] &= ~mask;
|
||||
}
|
||||
FI_ void regfile_free_reg(RegFile_R rf, Reg r_id) {
|
||||
/* never free the ABI set */
|
||||
if (regfile_abi_mask & (1u << r_id)) return;
|
||||
RegFile_RInfo info = regfile_rinfo(rf->GPR, r_id);
|
||||
info.section[0] &= ~info.mask;
|
||||
}
|
||||
FI_ void regfile_reset(RegFile_R rf) {
|
||||
rf->GPR[0] = u4_lo(regfile_abi_mask);
|
||||
rf->GPR[1] = u4_hi(regfile_abi_mask);
|
||||
}
|
||||
FI_ void regfile_reset_to_mask(RegFile_R rf, U4 mask) {
|
||||
rf->GPR[0] = u4_lo(mask);
|
||||
rf->GPR[1] = u4_hi(mask);
|
||||
}
|
||||
#pragma endregion RegFileArena (Register File Allocator)
|
||||
|
||||
#pragma region Mips Atom Procs
|
||||
/* RegUse structs are a convention to organize register allocations for a mips atom procedure.
|
||||
Unlike the usual enum-based declarations, they provide a namespaced scope and have view types via union declarations. */
|
||||
#define RegUse_(proc_name) (tmpl(RegUse,proc_name))
|
||||
|
||||
typedef Struct_(RegUse_example_atom_proc) {
|
||||
Reg const ro_register; // Scratch base carrier.
|
||||
Reg usual_modifiable;
|
||||
union { Reg view_1, view_2, view_3; } t1;
|
||||
};
|
||||
internal MipsAtom* example_atom_proc(AtomArena_R aa, U2 offset, RegUse_example_atom_proc r)
|
||||
MipsAtom_Proc_(aa, {
|
||||
add_si(r.usual_modifiable, r.ro_register, offset),
|
||||
or_u(r.t1.view_1, r.ro_register, 0),
|
||||
branch_lt_zero(r.t1.view_1, atom_offset(example_atom_proc, skip)), BdSlot_ nop,
|
||||
li_s(r.t1.view_2, 100),
|
||||
atom_label(skip)
|
||||
add_si(r.t1.view_3, r.usual_modifiable, 10),
|
||||
mac_yield(),
|
||||
})
|
||||
|
||||
#pragma endregion Mips Atom Procs
|
||||
|
||||
#pragma region Baked Mips Atoms
|
||||
// These atoms are resolved at compile time and are (usually) statically linked readonly data.
|
||||
|
||||
enum {
|
||||
bios_flushcache = 0x44,
|
||||
bios_table_addr = 0xA0,
|
||||
};
|
||||
|
||||
/* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0).
|
||||
* Sequence (per MIPS ABI; arguments in arg registers, RA pushed to stack):
|
||||
* 1. sp -= 8; sw $ra, 4($sp) ; save RA
|
||||
* 2. $a0 = bios_flushcache (arg0)
|
||||
* 3. $t0 = bios_table_addr ; t0 = &BIOS A-function table
|
||||
* 4. jalr $t0, $ra ; call BIOS(flushcache)
|
||||
* nop ; branch delay slot
|
||||
* 5. lw $ra, 4($sp); jr $ra ; restore & return
|
||||
* 6. sp += 8
|
||||
*/
|
||||
internal MipsAtom_(mips_flush_icache) {
|
||||
add_ui(rstack_ptr, rstack_ptr, -MipsStackAlignment), // sp -= 8
|
||||
store_word(rret_addr, rstack_ptr, S_(U4)), // sw $ra, 4($sp)
|
||||
add_ui(rret_0, rdiscard, bios_flushcache), // addiu $a0, $0, 0x44
|
||||
add_ui(rtmp_0, rdiscard, bios_table_addr), // addiu $t0, $0, 0xA0
|
||||
jump_link(rtmp_0, rret_addr), // jalr $t0, $ra
|
||||
nop, // BD slot
|
||||
load_word(rret_addr, rstack_ptr, S_(U4)), // lw $ra, 4($sp)
|
||||
jump_reg(rret_addr), // jr $ra
|
||||
add_ui(rstack_ptr, rstack_ptr, MipsStackAlignment), // sp += 8 (BD)
|
||||
mac_yield(),
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_SetGteWorld) {
|
||||
M3_S2* transform;
|
||||
};
|
||||
internal MipsAtom_(set_gte_world) {
|
||||
/* Pop matrix address from tape into R_T3 ($11) */
|
||||
load_word(R_T3, R_TapePtr, O_(Binds_SetGteWorld,transform)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_SetGteWorld)),
|
||||
/* Load 3x3 Rotation + 3x1 Translation from R_T3 into GTE CONTROL Regs (ctc2) */
|
||||
load_word(R_T0, R_T3, 0), load_word(R_T1, R_T3, 4),
|
||||
gte_mv_to_ctrl_r(R_T0, gte_cr_RT11), gte_mv_to_ctrl_r(R_T1, gte_cr_RT12),
|
||||
load_word(R_T0, R_T3, 8), load_word(R_T1, R_T3, 12), load_word(R_T2, R_T3, 16),
|
||||
gte_mv_to_ctrl_r(R_T0, gte_cr_RT13), gte_mv_to_ctrl_r(R_T1, gte_cr_RT21), gte_mv_to_ctrl_r(R_T2, gte_cr_RT22),
|
||||
load_word(R_T0, R_T3, 20), load_word(R_T1, R_T3, 24), load_word(R_T2, R_T3, 28),
|
||||
gte_mv_to_ctrl_r(R_T0, gte_cr_TRX), gte_mv_to_ctrl_r(R_T1, gte_cr_TRY), gte_mv_to_ctrl_r(R_T2, gte_cr_TRZ),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
/* DIAGNOSTIC 1: Pure tape loop test */
|
||||
internal MipsAtom_(diag_yield) { mac_yield() };
|
||||
|
||||
// TODO(Ed): Reduce magic numbers/offsets
|
||||
/* DIAGNOSTIC 2: Pure memory test (No GTE). Draws a fixed cyan triangle. */
|
||||
internal MipsAtom_(diag_color) {
|
||||
store_word( R_0, R_T7, 0),
|
||||
load_upper_i(R_AT, gp0_cmd_poly_f3 << 8 | 0xFF), /* High: MipsCode Poly_F3(0x20) + Color B:FF */
|
||||
or_i_self( R_AT, 0xFF00), /* Low: Color G:FF, R:00 (Cyan) */
|
||||
store_word( R_AT, R_T7, 4),
|
||||
|
||||
/* Fake coordinates - Swapped winding order to prevent GPU culling! */
|
||||
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 8), /* (16, 16) */
|
||||
load_upper_i(R_AT, 0x0050), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 12), /* (80, 16) */
|
||||
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0050), store_word(R_AT, R_T7, 16), /* (16, 80) */
|
||||
|
||||
add_ui( R_T1, R_0, 10),
|
||||
shift_lleft_self(R_T1, S_(U4)/2),
|
||||
add_u_self( R_T1, R_T6),
|
||||
|
||||
load_word( R_AT, R_T1, 0),
|
||||
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << polytag_len_bits),
|
||||
store_word( R_AT, R_T7, 0),
|
||||
shift_lleft(R_AT, R_T7, S_(polytag_len_bits)), shift_lright(R_AT, R_AT, S_(polytag_len_bits)),
|
||||
or_u_self( R_AT, R_V0),
|
||||
store_word( R_AT, R_T1, 0),
|
||||
|
||||
add_ui(R_T7, R_T7, 20),
|
||||
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
// TODO(Ed): Reduce magic numbers/offsets
|
||||
/* DIAGNOSTIC 3: Pure GTE test (No Memory Writes) */
|
||||
internal MipsAtom_(diag_gte) {
|
||||
/* Load 3 indices */
|
||||
load_half_u(R_T0, R_T4, 0),
|
||||
load_half_u(R_T1, R_T4, 2),
|
||||
load_half_u(R_T2, R_T4, 4),
|
||||
|
||||
/* Load Vertices into GTE */
|
||||
shift_lleft( R_AT, R_T0, 3), add_u( R_AT, R_AT, R_T5),
|
||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
|
||||
shift_lleft( R_AT, R_T1, 3), add_u(R_AT, R_AT, R_T5),
|
||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
||||
|
||||
shift_lleft(R_AT, R_T2, 3), add_u(R_AT, R_AT, R_T5),
|
||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
||||
|
||||
/* Run Math */
|
||||
nop2, gte_cmdw_rtpt,
|
||||
nop2, gte_cmdw_nclip,
|
||||
nop2,
|
||||
|
||||
/* Advance Face Cursor and Yield */
|
||||
add_ui(R_T4, R_T4, 8),
|
||||
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
#pragma endregion Baked Mips Atoms
|
||||
|
||||
@@ -0,0 +1,96 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# include "gen/macs.h"
|
||||
# include "gen/offsets.h"
|
||||
# include "math.h"
|
||||
# include "lottes_tape.h"
|
||||
#endif
|
||||
|
||||
ATOM_FILE_DEBUGGER_LINE_MARKER(math_atom_c);
|
||||
|
||||
#define v3s4_R_0() ((Reg_(V3_S4)){R_0,R_0,R_0})
|
||||
|
||||
typedef Struct_(Reg_V3_S2) { Reg x, y, z; };
|
||||
typedef Struct_(Reg_V3_S4) { Reg x, y, z; }; // Register allocation of a V3_S4
|
||||
typedef Struct_(Reg_P3_S4) { Reg x, y, z; }; // Register allocation of a P3_S4
|
||||
|
||||
#pragma region MACs (Mips Atom Component)
|
||||
|
||||
FI_ Slice_MipsCode ac_load_half_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
load_half(tx, base, offset + OA_(U2,[0])),
|
||||
load_half(ty, base, offset + OA_(U2,[1])),
|
||||
load_half(tz, base, offset + OA_(U2,[2])),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_load_v3s2(AtomBuilder_R ab, Reg_(V3_S2) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_half_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||
|
||||
FI_ Slice_MipsCode ac_load_v2s2(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
load_half(rs_x, r_base, offset + O_(V3_S2,x)),
|
||||
load_half(rs_y, r_base, offset + O_(V3_S2,y)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_store_v2s2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
store_half(rt_x, base, offset + O_(V2_S2,x)),
|
||||
store_half(rt_y, base, offset + O_(V2_S2,y)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_load_word_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
load_word(tx, base, offset + OA_(U4,[0])),
|
||||
load_word(ty, base, offset + OA_(U4,[1])),
|
||||
load_word(tz, base, offset + OA_(U4,[2])),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_load_v3s4(AtomBuilder_R ab, Reg_(V3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||
FI_ Slice_MipsCode ac_load_p3s4(AtomBuilder_R ab, Reg_(P3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||
|
||||
FI_ Slice_MipsCode ac_store_half_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
store_half(tx, base, offset + OA_(U2,[0])),
|
||||
store_half(ty, base, offset + OA_(U2,[1])),
|
||||
store_half(tz, base, offset + OA_(U2,[2])),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_store_v3s2(AtomBuilder_R ab, Reg_(V3_S2) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_store_half_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||
|
||||
FI_ Slice_MipsCode ac_store_word_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
store_word(tx, base, offset + OA_(U4,[0])),
|
||||
store_word(ty, base, offset + OA_(U4,[1])),
|
||||
store_word(tz, base, offset + OA_(U4,[2])),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_store_v3s4(AtomBuilder_R ab, Reg_(V3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||
FI_ Slice_MipsCode ac_store_p3s4(AtomBuilder_R ab, Reg_(P3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||
|
||||
FI_ Slice_MipsCode ac_add_si_v3s4(AtomBuilder_R ab, Reg rt_x, Reg rt_y, Reg rt_z, Reg base, U2 offset)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
add_si(rt_x, base, O_(V3_S4,x)),
|
||||
add_si(rt_y, base, O_(V3_S4,y)),
|
||||
add_si(rt_z, base, O_(V3_S4,z)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_sub_s_v3(AtomBuilder_R ab
|
||||
, Reg dx, Reg dy, Reg dz
|
||||
, Reg sx, Reg sy, Reg sz
|
||||
, Reg tx, Reg ty, Reg tz
|
||||
) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
sub_s(dx, sx, tx),
|
||||
sub_s(dy, sy, ty),
|
||||
sub_s(dz, sz, tz),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_sub_v3s4(AtomBuilder_R ab, Reg_(V3_S4) d, Reg_(V3_S4) s, Reg_(V3_S4) t) MipsAtomComp_ProcMap_(ab, mac_sub_s_v3(d.x, d.y, d.z, s.x, s.y, s.z, t.x, t.y, t.z))
|
||||
|
||||
FI_ Slice_MipsCode ac_sub_s_v3_self(AtomBuilder_R ab, Reg ds_x, Reg ds_y, Reg ds_z, Reg tx, Reg ty, Reg tz) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
sub_s(ds_x, ds_x, tx),
|
||||
sub_s(ds_y, ds_y, ty),
|
||||
sub_s(ds_z, ds_z, tz),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_sub_v3s4_self(AtomBuilder_R ab, Reg_(V3_S4) ds, Reg_(V3_S4) t) MipsAtomComp_ProcMap_(ab, mac_sub_s_v3_self(ds.x, ds.y, ds.z, t.x, t.y, t.z))
|
||||
|
||||
FI_ Slice_MipsCode ac_store_rects2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
store_half(rt_x, base, offset + O_(Rect_S2,x)),
|
||||
store_half(rt_y, base, offset + O_(Rect_S2,y)),
|
||||
store_half(rt_width, base, offset + O_(Rect_S2,width)),
|
||||
store_half(rt_height, base, offset + O_(Rect_S2,height)),
|
||||
})
|
||||
|
||||
#pragma endregion MACs (Mips Atom Component)
|
||||
+65
-11
@@ -7,10 +7,24 @@
|
||||
#define max(A, B) (((A) > (B)) ? (A) : (B))
|
||||
#define clamp_bot(X, B) max(X, B)
|
||||
|
||||
/* Convention
|
||||
<Type> ## <Width> _ <Component Type> ## <Component Width>
|
||||
For types with compound data (Ex: Rotation Matrix & Translation):
|
||||
<TypeA> ## <TypeB> ## <Width> _ <ComponentTypeA> ## <ComponentWidthA> ## <ComponentTypeB> ## <ComponentWidthB>
|
||||
|
||||
A: Array
|
||||
V: Vector
|
||||
R: Range
|
||||
M: Matrix
|
||||
T: Translation
|
||||
*/
|
||||
|
||||
enum {
|
||||
v3s2_byteoff = 3, // log2(8), used with shift_left_logical op for index via byte offset.
|
||||
};
|
||||
|
||||
typedef Array_(U1, 2);
|
||||
typedef Array_(U2, 2);
|
||||
typedef Array_(U4, 2);
|
||||
typedef Array_(S2, 2);
|
||||
typedef Array_(S2, 3);
|
||||
@@ -22,24 +36,46 @@ typedef S2 A3x3_S2[3][3];
|
||||
typedef Struct_(Extent2_S2) { S2 width; S2 height; };
|
||||
typedef Struct_(Extent2_S4) { S4 width; S4 height; };
|
||||
|
||||
typedef Struct_(V2_U1) { U1 x; U1 y; };
|
||||
typedef Struct_(V2_S2) { S2 x; S2 y; };
|
||||
typedef Struct_(V2_S4) { S4 x; S4 y; };
|
||||
typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; };
|
||||
typedef Struct_(V3_S4) { S4 x; S4 y; S4 z; S4 pad; };
|
||||
typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; }; // PSY-Q: SVECTOR
|
||||
typedef Struct_(V3_S4) { S4 x; S4 y; S4 z; S4 pad; }; // PSY-Q: VECTOR. RGA(Lengyel): Euclidean vector or direction. A zero-weight RGA point is stored as a V3_S4 with the implicit weight dropped.
|
||||
typedef Struct_(V4_S2) { S2 x; S2 y; S2 z; S2 w; };
|
||||
typedef Struct_(V4_S4) { S4 x; S4 y; S4 z; S4 w; };
|
||||
|
||||
typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; };
|
||||
typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; };
|
||||
// typedef Struct_(P3_S4) { S4 x; S4 y; S4 z; S4 w1; }; // RGA(Lengyel): Affine point with implicit weight one. Storage alias of V3_S4. Use P3_S4 when the value is a point.
|
||||
typedef V3_S4 P3_S4;
|
||||
|
||||
typedef Struct_(Rect_S2) { S2 x; S2 y; S2 width; S2 height; };
|
||||
typedef Struct_(Rect_S4) { S4 x; S4 y; S4 width; S4 height; };
|
||||
typedef Struct_(R1_U2) { U2 p0; U2 p1; };
|
||||
typedef Struct_(R1_S2) { S2 p0; S2 p1; };
|
||||
|
||||
typedef Struct_(M3_S2) { A3x3_S2 m; A3_S4 t; };
|
||||
typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; }; // Range-2 Signed 2-Byte (16-bit)
|
||||
typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; }; // Range-2 Signed 4-Byte (32-bit)
|
||||
|
||||
typedef Struct_(Rect_S2) { S2 x; S2 y; S2 width; S2 height; };
|
||||
typedef Struct_(Rect_S4) { S4 x; S4 y; S4 width; S4 height; };
|
||||
|
||||
typedef Struct_(MT3_S2S4) { A3x3_S2 m; A3_S4 t; }; // PSY-Q: MATRIX. RGA(Lengyel): Matrix expansion of a rigid transformation. GTE utilizes this representation; corresponding motor not constructed here.
|
||||
|
||||
/* RGA(Lengyel) reserved names (deferred):
|
||||
* P4_S4 - future flat point with explicit weight (Lengyel/TML FlatPoint3D analog).
|
||||
* B3_S4 - future 3D bivector (callers store a Complement(Wedge(...)) as a V3_S4).
|
||||
* Mo8_S4 - future motor. Not introduced until a course operation actually needs composition, interpolation, or inversion. */
|
||||
|
||||
typedef Array_(V2_U1, 2);
|
||||
typedef Array_(V2_S2, 2);
|
||||
typedef Array_(V2_S2, 3);
|
||||
typedef Array_(V2_S2, 4);
|
||||
|
||||
#define r1u2(p0,p1) (R1_U2){p0,p1}
|
||||
|
||||
enum {
|
||||
fp_one = (1 << 12),
|
||||
};
|
||||
|
||||
#define v3s4_fp_one() v3s4(fp_one, fp_one, fp_one)
|
||||
|
||||
#define v2s2(x,y) (V2_S2){x,y}
|
||||
#define v3s2(x,y,z) (V3_S2){x,y,z,0}
|
||||
#define v3s4(x,y,z) (V3_S4){x,y,z,0}
|
||||
@@ -58,10 +94,28 @@ FI_ void add_a3s4_fp(A3_S4_R out_a, A3_S4 b) {
|
||||
(out_a[0])[2] += b[2] >> 1;
|
||||
}
|
||||
|
||||
FI_ void add_v3s4(V3_S4_R out_a, V3_S4 b) {
|
||||
add_a3s4(pcast(A3_S4_R, out_a), pcast(A3_S4, b));
|
||||
FI_ void sub_a3s4(A3_S4_R out_a, A3_S4 b) {
|
||||
(out_a[0])[0] -= b[0];
|
||||
(out_a[0])[1] -= b[1];
|
||||
(out_a[0])[2] -= b[2];
|
||||
}
|
||||
|
||||
FI_ void add_v3s4_fp(V3_S4_R out_a, V3_S4 b) {
|
||||
add_a3s4_fp(pcast(A3_S4_R, out_a), pcast(A3_S4, b));
|
||||
FI_ void sub_a3s4_fp(A3_S4_R out_a, A3_S4 b) {
|
||||
(out_a[0])[0] -= b[0] >> 1;
|
||||
(out_a[0])[1] -= b[1] >> 1;
|
||||
(out_a[0])[2] -= b[2] >> 1;
|
||||
}
|
||||
|
||||
FI_ void mul_a3s4(A3_S4_R out_a, A3_S4 b) {
|
||||
(out_a[0])[0] *= b[0];
|
||||
(out_a[0])[1] *= b[1];
|
||||
(out_a[0])[2] *= b[2];
|
||||
}
|
||||
|
||||
FI_ void add_v3s4 (V3_S4_R out_a, V3_S4 b) { add_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
||||
FI_ void add_v3s4_fp(V3_S4_R out_a, V3_S4 b) { add_a3s4_fp(C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
||||
|
||||
FI_ void sub_v3s4 (V3_S4_R out_a, V3_S4 b) { sub_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
||||
FI_ void sub_v3s4_fp(V3_S4_R out_a, V3_S4 b) { sub_a3s4_fp(C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
||||
|
||||
FI_ void mul_v3s4 (V3_S4_R out_a, V3_S4 b) { mul_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
||||
|
||||
+36
-15
@@ -18,7 +18,7 @@ I_ U4 align_pow2(U4 x, U4 b) {
|
||||
|
||||
#define align_struct(type_width) ((U4)(((type_width) + 3) & ~3))
|
||||
|
||||
FI_ void mem_bump(U4 start, U4 cap, U4*R_ used, U4 amount) {
|
||||
FI_ void mem_bump(U4 cap, U4*R_ used, U4 amount) {
|
||||
assert(amount <= (cap - used[0]));
|
||||
used[0] += amount;
|
||||
}
|
||||
@@ -58,37 +58,44 @@ typedef Struct_(Str8) { UTF8* ptr; U4 len; };
|
||||
typedef Struct_(Slice_Str8) { Str8* ptr; U4 len; };
|
||||
#define slit(string_literal) (Str8){ (UTF8*) string_literal, S_(string_literal) - 1 }
|
||||
|
||||
typedef Struct_(Slice) { U4 ptr, len; }; // Untyped Slice
|
||||
FI_ Slice slice_ut_(U4 ptr, U4 len) { return (Slice){ptr, len}; }
|
||||
typedef Struct_(Slice) { B1* ptr; U4 len; }; // Untyped Slice (byte-addressable; .len in elements)
|
||||
FI_ Slice slice_ut_(U4 ptr, U4 len) { return (Slice){(B1*)ptr, len}; }
|
||||
|
||||
#define Slice_(type) Struct_(tmpl(Slice,type)) { type* ptr; U4 len; }
|
||||
typedef Slice_(B1);
|
||||
#define slice_assert(s) do { assert((s).ptr != 0); assert((s).len > 0); } while(0)
|
||||
#define slice_end(slice) ((slice).ptr + (slice).len)
|
||||
#define slice_end(slice) ((slice).ptr + S_slice(slice) / S_(B1)) /* byte-ptr arithmetic; .len is in elements per slice convention */
|
||||
#define S_slice(s) ((s).len * S_((s).ptr[0]))
|
||||
|
||||
#define slice_ut(ptr,len) slice_ut_(u4_(ptr), u4_(len))
|
||||
#define slice_ut_arr(a) slice_ut_(u4_(a), S_(a))
|
||||
#define slice_ut(ptr,len) slice_ut_(u4_(ptr), u4_(len))
|
||||
#define slice_ut_arr(a) slice_ut_(u4_(a), S_(a))
|
||||
#define slice_to_ut(s) slice_ut_(u4_((s).ptr), S_slice(s))
|
||||
|
||||
#define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter)
|
||||
#define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = array_decl(type,__VA_ARGS__), .len = array_len( array_decl(type,__VA_ARGS__)) }
|
||||
#define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = S_(array) }
|
||||
#define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = Array_decl(type,__VA_ARGS__), .len = Array_len( Array_decl(type,__VA_ARGS__)) }
|
||||
#define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = Array_len(array) }
|
||||
|
||||
FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(s.ptr, s.len); }
|
||||
FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(u4_(s.ptr), s.len); }
|
||||
#define slice_zero(s) slice_zero_(slice_to_ut(s))
|
||||
|
||||
FI_ void slice_copy_(Slice dest, Slice src) {
|
||||
assert(dest.len >= src.len);
|
||||
assert(S_slice(dest) >= S_slice(src));
|
||||
slice_assert(dest);
|
||||
slice_assert(src);
|
||||
mem_copy(dest.ptr, src.ptr, src.len);
|
||||
mem_copy(u4_(dest.ptr), u4_(src.ptr), S_slice(src));
|
||||
}
|
||||
#define slice_copy(dest, src) do { \
|
||||
static_assert(T_same(dest, src)); \
|
||||
slice_copy_(slice_to_ut(dest), slice_to_ut(src)); \
|
||||
} while(0)
|
||||
|
||||
FI_ Slice slice_bump(U4_R used, U4 start, U4 len, U4 amount) {
|
||||
assert(len - used[0] - amount);
|
||||
U4 ptr = start + used[0]; used[0] += amount;
|
||||
return slice_ut(ptr, amount);
|
||||
}
|
||||
|
||||
typedef Slice_(U1);
|
||||
typedef Slice_(U4);
|
||||
|
||||
#pragma endregion Slice
|
||||
@@ -98,18 +105,19 @@ typedef Slice_(U4);
|
||||
typedef Opt_(farena) { U4 alignment, type_width; };
|
||||
typedef Struct_(FArena) { U4 start, capacity, used; };
|
||||
FI_ void farena_init(FArena_R arena, Slice mem) { assert(arena != nullptr);
|
||||
arena->start = mem.ptr;
|
||||
arena->start = u4_(mem.ptr);
|
||||
arena->capacity = mem.len;
|
||||
arena->used = 0;
|
||||
}
|
||||
FI_ FArena farena_make(Slice mem) { FArena a; farena_init(& a, mem); return a; }
|
||||
I_ Slice farena_push(FArena_R arena, U4 amount, Opt_farena o) {
|
||||
FI_ Slice farena_bump(FArena_R a, U4 amount) { return slice_bump(& a->used, a->start, a->capacity, amount); }
|
||||
I_ Slice farena_push(FArena_R arena, U4 amount, Opt_farena o) {
|
||||
if (amount == 0) { return (Slice){}; }
|
||||
U4 desired = amount * (o.type_width == 0 ? 1 : o.type_width);
|
||||
U4 to_commit = align_pow2(desired, o.alignment ? o.alignment : MEM_ALIGNMENT_DEFAULT);
|
||||
U4 ptr = arena->start + arena->used;
|
||||
mem_bump(arena->start, arena->capacity, & arena->used, to_commit);
|
||||
return (Slice){ ptr, to_commit };
|
||||
mem_bump(arena->capacity, & arena->used, to_commit);
|
||||
return (Slice){ (B1*)ptr, to_commit };
|
||||
}
|
||||
FI_ void farena_reset (FArena_R arena) { arena->used = 0; }
|
||||
FI_ void farena_rewind(FArena_R arena, U4 save_point) {
|
||||
@@ -117,8 +125,21 @@ FI_ void farena_rewind(FArena_R arena, U4 save_point) {
|
||||
arena->used -= save_point - arena->start;
|
||||
}
|
||||
FI_ U4 farena_save(FArena arena) { return arena.used; }
|
||||
FI_ U4 farena_unused_start(FArena arena) { return arena.start + arena.used; }
|
||||
#define farena_push_(arena, amount, ...) farena_push((arena), (amount), opt_(farena, __VA_ARGS__))
|
||||
#define farena_push_type(arena, type, ...) C_(type*, farena_push((arena), 1, opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr)
|
||||
#define farena_push_array(arena, type, amount, ...) (tmpl(Slice,type)){ C_(type*, farena_push((arena), (amount), opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr), (amount) }
|
||||
|
||||
#pragma endregion FArena
|
||||
|
||||
#pragma region BIOS Scratchpad
|
||||
/* BIOS scratchpad location. 1 KB at 0x1F800000.
|
||||
* TapeHostFrame occupies the final 44 bytes while tape code executes.
|
||||
* Atom scratch is bounded by the TapeHostFrame_Loc declaration in lottes_tape.h. */
|
||||
enum {
|
||||
Scratchpad_Loc = 0x1F800000,
|
||||
Scratchpad_Len = 0x400, /* 1 KB */
|
||||
Scratchpad_End = Scratchpad_Loc + Scratchpad_Len, /* 0x1F800400 */
|
||||
};
|
||||
#define C_scratch(type) C_(type, Scratchpad_Loc)
|
||||
#pragma endregion BIOS Scratchpad
|
||||
|
||||
@@ -0,0 +1,80 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# include "gen/macs.h"
|
||||
# include "gen/offsets.h"
|
||||
# include "bios.h"
|
||||
# include "mips.h"
|
||||
# include "lottes_tape.h"
|
||||
#endif
|
||||
|
||||
ATOM_FILE_DEBUGGER_LINE_MARKER(mips_atom_c);
|
||||
|
||||
#pragma region MACs (Mips Atom Components)
|
||||
|
||||
FI_ Slice_MipsCode ac_load_word_imm(AtomBuilder_R ab, Reg dst, U4 imm)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
load_upper_i(dst, u4_hi(imm)),
|
||||
or_i_self( dst, u4_lo(imm)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_shift_aright_v3_self(AtomBuilder_R ab, Reg dt_x, Reg dt_y, Reg dt_z, U2 shift_amount)
|
||||
MipsAtomComp_Proc_( ab, {
|
||||
shift_aright(dt_x, dt_x, shift_amount),
|
||||
shift_aright(dt_y, dt_y, shift_amount),
|
||||
shift_aright(dt_z, dt_z, shift_amount),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_shift_aright_v3s4_self(AtomBuilder_R ab, Reg_(V3_S4) dt, U2 shift) MipsAtomComp_ProcMap_(ab, mac_shift_aright_v3_self(dt.x, dt.y, dt.z, shift))
|
||||
|
||||
FI_ Slice_MipsCode ac_shift_aright_var_v3(AtomBuilder_R ab
|
||||
, Reg rd_v0, Reg rd_v1, Reg rd_v2
|
||||
, Reg rs_v0, Reg rs_v1, Reg rs_v2
|
||||
, Reg r_shift)
|
||||
MipsAtomComp_Proc_(ab, {
|
||||
shift_aright_var(rd_v0, rs_v0, r_shift),
|
||||
shift_aright_var(rd_v1, rs_v1, r_shift),
|
||||
shift_aright_var(rd_v2, rs_v2, r_shift),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_shift_aright_var_v3_self(AtomBuilder_R ab, Reg rds_v0, Reg rds_v1, Reg rds_v2, Reg r_shift)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
shift_aright_var(rds_v0, rds_v0, r_shift),
|
||||
shift_aright_var(rds_v1, rds_v1, r_shift),
|
||||
shift_aright_var(rds_v2, rds_v2, r_shift),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_shift_aright_var_v3s4_self(AtomBuilder_R ab, Reg_(V3_S4) ds, Reg shift) MipsAtomComp_ProcMap_(ab, mac_shift_aright_var_v3_self(ds.x, ds.y, ds.z, shift))
|
||||
|
||||
#pragma endregion MACs (Mips Atom Components)
|
||||
|
||||
#pragma region Baked Atoms
|
||||
|
||||
/* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0).
|
||||
* Sequence (per MIPS ABI; arguments in arg registers, RA pushed to stack):
|
||||
* 1. sp -= 8; sw $ra, 4($sp) ; save RA
|
||||
* 2. $a0 = bios_flushcache (arg0)
|
||||
* 3. $t0 = bios_table_addr ; t0 = &BIOS A-function table
|
||||
* 4. jalr $t0, $ra ; call BIOS(flushcache)
|
||||
* nop ; branch delay slot
|
||||
* 5. lw $ra, 4($sp)
|
||||
* 6. sp += 8 ; load-delay
|
||||
* 7. jr $ra
|
||||
* nop ; BD
|
||||
*/
|
||||
|
||||
#if 0
|
||||
// Note: Can't do this without having a way to do C-Runtime frame call from Tape ABI.
|
||||
// Don't support this without adjusting scratchpad to save tape frame in some way.
|
||||
internal MipsAtom_(mips_flush_icache) {
|
||||
add_ui(R_SP, R_SP, -MipsStackAlignment), // sp -= 8
|
||||
store_word(R_RA, R_SP, S_(U4)), // sw $ra, 4($sp)
|
||||
add_ui(R_V0, R_0, bios_flushcache), // addiu $a0, $0, 0x44
|
||||
add_ui(R_T0, R_0, bios_table_addr), // addiu $t0, $0, 0xA0
|
||||
jump_link(R_T0, R_RA), nop, // jalr $t0, $ra, BD slot
|
||||
load_word(R_RA, R_SP, S_(U4)), // lw $ra, 4($sp)
|
||||
add_ui(R_SP, R_SP, MipsStackAlignment), // sp += 8 (load-delay)
|
||||
jump_reg(R_RA), nop, // jr $ra, BD slot
|
||||
// mac_yield(),
|
||||
};
|
||||
#endif
|
||||
|
||||
#pragma endregion Baked Atoms
|
||||
+156
-170
@@ -1,38 +1,28 @@
|
||||
/* ============================================================================
|
||||
* duffle DSL Suffix Conventions
|
||||
* ============================================================================
|
||||
*
|
||||
* Every mnemonic in this header follows the same suffix grammar:
|
||||
*
|
||||
* _i Immediate value (16-bit constant operand). Combine with
|
||||
* _u or _s (single-letter modifier + type combined): add_ui,
|
||||
* add_si. Examples: add_ui, add_si, and_i, or_i, xor_i,
|
||||
* load_upper_i. and_i is sign-agnostic (andi zero-extends).
|
||||
* load_upper_i is a unique verb; _i is the immediate marker,
|
||||
* not a modifier+type combination.
|
||||
*
|
||||
* _u Unsigned (no-overflow, no-sign-extension). R-type
|
||||
* arithmetic examples: add_u, sub_u, mult_u, div_u. I-type
|
||||
* (combined with _i): add_ui.
|
||||
*
|
||||
* _s Signed (overflow-traps, sign-extends). R-type: add_s,
|
||||
* sub_s, mult_s, div_s, set_lt_s. I-type (combined with _i):
|
||||
* add_si.
|
||||
* _i: Immediate value (16-bit constant operand).
|
||||
* Combine with _u or _s (single-letter modifier + type combined): add_ui, add_si.
|
||||
* Examples: add_ui, add_si, and_i, or_i, xor_i, load_upper_i. and_i is sign-agnostic (andi zero-extends).
|
||||
* load_upper_i is a unique verb; _i is the immediate marker, not a modifier+type combination.
|
||||
* _u: Unsigned (no-overflow, no-sign-extension).
|
||||
* R-type arithmetic examples: add_u, sub_u, mult_u, div_u. I-type (combined with _i): add_ui.
|
||||
* _s: Signed (overflow-traps, sign-extends).
|
||||
* R-type: add_s, sub_s, mult_s, div_s, set_lt_s. I-type (combined with _i): add_si.
|
||||
*
|
||||
* --- Shift family (R-type): verb-modifier-direction ---
|
||||
* The shift macros use `shift_<modifier><direction>`. Modifier is
|
||||
* the single letter `l` (logical) or `a` (arithmetic). Direction
|
||||
* is the word `left` or `right`. Combined: `_lleft`, `_lright`,
|
||||
* `_aright`. Examples: shift_lleft( rd, rt, shamt) (= sll)
|
||||
* shift_lright(rd, rt, shamt) (= srl)
|
||||
* shift_aright(rd, rt, shamt) (= sra)
|
||||
* (no `_aleft`; MIPS has no `sla` — arithmetic-left is bit-identical
|
||||
* to logical-left, so use shift_lleft for that case)
|
||||
* The shift macros use `shift_<modifier><direction>`.
|
||||
* Modifier is the single letter `l` (logical) or `a` (arithmetic).
|
||||
* Direction is the word `left` or `right`. Combined: `_lleft`, `_lright`, `_aright`.
|
||||
* Examples: shift_lleft( rd, rt, shamt) (= sll)
|
||||
* shift_lright(rd, rt, shamt) (= srl)
|
||||
* shift_aright(rd, rt, shamt) (= sra)
|
||||
* (no `_aleft`; MIPS has no `sla` — arithmetic-left is bit-identical to logical-left, so use shift_lleft for that case)
|
||||
*
|
||||
* --- Jump/Call family ---
|
||||
* Simple jumps keep the original short names: jump (j), jump_reg
|
||||
* (jr), jump_link (jalr rs, rd). The jump-and-link-to variants
|
||||
* (jal, jalr rs with default $ra) get the `call_` verb instead:
|
||||
* Simple jumps keep the original short names: jump (j), jump_reg (jr), jump_link (jalr rs, rd).
|
||||
* The jump-and-link-to variants (jal, jalr rs with default $ra) get the `call_` verb instead:
|
||||
* call_addr (jal), call_reg (jalr rs, default $ra).
|
||||
* Examples: jump(off) (= j)
|
||||
* jump_reg(rs) (= jr)
|
||||
@@ -40,32 +30,22 @@
|
||||
* call_reg(rs) (= jalr rs, default $ra)
|
||||
* call_addr(off) (= jal)
|
||||
*
|
||||
* _r Register marker — used only when the register type needs
|
||||
* disambiguation (e.g., GTE data register vs control
|
||||
* register). NOT used in plain R-type arithmetic (the
|
||||
* R-type is implicit). Examples: gte_mv_to_data_r,
|
||||
* gte_mv_to_ctrl_r.
|
||||
* _r: Register marker — used only when the register type needs disambiguation (e.g., GTE data register vs control register).
|
||||
* NOT used in plain R-type arithmetic (the R-type is implicit). Examples: gte_mv_to_data_r, gte_mv_to_ctrl_r.
|
||||
* _self: Destination equals one source operand.
|
||||
* Examples: add_ui_self (I-type, to self), add_u_self (R-type, to self).
|
||||
* _mv_to_: Direction: data flows into X.
|
||||
* Example: gte_mv_to_data_r, gte_mv_to_ctrl_r.
|
||||
* _mv_from_: Direction: data flows out of X.
|
||||
* Example: gte_mv_from_data_r, gte_mv_from_ctrl_r.
|
||||
* _str: String-form — emits inline-asm string instead of `.word`.
|
||||
* Example: gte_rtpt_asm_str.
|
||||
* _2w / _1w: Word count of the emitted sequence.
|
||||
* Example: load_imm_2w.
|
||||
*
|
||||
* _self Destination equals one source operand.
|
||||
* Examples: add_ui_self (I-type, to self),
|
||||
* add_u_self (R-type, to self).
|
||||
*
|
||||
* _mv_to_ Direction: data flows into X.
|
||||
* Example: gte_mv_to_data_r, gte_mv_to_ctrl_r.
|
||||
*
|
||||
* _mv_from_ Direction: data flows out of X.
|
||||
* Example: gte_mv_from_data_r, gte_mv_from_ctrl_r.
|
||||
*
|
||||
* _str String-form — emits inline-asm string instead of `.word`.
|
||||
* Example: gte_rtpt_asm_str.
|
||||
*
|
||||
* _2w / _1w Word count of the emitted sequence.
|
||||
* Example: load_imm_2w.
|
||||
*
|
||||
* _cop2 RESERVED — DO NOT USE in macro names. The `gte_` namespace
|
||||
* prefix already implies coprocessor 2. Use `c2` only in:
|
||||
* (a) integer opcode enums (op_lwc2 = 0x32, op_swc2 = 0x3A)
|
||||
* (b) vendor-mnemonic macro aliases (gte_mtc2, gte_mfc2)
|
||||
* _cop2: RESERVED — DO NOT USE in macro names. The `gte_` namespace prefix already implies coprocessor 2. Use `c2` only in:
|
||||
* (a) integer opcode enums (op_lwc2 = 0x32, op_swc2 = 0x3A)
|
||||
* (b) vendor-mnemonic macro aliases (gte_mtc2, gte_mfc2)
|
||||
*
|
||||
* Primitive commands: gp0_cmd_poly_f3 = 0x20 (byte opcode)
|
||||
* Packed 32-bit cmd: gp0_word_poly_f3(r, g, b) (32-bit, shifted)
|
||||
@@ -80,9 +60,8 @@
|
||||
* gte_lw_v0_xy(base) (gte + lw + v0 + xy)
|
||||
* load_upper_i (load-upper + immediate, unique verb)
|
||||
*
|
||||
* Vendor mnemonics (sll, srl, sra, jr, j, jal, jalr) are NOT in this
|
||||
* header. They live in the opt-in `mips_vendor_sym.h` for users who
|
||||
* prefer the textbook MIPS assembly mnemonics.
|
||||
* Vendor mnemonics (sll, srl, sra, jr, j, jal, jalr) are NOT in this header.
|
||||
* They live in the opt-in `mips_vendor_sym.h` for users who prefer the textbook MIPS assembly mnemonics.
|
||||
* ============================================================================ */
|
||||
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
@@ -98,19 +77,17 @@ enum {
|
||||
/* ============================================================================
|
||||
* REGISTER INTEGER IDS (preprocessor-visible)
|
||||
* ============================================================================
|
||||
* Every R_* enum below has a parallel R_*_Code `#define` so that the
|
||||
* preprocessor can stringify the integer (e.g. for asm clobber lists and
|
||||
* register-variable declarations via `rgcc(R_X)`). The enum value is
|
||||
* bound to the `#define` so the two forms cannot drift apart.
|
||||
* Every R_* enum below has a parallel R_*_Code `#define` so that the preprocessor can stringify the integer
|
||||
* (e.g. for asm clobber lists and register-variable declarations via `rgcc(R_X)`).
|
||||
* The enum value is bound to the `#define` so the two forms cannot drift apart.
|
||||
*
|
||||
* Only registers that get stringified need a `_Code` form; the rest are
|
||||
* plain enum values. If you need to add a new one, follow the pattern:
|
||||
* Only registers that get stringified need a `_Code` form; the rest are plain enum values.
|
||||
* If you need to add a new one, follow the pattern:
|
||||
* #define R_T7_Code 15
|
||||
* R_T7 = R_T7_Code, // in the enum
|
||||
* R_T7 = R_T7_Code, // in the enum
|
||||
*
|
||||
* User code should always reference the enum form (`R_T4`) at arithmetic
|
||||
* sites and let `rlit(R_T4_Code)` / `rgcc(R_T4)` handle the stringify
|
||||
* cases — never write the bare number `12`.
|
||||
* User code should always reference the enum form (`R_T4`) at arithmetic sites and let
|
||||
* `rlit(R_T4_Code)` / `rgcc(R_T4)` handle the stringify cases — never write the bare number `12`.
|
||||
* ============================================================================ */
|
||||
#define R_0_Code 0
|
||||
#define R_AT_Code 1
|
||||
@@ -159,31 +136,31 @@ enum {
|
||||
|
||||
/* Semantic Aliases for MIPS Registers (O32 ABI) */
|
||||
|
||||
, rdiscard = R_0 /* Hardwired to 0 */
|
||||
, rasm_tmp = R_AT /* Assembler temporary (destroyed by some assembler pseudoinstructions!) */
|
||||
, rret_0 = R_V0 /* Function return value */
|
||||
, rret_1 = R_V1 /* Second return value (e.g., 64-bit) */
|
||||
, rarg_0 = R_A0 /* First function argument */
|
||||
, rarg_1 = R_A1 /* Second function argument */
|
||||
, rarg_2 = R_A2 /* Third function argument */
|
||||
, rarg_3 = R_A3 /* Fourth function argument */
|
||||
, rtmp_0 = R_T0 /* Temporary (Caller saved) */
|
||||
, rtmp_1 = R_T1 /* Temporary (Caller saved) */
|
||||
, rtmp_2 = R_T2 /* Temporary (Caller saved) */
|
||||
, rtmp_3 = R_T3 /* Temporary (Caller saved) */
|
||||
, rtmp_4 = R_T4 /* Temporary (Caller saved) — common GTE base pointer */
|
||||
, rtmp_9 = R_T9 /* Temporary (Caller saved) — common GTE base pointer */
|
||||
, rstatic_0 = R_S0 /* Static (Callee saved, preserved across calls) */
|
||||
, rstatic_1 = R_S1
|
||||
, rstatic_2 = R_S2
|
||||
, rstatic_3 = R_S3
|
||||
, rstatic_4 = R_S4
|
||||
, rstatic_5 = R_S5
|
||||
, rstatic_6 = R_S6
|
||||
, rstatic_7 = R_S7
|
||||
, rsaved_0 = R_S0 /* Alias for rstatic_0 (alternate vocabulary) */
|
||||
, rstack_ptr = R_SP /* Stack Pointer */
|
||||
, rret_addr = R_RA /* Return Address (populated by JAL) */
|
||||
// , rdiscard = R_0 /* Hardwired to 0 */
|
||||
// , rasm_tmp = R_AT /* Assembler temporary (destroyed by some assembler pseudoinstructions!) */
|
||||
// , rret_0 = R_V0 /* Function return value */
|
||||
// , rret_1 = R_V1 /* Second return value (e.g., 64-bit) */
|
||||
// , rarg_0 = R_A0 /* First function argument */
|
||||
// , rarg_1 = R_A1 /* Second function argument */
|
||||
// , rarg_2 = R_A2 /* Third function argument */
|
||||
// , rarg_3 = R_A3 /* Fourth function argument */
|
||||
// , rtmp_0 = R_T0 /* Temporary (Caller saved) */
|
||||
// , rtmp_1 = R_T1 /* Temporary (Caller saved) */
|
||||
// , rtmp_2 = R_T2 /* Temporary (Caller saved) */
|
||||
// , rtmp_3 = R_T3 /* Temporary (Caller saved) */
|
||||
// , rtmp_4 = R_T4 /* Temporary (Caller saved) — common GTE base pointer */
|
||||
// , rtmp_9 = R_T9 /* Temporary (Caller saved) — common GTE base pointer */
|
||||
// , rstatic_0 = R_S0 /* Static (Callee saved, preserved across calls) */
|
||||
// , rstatic_1 = R_S1
|
||||
// , rstatic_2 = R_S2
|
||||
// , rstatic_3 = R_S3
|
||||
// , rstatic_4 = R_S4
|
||||
// , rstatic_5 = R_S5
|
||||
// , rstatic_6 = R_S6
|
||||
// , rstatic_7 = R_S7
|
||||
// , rsaved_0 = R_S0 /* Alias for rstatic_0 (alternate vocabulary) */
|
||||
// , rstack_ptr = R_SP /* Stack Pointer */
|
||||
// , rret_addr = R_RA /* Return Address (populated by JAL) */
|
||||
|
||||
/* --- MIPS CPU Opcodes (Bits 31-26) --- */
|
||||
|
||||
@@ -225,7 +202,6 @@ enum {
|
||||
/* 2F: N/A */
|
||||
// , op_lwc0
|
||||
|
||||
|
||||
// , op_load_addr = op_la
|
||||
// , op_load_imm = op_li
|
||||
, op_jump = op_j
|
||||
@@ -276,29 +252,29 @@ enum {
|
||||
enum { _BitOffsets = 0
|
||||
/* Bit Offsets for MIPS Instruction Fields */
|
||||
|
||||
, OPCODE_SHIFT = 26
|
||||
, RS_SHIFT = 21
|
||||
, RT_SHIFT = 16
|
||||
, RD_SHIFT = 11
|
||||
, SHAMT_SHIFT = 6 /* Shift Amount */
|
||||
, FC_SHIFT = 0
|
||||
, OPCODE_POS = 26
|
||||
, RS_POS = 21
|
||||
, RT_POS = 16
|
||||
, RD_POS = 11
|
||||
, SHAMT_POS = 6 /* Shift Amount: Offset Position */
|
||||
, FC_POS = 0
|
||||
|
||||
/* Bit Masks to prevent overflow into adjacent fields */
|
||||
/* IMM_MASK is the 16-bit two's-complement truncation for the immediate field.
|
||||
* It is NOT a range guard — it is load-bearing for negative branch offsets
|
||||
* (the metaprogram emits raw signed offsets; the mask truncates them to the
|
||||
* 16-bit representation the hardware expects). The static analysis
|
||||
* `immediate_field_width` check validates ranges at build time. */
|
||||
|
||||
, OPCODE_MASK = 0x3F
|
||||
, REG_MASK = 0x1F
|
||||
, SHAMT_MASK = 0x1F /* Shift Amount */
|
||||
, FC_MASK = 0x3F
|
||||
, IMM_MASK = 0xFFFF
|
||||
};
|
||||
|
||||
#define enc_op(op) (((op) & OPCODE_MASK) << OPCODE_SHIFT)
|
||||
#define enc_rs(rs) (((rs) & REG_MASK) << RS_SHIFT)
|
||||
#define enc_rt(rt) (((rt) & REG_MASK) << RT_SHIFT)
|
||||
#define enc_rd(rd) (((rd) & REG_MASK) << RD_SHIFT)
|
||||
#define enc_shamt(shamt) (((shamt) & SHAMT_MASK) << SHAMT_SHIFT)
|
||||
#define enc_fc(fc) (((fc) & FC_MASK) << FC_SHIFT)
|
||||
#define enc_imm(imm) (((imm) & IMM_MASK))
|
||||
#define enc_op(op) ((op) << OPCODE_POS)
|
||||
#define enc_rs(rs) ((rs) << RS_POS)
|
||||
#define enc_rt(rt) ((rt) << RT_POS)
|
||||
#define enc_rd(rd) ((rd) << RD_POS)
|
||||
#define enc_shamt(shamt) ((shamt) << SHAMT_POS)
|
||||
#define enc_fc(fc) ((fc) << FC_POS)
|
||||
#define enc_imm(imm) ((imm) & IMM_MASK)
|
||||
|
||||
/* MIPS R-Type Instruction Format (Register-to-Register) */
|
||||
#define enc_r(op, rs, rt, rd, shamt, fc) (enc_op(op) | enc_rs(rs) | enc_rt(rt) | enc_rd(rd) | enc_shamt(shamt) | enc_fc(fc))
|
||||
@@ -327,22 +303,25 @@ enum { _BitOffsets = 0
|
||||
* Argument order matches the MIPS assembly syntax:
|
||||
* dest-first, then source operands, then immediate last.
|
||||
*
|
||||
* load_word(rt, base, off) → lw rt, off(base)
|
||||
* store_word(rt, base, off) → sw rt, off(base)
|
||||
* add_ui(rt, rs, imm) → addiu rt, rs, imm
|
||||
* shift_lleft(rd, rt, shamt) → sll rd, rt, shamt
|
||||
* shift_lright(rd, rt, shamt) → srl rd, rt, shamt
|
||||
* shift_aright(rd, rt, shamt) → sra rd, rt, shamt
|
||||
* jump_reg(rs) → jr rs
|
||||
* jump_link(rs, rd) → jalr rs (link in rd, default $ra)
|
||||
* nop → sll $0, $0, 0
|
||||
* load_word(rt, base, off) → lw rt, off(base)
|
||||
* store_word(rt, base, off) → sw rt, off(base)
|
||||
* add_ui(rt, rs, imm) → addiu rt, rs, imm
|
||||
* shift_lleft(rd, rt, shamt) → sll rd, rt, shamt
|
||||
* shift_lright(rd, rt, shamt) → srl rd, rt, shamt
|
||||
* shift_aright(rd, rt, shamt) → sra rd, rt, shamt
|
||||
* jump_reg(rs) → jr rs
|
||||
* jump_link(rs, rd) → jalr rs (link in rd, default $ra)
|
||||
* nop → sll $0, $0, 0
|
||||
*/
|
||||
#define load_word(rt, base, off) enc_i(op_lw, (base), (rt), (off))
|
||||
#define load_byte(rt, base, off) enc_i(op_lb, (base), (rt), (off))
|
||||
#define load_half(rt, base, off) enc_i(op_lh, (base), (rt), (off))
|
||||
#define load_byte_u(rt, base, off) enc_i(op_lbu, (base), (rt), (off))
|
||||
#define load_half_u(rt, base, off) enc_i(op_lhu, (base), (rt), (off))
|
||||
#define LdSlot_
|
||||
|
||||
#define store_word(rt, base, off) enc_i(op_sw, (base), (rt), (off))
|
||||
|
||||
#define add_ui(rt, rs, imm) enc_i(op_addiu, (rs), (rt), (imm))
|
||||
#define and_i(rt, rs, imm) enc_i(op_andi, (rs), (rt), (imm))
|
||||
// #define and_si and_i
|
||||
@@ -360,10 +339,10 @@ enum { _BitOffsets = 0
|
||||
|
||||
/* Logic Opcodes */
|
||||
|
||||
#define and_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_and)
|
||||
#define or_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_or)
|
||||
#define xor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_xor)
|
||||
#define nor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_nor)
|
||||
#define and_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_and)
|
||||
#define or_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_or)
|
||||
#define xor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_xor)
|
||||
#define nor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_nor)
|
||||
|
||||
#define or_u_self(rd_rs, rt) enc_r(op_special, (rd_rs), (rt), (rd_rs), 0, fc_or)
|
||||
|
||||
@@ -372,6 +351,12 @@ enum { _BitOffsets = 0
|
||||
#define shift_lright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_srl)
|
||||
#define shift_aright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_sra)
|
||||
|
||||
/* Shift Variable — register-shift forms.
|
||||
* shift_lleft_var(rd, rt, rs) → sllv rd, rt, rs (shamt in low 5 bits of rs)
|
||||
* shift_aright_var(rd, rt, rs) → srav rd, rt, rs */
|
||||
#define shift_lleft_var(rd, rt, rs) enc_r(op_special, (rs), (rt), (rd), 0, fc_sllv)
|
||||
#define shift_aright_var(rd, rt, rs) enc_r(op_special, (rs), (rt), (rd), 0, fc_srav)
|
||||
|
||||
#define shift_lleft_self(rd_rt, shamt) enc_r(op_special, R_0, (rd_rt), (rd_rt), (shamt), fc_sll)
|
||||
|
||||
#define mask_upper(rd, rt, shamt) shift_lleft(rd, rt, shamt), shift_lright(rd, rt, shamt)
|
||||
@@ -386,10 +371,29 @@ enum { _BitOffsets = 0
|
||||
/* call_reg rs — jump-and-link to register-held address; link in $ra. */
|
||||
#define call_reg(rs) jump_link((rs), R_RA)
|
||||
|
||||
/* j target — absolute jump within the current 256MB region. */
|
||||
/* j target — absolute jump within the current 256MB region.
|
||||
* WARNING: `jump(off)` CANNOT BE USED for within-atom jumps in the current pipeline.
|
||||
* The MIPS j opcode encodes `(target_addr >> 2)` in its 26-bit immediate field; an ABSOLUTE byte address, not a relative word offset.
|
||||
* The metaprogram computes `off` as a relative word offset (`target_word_idx - branch_word_idx - 1`), which the assembler/linker does NOT resolve.
|
||||
* `jump(off)` is only safe when the BUILD PIPELINE owns the absolute position of the emitted code — i.e. when: s
|
||||
* - the build emits a symbol-relative `.word` expression that the linker resolvess via `R_MIPS_26`, OR
|
||||
* - the code is hand-assembled with explicit absolute targets, OR a custom post-build patcher resolves the 26-bit field.
|
||||
* TODO(Ed): Review this.. technically we can resolve aboslute jumps on baked atoms? (Even proedurally generated ones...)
|
||||
*/
|
||||
#define jump(off) enc_i(op_j, R_0, R_0, (off))
|
||||
|
||||
/* call_addr off — jump-and-link to immediate address. */
|
||||
// Annotate an instruction as filling a branch-delay slot.
|
||||
#define BdSlot_
|
||||
|
||||
/* jump_rel off — unconditional relative jump (the within-atom-safe `jump`).
|
||||
* MIPS I R3000A has no "branch always" opcode. The idiom for an unconditional relative jump is `beq $0, $0, off`. */
|
||||
#define jump_rel(off) branch_equal(R_0, R_0, (off))
|
||||
|
||||
/* call_addr off — jump-and-link to immediate address.
|
||||
* Same WARNING as `jump(off)` above: the jal opcode also encodes an absolute 26-bit target.
|
||||
* For within-atom calls, the current pipeline has no equivalent always-taken call-and-link idiom.
|
||||
* Workaround: `branch_link` (always-taken branch + explicit `la $ra, next_word_addr; jr $ra`), or just use `call_reg($tmp)` after loading the target into a register.
|
||||
*/
|
||||
#define call_addr(off) enc_i(op_jal, R_0, R_0, (off))
|
||||
|
||||
/* --- Store family (mirrors the load family) --- */
|
||||
@@ -403,16 +407,7 @@ enum { _BitOffsets = 0
|
||||
* sub_s / sub_u → sub / subu
|
||||
* mult_s / mult_u → mult / multu (writes HI/LO; result in LO)
|
||||
* div_s / div_u → div / divu (LO = quot, HI = rem)
|
||||
*
|
||||
* NOTE: dsl.h defines `add_s`/`sub_s`/`mut_s`/`gt_s`/etc. as
|
||||
* _Generic-based signed integer-arithmetic helpers for U1/U2/U4. Those
|
||||
* live in a different conceptual layer (generic arithmetic on DSL
|
||||
* types) and would collide with the instruction encoders here. The
|
||||
* `#undef` below lets the gas-style names below win; if a file needs
|
||||
* both, the dsl.h versions can be reached via their long forms
|
||||
* (e.g. `def_signed_op`-style or the underlying `add_s1/s2/s4`). */
|
||||
#undef add_s
|
||||
#undef sub_s
|
||||
*/
|
||||
#define add_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_add)
|
||||
#define add_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_addu)
|
||||
#define sub_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_sub)
|
||||
@@ -422,6 +417,7 @@ enum { _BitOffsets = 0
|
||||
#define div_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_div)
|
||||
#define div_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_divu)
|
||||
|
||||
// TODO(Ed): Change convention of 'self' to ds for (destination is source)?
|
||||
#define add_u_self(rd_rs, rt) add_u(rd_rs, rd_rs, rt)
|
||||
|
||||
/* --- Arithmetic I-type (immediate) --- */
|
||||
@@ -441,7 +437,7 @@ enum { _BitOffsets = 0
|
||||
#define mov_to_low(rs) enc_r(op_special, (rs), R_0, R_0, 0, fc_mtlo)
|
||||
|
||||
/* --- Atomic branches (no pseudos like bgt/bge; compose with slt_* + branch_ne) ---
|
||||
* branch_equal rs, rt, off → beq rs, rt, off
|
||||
* branch_equal rs, rt, off → beq rs, rt, off
|
||||
* branch_ne rs, rt, off → bne rs, rt, off
|
||||
* branch_lt_zero rs, off → bltz rs, off
|
||||
* branch_gt_zero rs, off → bgtz rs, off
|
||||
@@ -463,31 +459,28 @@ enum { _BitOffsets = 0
|
||||
/* --- Shift-amount alias (matches the gas convention `\p3 = shamt`) --- */
|
||||
#define shift_amount(rd, rt, n) shift_lleft(rd, rt, n)
|
||||
|
||||
/* nop — canonical sll $0, $0, 0 */
|
||||
#define nop shift_lleft(rdiscard, rdiscard, 0)
|
||||
/* nop — sll $0, $0, 0 */
|
||||
#define nop shift_lleft(R_0, R_0, 0)
|
||||
#define nop2 nop, nop
|
||||
|
||||
// li_s — load signed 16-bit immediate into GPR (addiu rt, $0, imm — sign-extends).
|
||||
#define li_s(rt, imm) add_ui((rt), R_0, (imm))
|
||||
// #define load_imm_s(rt, imm) add_ui((rt), R_0, (imm))
|
||||
|
||||
#define load_imm_1w(rt, imm) add_ui((rt), R_0, (imm))
|
||||
#define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm))
|
||||
|
||||
/* load_imm_2w — unconditional 2-word `li` form: `lui` + (ori | addi).
|
||||
*
|
||||
* Granular companion to `load_imm`: skips the compile-time range checks
|
||||
* and always emits 2 .words. Use this when:
|
||||
* Granular companion to `load_imm`: skips the compile-time range checks and always emits 2 .words. Use this when:
|
||||
* - you know `imm` is > 0xFFFF (otherwise you're wasting a word), OR
|
||||
* - `imm` is not a compile-time constant and you want predictable
|
||||
* 2-word emission without the `__builtin_constant_p` branches.
|
||||
* - `imm` is not a compile-time constant and you want predictable 2-word emission without the `__builtin_constant_p` branches.
|
||||
*
|
||||
* The lo16 strategy is still chosen at expansion time on the lo half:
|
||||
* lo16 in 0x0000..0x7FFF → addi (sign-ext is harmless, the lui
|
||||
* already cleared bits 15..0)
|
||||
* lo16 in 0x8000..0xFFFF → ori (zero-extends to preserve the
|
||||
* intended bit pattern)
|
||||
* lo16 in 0x0000..0x7FFF → addi (sign-ext is harmless, the lui already cleared bits 15..0)
|
||||
* lo16 in 0x8000..0xFFFF → ori (zero-extends to preserve the intended bit pattern)
|
||||
*
|
||||
* For situations where you need to bypass even this choice (e.g. to
|
||||
* force a specific encoding for a known discontiguous high/low pair),
|
||||
* For situations where you need to bypass even this choice (e.g. to force a specific encoding for a known discontiguous high/low pair),
|
||||
* see `load_imm_2w_ori_forced` and `load_imm_2w_addi_forced` below.
|
||||
*
|
||||
* Statement-level (not expression-level): emits its own `asm volatile(...)`.
|
||||
*/
|
||||
#define load_imm_2w(rt, imm) do { \
|
||||
@@ -518,9 +511,8 @@ enum { _BitOffsets = 0
|
||||
} while (0)
|
||||
|
||||
/* load_imm_2w_addi_forced — force the `lui` + `addi` form regardless of lo16 sign.
|
||||
* Use when you know sign-extension is fine (e.g. lo16 is treated as
|
||||
* signed downstream) and you want a smaller effective instruction
|
||||
* (the assembler/MIPS hardware will sign-extend the imm16). */
|
||||
* Use when you know sign-extension is fine (e.g. lo16 is treated as signed downstream)
|
||||
* and you want a smaller effective instruction (the assembler/MIPS hardware will sign-extend the imm16). */
|
||||
#define load_imm_2w_addi_forced(rt, imm) do { \
|
||||
/*U4 _li2a_imm_ = (U4)(imm);*/ \
|
||||
asm volatile(asm_words( \
|
||||
@@ -532,23 +524,17 @@ enum { _BitOffsets = 0
|
||||
|
||||
/* load_imm rt, imm — true `li` semantics (assembler `li` pseudo)
|
||||
*
|
||||
* Dispatches at compile time on the immediate's range, picking the
|
||||
* smallest single-instruction form when possible:
|
||||
*
|
||||
* imm in 0 .. 0x7FFF → addi rt, $0, imm (1 word)
|
||||
* imm in 0x8000 .. 0xFFFF → ori rt, $0, imm (1 word; sign-bit must be zeroed)
|
||||
* imm in 0x10000 .. 0xFFFFFFFF → lui + (ori | addi) (2 words)
|
||||
*
|
||||
* Statement-level (not expression-level): the macro emits its own
|
||||
* `asm volatile(...)` block with 1 or 2 .word constants. Callers can
|
||||
* group multiple `load_imm` calls in a single volatile by using the
|
||||
* lower-level encoders directly:
|
||||
* Dispatches at compile time on the immediate's range, picking the smallest single-instruction form when possible:
|
||||
* imm in 0 .. 0x7FFF → addi rt, $0, imm (1 word)
|
||||
* imm in 0x8000 .. 0xFFFF → ori rt, $0, imm (1 word; sign-bit must be zeroed)
|
||||
* imm in 0x10000 .. 0xFFFFFFFF → lui + (ori | addi) (2 words)
|
||||
*
|
||||
* Statement-level (not expression-level): the macro emits its own `asm volatile(...)` block with 1 or 2 .word constants.
|
||||
* Callers can group multiple `load_imm` calls in a single volatile by using the lower-level encoders directly:
|
||||
* load_imm(R_T4, 0x12345678); // emits 2 .words
|
||||
*
|
||||
* Falls back to a 2-word form if `imm` is not a compile-time constant,
|
||||
* but that path is unusual (load_imm is most useful with literal
|
||||
* addresses and magic numbers). */
|
||||
* Falls back to a 2-word form if `imm` is not a compile-time constant, but that path is unusual
|
||||
* (load_imm is most useful with literal addresses and magic numbers). */
|
||||
#define load_imm(rt, imm) do { \
|
||||
if (cexpr_(imm) && ((imm) <= 0x7FFFU)) { \
|
||||
/* Small positive: addi rt, $0, imm */ \
|
||||
@@ -588,9 +574,8 @@ enum { _BitOffsets = 0
|
||||
|
||||
|
||||
/* Standard clobber list for pure-MIPS asm volatile blocks: caller-saved
|
||||
* GPRs that the kernel treats as volatile (v0/v1/t0/t1/ra) plus the
|
||||
* "memory" barrier. The register ids are passed through `rlit` so
|
||||
* the R_*_Code `#define`s are stringified into "$N" at expansion time. */
|
||||
* GPRs that the kernel treats as volatile (v0/v1/t0/t1/ra) plus the "memory" barrier.
|
||||
* The register ids are passed through `rlit` so the R_*_Code `#define`s are stringified into "$N" at expansion time. */
|
||||
#define clbr_volatile_gprs rlit(R_V0), rlit(R_T0), rlit(R_T1), rlit(R_RA), clb_mem_drain
|
||||
|
||||
#define asm_mips_flush_icache() asm volatile( asm_words( \
|
||||
@@ -601,6 +586,7 @@ enum { _BitOffsets = 0
|
||||
, jump_link(rtmp_0, rret_addr) \
|
||||
, nop \
|
||||
, load_word(rret_addr, rstack_ptr, 4) \
|
||||
, jump_reg(rret_addr) \
|
||||
, add_ui(rstack_ptr, rstack_ptr, MipsStackAlignment) \
|
||||
, jump_reg(rret_addr) \
|
||||
, nop \
|
||||
) asm_clobber: clbr_volatile_gprs )
|
||||
|
||||
@@ -2,9 +2,8 @@
|
||||
* duffle DSL — MIPS Vendor Mnemonics (opt-in)
|
||||
* ============================================================================
|
||||
*
|
||||
* Provides the textbook MIPS assembly mnemonics as thin aliases to the
|
||||
* canonical duffle macros in mips.h. The duffle names are primary; this
|
||||
* header is for users who prefer the textbook mnemonics.
|
||||
* Provides the textbook MIPS assembly mnemonics as thin aliases to the duffle macros in mips.h.
|
||||
* The duffle names are primary; this header is for users who prefer the textbook mnemonics.
|
||||
*
|
||||
* USAGE: #include "duffle/mips_vendor_sym.h" // after mips.h
|
||||
*
|
||||
@@ -21,12 +20,6 @@
|
||||
* jal -> call_addr (jump-and-link to immediate address)
|
||||
* jalr -> call_reg (jump-and-link to register, default $ra)
|
||||
* (for the 2-arg `jalr rs, rd`, use `jump_link(rs, rd)` directly)
|
||||
*
|
||||
* The vendor mnemonics are NOT registered with the duffle word-count
|
||||
* metadata (tape_atom.metadata.h). They expand to the duffle canonical
|
||||
* macros which DO have word-count entries. Verification: V2 (objdump
|
||||
* byte-identical) holds.
|
||||
*
|
||||
* ============================================================================ */
|
||||
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
|
||||
@@ -0,0 +1,196 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# include "gen/macs.h"
|
||||
# include "gen/offsets.h"
|
||||
# include "mips.h"
|
||||
# include "dsl.atom.h"
|
||||
# include "lottes_tape.h"
|
||||
# include "pad.h"
|
||||
#endif
|
||||
|
||||
ATOM_FILE_DEBUGGER_LINE_MARKER(pad_atom_c);
|
||||
|
||||
#pragma region MACs (Mips Atom Components)
|
||||
|
||||
FI_ Slice_MipsCode ac_pad_set_centered_axes(AtomBuilder_R ab, Reg state, Reg scratch) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
load_upper_i(scratch, (PadAxis_Centered >> 16) & 0xFFFF),
|
||||
or_i_self( scratch, PadAxis_Centered & 0xFFFF), // mac_load_word_imm(scratch, PadAxis_Centered),
|
||||
store_word( scratch, state, O_(PadState,axes)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_pad_set_id_byte(AtomBuilder_R ab, Reg state, Reg r_id, U1 id_value) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
add_ui( r_id, R_0, id_value),
|
||||
store_byte(r_id, state, O_(PadState,id)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_pad_set_status(AtomBuilder_R ab, U4 r_tmp, U1 r_state, U4 pad_status) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
add_ui( r_tmp, R_0, pad_status),
|
||||
store_word(r_tmp, r_state, O_(PadState,status)),
|
||||
})
|
||||
|
||||
/* Invert r_buttons (active-low → active-high) and store to PadState.buttons.
|
||||
* r_buttons must already be loaded (the caller is responsible for filling the load-delay slot of
|
||||
* the preceding load_half_u with an instruction that doesn't read r_buttons). */
|
||||
FI_ Slice_MipsCode ac_pad_store_inverted_buttons(AtomBuilder_R ab, U1 r_buttons, U1 r_pad_state) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
nor_u( r_buttons, r_buttons, R_0),
|
||||
store_half(r_buttons, r_pad_state, O_(PadState,buttons)),
|
||||
})
|
||||
|
||||
#pragma endregion MACs (Mips Atom Components)
|
||||
|
||||
#pragma region Baked Atoms
|
||||
|
||||
/* ----- pad_bios_snapshot -----
|
||||
* Per-frame snapshot of one BIOS pad buffer into PadState.
|
||||
* Decoder (branch ladder on raw[0] status + raw[1] id):
|
||||
* 1. raw[0] == 0xFF -> Disconnected (buttons=0, axes=0x80)
|
||||
* 2. raw[0]==0 && raw[1]==0 -> Pending (buttons=0, axes=0x80)
|
||||
* 3. raw[1] == 0x41 -> Digital (buttons normalized; axes=0x80)
|
||||
* 4. raw[1] == 0x53 -> AnalogStick (buttons normalized; axes from raw[4..7])
|
||||
* 5. raw[1] in 0x7x -> AnalogPad (buttons normalized; axes from raw[4..7])
|
||||
* 6. else -> Unsupported (buttons=0, axes=0x80)
|
||||
*
|
||||
* Buttons normalization: byte_swap16((~raw_buttons) & 0xFFFF).
|
||||
* raw_buttons = load_half_u(raw, 2) = raw[2] | (raw[3] << 8).
|
||||
* byte_swap16(x) = (x >> 8) | (x << 8); nor(x, R_0) = ~x. store_half truncates to 16 bits so the upper-16 mask is implicit in the store.
|
||||
*
|
||||
* Register use (atom-local; no wave-context touched):
|
||||
* R_T0 = raw base : Kept throughout; axes loads read raw[4..7] from R_T0.
|
||||
* R_T1 = state base : Kept throughout; all stores go through R_T1.
|
||||
* R_T2 = raw[0] status : Alive across the disc/pending/id dispatch, then dead.
|
||||
* R_T3 = raw[1] id : Alive across the id dispatch, then dead.
|
||||
* R_T4 = scratch : Shifts, compares, immediate loads, store values.
|
||||
* R_T5 = scratch : Parallel lui + ori for the 0x80808080 axes constant + byte-swap target.
|
||||
*/
|
||||
enum {
|
||||
R_PadRaw = R_T0 atom_reg atom_type(U1),
|
||||
R_PadState = R_T1 atom_reg atom_type(PadState*),
|
||||
R_RawStatus = R_T2 atom_reg,
|
||||
R_RawId = R_T3 atom_reg,
|
||||
};
|
||||
typedef Struct_(Binds_PadBiosSnapshot) {
|
||||
PadBiosRaw* raw;
|
||||
PadState* state;
|
||||
};
|
||||
internal MipsAtom_(pad_bios_snapshot) atom_info(atom_bind(Binds_PadBiosSnapshot)
|
||||
, atom_reads( R_PadRaw, R_PadState, R_RawStatus, R_RawId)
|
||||
, atom_writes(R_PadRaw, R_PadState, R_RawStatus, R_RawId)
|
||||
) {
|
||||
/* === Bind consumption: T0 = raw, T1 = state, advance R_TapePtr by 8. */
|
||||
load_word(R_PadRaw, R_TapePtr, O_(Binds_PadBiosSnapshot,raw)),
|
||||
load_word(R_PadState, R_TapePtr, O_(Binds_PadBiosSnapshot,state)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_PadBiosSnapshot)),
|
||||
|
||||
/* === Read raw[0] (status) + raw[1] (id) */
|
||||
load_byte_u(R_RawStatus, R_PadRaw, O_(PadBiosRaw,status)),
|
||||
load_byte_u(R_RawId, R_PadRaw, O_(PadBiosRaw,id)),
|
||||
|
||||
atom_label(snap_root) /* === Case 1: Disconnected (status == 0xFF). */
|
||||
add_ui(R_T4, R_0, PadRawStatus_Timeout), branch_ne(R_RawStatus, R_T4, atom_offset(snap_root, skip_disconnected)),
|
||||
/* BD-slot: pre-compute PadStatus_Disconnected. Branch reads R_T4=0xFF in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to pending/id_dispatch), R_T4 is overwritten by the next case body's add_ui — harmless. */
|
||||
|
||||
atom_label(disconnected) /* === Disconnected body. */
|
||||
mac_pad_set_status(R_T4, R_PadState, PadStatus_Disconnected),
|
||||
store_half( R_0, R_PadState, O_(PadState,buttons)),
|
||||
mac_pad_set_centered_axes(R_PadState, R_T4),
|
||||
mac_pad_set_id_byte(R_PadState, R_RawId, PadRawStatus_Timeout),
|
||||
jump_rel(atom_offset(disconnected, snap_end)),
|
||||
/* BD-slot: load next atom's entry point (replaces the nop).
|
||||
* Always jumps to snap_end, where mac_yield_tail() transfers control to R_AtomJmp without re-loading it. */
|
||||
mac_yield_load(),
|
||||
atom_label(skip_disconnected)
|
||||
|
||||
/* === Case 2: Pending (status == 0 && id == 0)
|
||||
* Combined check: if (status | id) != 0 then skip to id_dispatch. Falls through to the Pending case only when both are zero. */
|
||||
or_u_self(R_RawStatus, R_RawId), branch_ne(R_RawStatus, R_0, atom_offset(case_2, id_dispatch)),
|
||||
/* BD-slot: pre-compute PadStatus_Pending. Branch reads R_RawStatus in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to id_dispatch), R_T4 is overwritten by the digital/analog body add_ui - harmless. */
|
||||
|
||||
atom_label(pending) /* === Pending body (status=0, id=0 — pre-IRQ-empty buffer). */
|
||||
mac_pad_set_status(R_T4, R_PadState, PadStatus_Pending),
|
||||
store_half( R_0, R_PadState, O_(PadState,buttons)),
|
||||
mac_pad_set_centered_axes(R_PadState, R_T4),
|
||||
store_byte(R_RawId, R_PadState, O_(PadState,id)),
|
||||
jump_rel(atom_offset(pending, snap_end)),
|
||||
mac_yield_load(),
|
||||
|
||||
atom_label(id_dispatch) /* === Case 3-6: ID dispatch */
|
||||
add_ui(R_T4, R_0, PadRawId_Digital), branch_ne(R_RawId, R_T4, atom_offset(id_dispatch, try_analog_stick)),
|
||||
/* BD-slot: pre-compute PadStatus_Digital. Branch reads R_RawId in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to try_analog_stick), R_T4 is overwritten by the analog body add_ui. */
|
||||
|
||||
/* === Digital body (status, buttons normalize, axes=0x80, id, branch.
|
||||
* R_T5 holds the 0x80808080 axes constant (loaded into the load-delay slot of the buttons-load).
|
||||
* R_T5 is then "dead" — only consumed at the analog_pad range check downstream. */
|
||||
mac_pad_set_status(R_T4, R_PadState, PadStatus_Digital),
|
||||
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw, buttons)), /* R_T4 = raw_buttons; */
|
||||
mac_load_word_imm( R_T5, PadAxis_Centered), /* fills the buttons-load's delay slot (doesn't read R_T4) */
|
||||
// load_upper_i(R_T5, PadAxis_Centered_Hi), or_i_self(R_T5, PadAxis_Centered_Lo),
|
||||
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
|
||||
store_word(R_T5, R_PadState, O_(PadState, axes)), /* single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y) */
|
||||
mac_pad_set_id_byte(R_PadState, R_T4, PadRawId_Digital),
|
||||
|
||||
jump_rel(atom_offset(id_dispatch, snap_end)),
|
||||
mac_yield_load(),
|
||||
|
||||
atom_label(try_analog_stick) /* === Case 4: AnalogStick (id == 0x53)*/
|
||||
add_ui(R_T4, R_0, PadRawId_AnalogStick), branch_ne(R_RawId, R_T4, atom_offset(try_analog_stick, try_analog_pad)),
|
||||
/* BD-slot: pre-compute PadStatus_AnalogStick. Branch reads R_RawId in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to try_analog_pad), R_T4 is overwritten by the analog_pad body add_ui. */
|
||||
|
||||
atom_label(analog_stick) /* === AnalogStick body
|
||||
* R_T5 holds left_xy (loaded into the load-delay slot of the buttons-load via the left-axis load_half_u).
|
||||
* R_T4 holds right_xy (loaded into the load-delay slot of the left-load).
|
||||
* R_T5 is then "dead" — reused for the id-byte value load in mac_pad_write_id_byte.
|
||||
* The buttons invert+store happens BEFORE R_T4 is overwritten by the right_xy load. */
|
||||
mac_pad_set_status(R_T4, R_PadState, PadStatus_AnalogStick),
|
||||
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,buttons)), /* R_T4 = raw_buttons; delay slot at the next instruction */
|
||||
load_half_u( R_T5, R_PadRaw, O_(PadBiosRaw,left)), /* fills the buttons-load's delay slot (doesn't read R_T4) */
|
||||
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
|
||||
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,right)), /* fills R_T5's load-delay slot (doesn't read R_T5); overwrites R_T4 (was buttons) with right_xy */
|
||||
store_half( R_T5, R_PadState, O_(PadState, left)),
|
||||
store_half( R_T4, R_PadState, O_(PadState, right)),
|
||||
mac_pad_set_id_byte(R_PadState, R_T5, PadRawId_AnalogStick),
|
||||
jump_rel(atom_offset(analog_stick, snap_end)),
|
||||
mac_yield_load(),
|
||||
|
||||
atom_label(try_analog_pad) /* === Case 5-6: AnalogPad (id & 0xF0 == 0x70) */
|
||||
and_i( R_T4, R_RawId, PadRawId_AnalogPadMask),
|
||||
add_ui( R_T5, R_0, PadRawId_AnalogPadValue),
|
||||
branch_ne(R_T4, R_T5, atom_offset(try_analog_pad, try_unsupported)),
|
||||
/* BD-slot: pre-compute PadStatus_AnalogPad. Branch reads R_T4 in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to try_unsupported), R_T4 is overwritten by the unsupported body add_ui. */
|
||||
|
||||
atom_label(analog_pad) /* === AnalogPad body
|
||||
* Same shape as AnalogStick with AnalogPad status. R_T5 holds left_xy (it's dead on this path).
|
||||
* The id byte is raw id from the BIOS buffer (R_RawId already holds raw[1]).
|
||||
* Buttons invert + store happens before R_T4 is overwritten by the right_xy load. */
|
||||
mac_pad_set_status(R_T4, R_PadState, PadStatus_AnalogPad),
|
||||
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,buttons)), /* R_T4 = raw_buttons; delay slot at the next instruction */
|
||||
load_half_u( R_T5, R_PadRaw, O_(PadBiosRaw,left)), /* fills the buttons-load's delay slot (doesn't read R_T4) */
|
||||
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
|
||||
load_half_u(R_T4, R_PadRaw, O_(PadBiosRaw,right)), /* fills R_T5's load-delay slot (doesn't read R_T5); overwrites R_T4 with right_xy */
|
||||
store_half( R_T5, R_PadState, O_(PadState, left)),
|
||||
store_half( R_T4, R_PadState, O_(PadState, right)),
|
||||
store_byte( R_RawId, R_PadState, O_(PadState, id)),
|
||||
|
||||
jump_rel(atom_offset(analog_pad, snap_end)),
|
||||
mac_yield_load(),
|
||||
|
||||
atom_label(try_unsupported) /* === Case 7: Unsupported — fall through from the AnalogPad range-check miss. */
|
||||
add_ui( R_T4, R_0, PadStatus_Unsupported),
|
||||
store_word(R_T4, R_PadState, O_(PadState,status)),
|
||||
store_half(R_0, R_PadState, O_(PadState,buttons)),
|
||||
mac_pad_set_centered_axes(R_PadState, R_T4),
|
||||
mac_pad_set_id_byte(R_PadState, R_RawId, PadUnknownId_Sentinel),
|
||||
/* Fall through to snap_end. */
|
||||
|
||||
atom_label(no_jump_fallthrough)
|
||||
mac_yield_load(),
|
||||
|
||||
atom_label(snap_end)
|
||||
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the case-exit branch. */
|
||||
mac_yield_tail(),
|
||||
};
|
||||
|
||||
#pragma endregion Baked Atoms
|
||||
@@ -0,0 +1,78 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# include "dsl.h"
|
||||
# include "gcc_asm.h"
|
||||
# include "mips.h"
|
||||
# include "bios.h"
|
||||
# include "pad.h"
|
||||
#endif
|
||||
|
||||
/* Uses ONE 8-byte frame allocated via the compiler's standard prologue.
|
||||
* 4 wasted-arg words for B(12h) InitPAD2 are at [SP+0..15] but are not explicitly allocated.
|
||||
* Compiler handles the MIPS O32 "wasted stack" convention for us by treating the B-call as a 4-arg call.
|
||||
*
|
||||
* The buffer pointers are passed as arguments so the compiler keeps them in callee-saved registers;
|
||||
* The B(12h) asm volatile block does NOT clobber those registers (it clobbers only the volatile GPRs + B-table arg registers explicitly).
|
||||
* The C-level writes after the call re-load the pointers from their callee-saved homes.
|
||||
*
|
||||
* The clobber list for both B-calls names the full BIOS destroy set documented in kernelbios.md:167-174 (R1..R15, R24..R25, R31, HI/LO).
|
||||
* The kernel-ABI "volatile GPRs" subset is clb_mem_drain; the rest of the destroy set is enumerated explicitly here. */
|
||||
NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
|
||||
{
|
||||
/* Pin raw0 + raw1 to $a0 + $a1 via rgcc; the B(12h) call uses these directly.
|
||||
* The `(void)` casts mark them as unread after the call so the compiler doesn't need to move them back. */
|
||||
register PadBiosRaw* p0 rgcc(R_A0) = raw0;
|
||||
register PadBiosRaw* p1 rgcc(R_A1) = raw1;
|
||||
(void)p0; (void)p1;
|
||||
|
||||
// TODO(Ed): Properly annotate the raw values in the inline asm instructions.
|
||||
// Use enums.
|
||||
|
||||
/* B(12h) InitPAD2(raw0, 0x22, raw1, 0x22)
|
||||
* $a0 = raw0 (rgcc-bound; survives the sequence below)
|
||||
* $a1 = raw1 (preserved into $a2 before $a1 is overwritten)
|
||||
* $a2 = raw1 (moved from $a1; survives $a1's overwrite)
|
||||
* $a3 = 0x22 (immediate)
|
||||
* $t1 = 0x12 (function number)
|
||||
* $t2 = 0xB0 (BIOS B-table address) */
|
||||
asm volatile(
|
||||
asm_words(
|
||||
or_u( R_A2, R_A1, R_0), /* $a2 = $a1 = raw1 */
|
||||
add_ui( R_A1, R_0, bios_pad_buffer_size), /* $a1 = 0x22 */
|
||||
add_ui( R_A3, R_0, bios_pad_buffer_size), /* $a3 = 0x22 */
|
||||
add_ui( R_T1, R_0, bios_init_pad_2), /* $t1 = 0x12 */
|
||||
add_ui( R_T2, R_0, bios_btable_addr), /* $t2 = 0xB0 */
|
||||
call_reg(R_T2), /* jalr $t2, $ra */
|
||||
nop /* BD slot */
|
||||
)
|
||||
asm_rpins, r_use(p0), r_use(p1)
|
||||
asm_clobber:
|
||||
rlit(R_AT),
|
||||
rlit(R_V0), rlit(R_V1),
|
||||
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
||||
rlit(R_RA),
|
||||
clb_mem_drain
|
||||
);
|
||||
|
||||
/* The C-level writes re-load the pointers via the parameter names and write 0xFF to each
|
||||
* buffer's status byte to mark the initial-state hazard documented in kernelbios.md:1621-1624. */
|
||||
u1_v(raw0)[0] = 0xFF;
|
||||
u1_v(raw1)[0] = 0xFF;
|
||||
|
||||
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */
|
||||
asm volatile(
|
||||
asm_words(
|
||||
add_ui( R_T1, R_0, bios_start_pad_2), /* $t1 = 0x13 */
|
||||
add_ui( R_T2, R_0, bios_btable_addr), /* $t2 = 0xB0 (re-load) */
|
||||
call_reg(R_T2), /* jalr $t2, $ra */
|
||||
nop /* BD slot */
|
||||
)
|
||||
asm_clobber:
|
||||
rlit(R_AT),
|
||||
rlit(R_V0), rlit(R_V1),
|
||||
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
||||
rlit(R_RA),
|
||||
clb_mem_drain
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,108 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "dsl.h"
|
||||
# include "math.h"
|
||||
#endif
|
||||
|
||||
// PSX button bit positions: PSX-SPX docs/psx-spx/docs/controllersandmemorycards.md:405-421.
|
||||
typedef Enum_(U2, PadBtns) {
|
||||
Bit_(Pad_Select, 0),
|
||||
Bit_(Pad_L3, 1),
|
||||
Bit_(Pad_R3, 2),
|
||||
Bit_(Pad_Start, 3),
|
||||
Bit_(Pad_Up, 4),
|
||||
Bit_(Pad_Right, 5),
|
||||
Bit_(Pad_Down, 6),
|
||||
Bit_(Pad_Left, 7),
|
||||
Bit_(Pad_L2, 8),
|
||||
Bit_(Pad_R2, 9),
|
||||
Bit_(Pad_L1, 10),
|
||||
Bit_(Pad_R1, 11),
|
||||
Bit_(Pad_Triangle, 12),
|
||||
Bit_(Pad_Circle, 13),
|
||||
Bit_(Pad_Cross, 14),
|
||||
Bit_(Pad_Square, 15),
|
||||
};
|
||||
|
||||
enum {
|
||||
PadId_Offset = 4,
|
||||
|
||||
Pad0 = 0 << PadId_Offset,
|
||||
Pad1 = 1 << PadId_Offset,
|
||||
};
|
||||
|
||||
/* =============================================================================
|
||||
* BIOS pad-buffer subsystem: docs/psx-spx/docs/kernelbios.md (B(12h) + B(13h))
|
||||
* ============================================================================= */
|
||||
|
||||
enum {
|
||||
PAD_BIOS_RAW_SIZE = 0x22,
|
||||
};
|
||||
// BIOS pad buffer layout (docs/psx-spx/docs/kernelbios.md (InitPAD2 returns 0x22 = 34 bytes per port)).
|
||||
// Bytes 0..7 are the named snapshot region; bytes 8..33 are reserved (the BIOS writes the buffer raw; we only read bytes 0..7 via O_(PadBiosRaw, ...)).
|
||||
typedef Struct_(PadBiosRaw) {
|
||||
U1 status; /* offset 0 (PadRawStatus_Ok / PadRawStatus_Timeout) */
|
||||
U1 id; /* offset 1 (PadRawId_Digital / PadRawId_AnalogStick / 0x7x AnalogPad) */
|
||||
U2 buttons; /* offset 2-3 (active-low 16-bit button map) */
|
||||
V2_U1 right; /* offset 4-5 (right stick x, y) */
|
||||
V2_U1 left; /* offset 6-7 (left stick x, y) */
|
||||
U1 reserved[PAD_BIOS_RAW_SIZE - 8]; /* offset 8..33 */
|
||||
};
|
||||
|
||||
typedef Enum_(U4, PadStatus) {
|
||||
PadStatus_Disconnected,
|
||||
PadStatus_Digital,
|
||||
PadStatus_AnalogStick,
|
||||
PadStatus_AnalogPad,
|
||||
PadStatus_Unsupported,
|
||||
PadStatus_Pending,
|
||||
PadStatus_Invalid,
|
||||
};
|
||||
|
||||
// Distinct from the game-facing PadStatus enum: PadRawStatus_Ok and PadRawStatus_Timeout are raw BIOS values
|
||||
typedef Enum_(U1, PadRawStatus) {
|
||||
PadRawStatus_Ok = 0x00,
|
||||
PadRawStatus_Timeout = 0xFF,
|
||||
};
|
||||
typedef Enum_(U1, PadRawId) {
|
||||
PadRawId_Digital = 0x41,
|
||||
PadRawId_AnalogStick = 0x53,
|
||||
PadRawId_AnalogPadMask = 0xF0,
|
||||
PadRawId_AnalogPadValue = 0x70,
|
||||
};
|
||||
typedef Enum_(U1, PadUnknownId) {
|
||||
PadUnknownId_Sentinel = 0xFF,
|
||||
};
|
||||
typedef Enum_(U4, PadAxisCentered) {
|
||||
PadAxis_Centered_Hi = 0x8080,
|
||||
PadAxis_Centered_Lo = 0x8080,
|
||||
PadAxis_Centered = 0x80808080U,
|
||||
};
|
||||
typedef Enum_(U1, PadDeadZone) {
|
||||
PadDeadZone_LowBound = 0x70, /* left_x < LowBound → active; delta = 0x80 - left_x > 0 (rightward pull) */
|
||||
PadDeadZone_Center = 0x80, /* analog rest position; left_x == Center → delta = 0 (no rotation) */
|
||||
PadDeadZone_HighBound = 0x90, /* left_x > HighBound → active; delta = 0x80 - left_x < 0 (leftward pull) */
|
||||
};
|
||||
|
||||
|
||||
typedef Struct_(PadAxes) {
|
||||
V2_U1 left; /* offset 8-9 */
|
||||
V2_U1 right; /* offset 10-11 */
|
||||
};
|
||||
// Field order is chosen so that the 4 axes (left_x, left_y, right_x, right_y)
|
||||
// form a contiguous 4-byte block at offset 8, allowing a single `store_word` to clear-or-write all 4 axes in one MIPS instruction.
|
||||
typedef Struct_(PadState) {
|
||||
PadStatus status; /* offset 0, (U4) */
|
||||
PadBtns buttons; /* offset 4, */
|
||||
U1 id; /* offset 6, */
|
||||
byte_pad(1); /* offset 7, explicit pad to align the axes block */
|
||||
union {
|
||||
A2_V2_U1 axes; /* offset 8-11 store_target (4-byte aligned)*/
|
||||
struct {
|
||||
V2_U1 left; /* offset 8-9 */
|
||||
V2_U1 right; /* offset 10-11 */
|
||||
};
|
||||
};
|
||||
};
|
||||
|
||||
internal void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1);
|
||||
@@ -0,0 +1,7 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# include "gen/macs.h"
|
||||
# include "gen/offsets.h"
|
||||
# include "psyq.h"
|
||||
#endif
|
||||
|
||||
ATOM_FILE_DEBUGGER_LINE_MARKER(pysq_atom_c);
|
||||
@@ -0,0 +1,121 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "dsl.h"
|
||||
# include "math.h"
|
||||
# include "gp.h"
|
||||
#endif
|
||||
|
||||
typedef Struct_(DrawEnv_Packed) { U4 tag; U4 code[15]; };
|
||||
typedef Struct_(DrawEnv) {
|
||||
Rect_S2 clip_area;
|
||||
V2_S2 drawing_offset[2];
|
||||
Rect_S2 texture_window;
|
||||
S2 texture_page;
|
||||
B1 flag_dither;
|
||||
B1 flag_draw_on_display;
|
||||
B1 enable_auto_clear;
|
||||
RGB8 initial_bg_color;
|
||||
DrawEnv_Packed dr_env; // reserved
|
||||
};
|
||||
typedef Struct_(DisplayEnv) {
|
||||
Rect_S2 display_area;
|
||||
Rect_S2 screen;
|
||||
B1 vinterlace;
|
||||
B1 color24;
|
||||
B1 pad0;
|
||||
B1 pad1;
|
||||
};
|
||||
typedef Array_(DrawEnv, 2);
|
||||
typedef Array_(DisplayEnv, 2);
|
||||
|
||||
typedef Struct_(DoubleBuffer) {
|
||||
A2_DrawEnv draw;
|
||||
A2_DisplayEnv display;
|
||||
};
|
||||
|
||||
DisplayEnv* displayenv_init(DisplayEnv* env, S4 x, S4 y, S4 w, S4 h) asm("SetDefDispEnv");
|
||||
DrawEnv* drawenv_init (DrawEnv* env, S4 x, S4 y, S4 w, S4 h) asm("SetDefDrawEnv");
|
||||
|
||||
DisplayEnv* displayenv_put(DisplayEnv* env) asm("PutDispEnv");
|
||||
DrawEnv* drawenv_put (DrawEnv* env) asm("PutDrawEnv");
|
||||
|
||||
U4 geom_init(void) asm("InitGeom");
|
||||
void geom_set_offset(U4 x, U4 y) asm("SetGeomOffset");
|
||||
void geom_set_screen(U4 h) asm("SetGeomScreen");
|
||||
|
||||
U4* orderingtbl_clear_reverse(U4* ot, U4 len) asm("ClearOTagR");
|
||||
|
||||
U4 reset_graph(U4 mode) asm("ResetGraph");
|
||||
void set_display_enabled(U4 mask) asm("SetDispMask");
|
||||
|
||||
U4 draw_sync(U4 mode) asm("DrawSync");
|
||||
U4 vsync(U4 mode) asm("VSync");
|
||||
|
||||
void draw_orderingtbl(U4* buf) asm("DrawOTag");
|
||||
|
||||
typedef Struct_(Tile) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
Rect_S2 rect;
|
||||
};
|
||||
|
||||
/*
|
||||
Linear Algebra
|
||||
*/
|
||||
|
||||
MT3_S2S4* mt3s2s4_rotation (V3_S2* vec, MT3_S2S4* mat) asm("RotMatrix");
|
||||
MT3_S2S4* mt3s2s4_translation(MT3_S2S4* mat, V3_S4* vec) asm("TransMatrix");
|
||||
MT3_S2S4* mt3s2s4_scale (MT3_S2S4* mat, V3_S4* vec) asm("ScaleMatrix");
|
||||
|
||||
// Rotation, Translation, Perspective
|
||||
|
||||
S4 rtp_v3s2_raw(V3_S2* vec, S4* xy, S4* pp, S4* flag) asm("RotTransPers");
|
||||
FI_ S4 rtp_v3s2(V3_S2* vec, V2_S2* xy, A2_S2* pp, S4* flag) { return rtp_v3s2_raw(vec, C_(S4*R_, & xy->x), C_(S4*R_, pp), r_(flag)); }
|
||||
|
||||
S4 rtp_avg_nclip_a3_v3s2_raw(V3_S2* v0, V3_S2* v1, V3_S2* v2, S4* xy1, S4* xy2, S4* xy3, S4* pp, S4* otz, S4* flag) asm("RotAverageNclip3");
|
||||
FI_ S4 rtp_avg_nclip_a3_v3s2(
|
||||
V3_S2* v0, V3_S2* v1, V3_S2* v2,
|
||||
V2_S2* xy0, V2_S2* xy1, V2_S2* xy2,
|
||||
A2_S2* pp, S4* otz, S4* flag
|
||||
){
|
||||
return rtp_avg_nclip_a3_v3s2_raw(
|
||||
v0, v1, v2,
|
||||
C_(S4*R_, xy0), C_(S4*R_, xy1), C_(S4*R_, xy2),
|
||||
C_(S4*R_, pp), C_(S4*R_, otz), C_(S4*R_, flag)
|
||||
);
|
||||
}
|
||||
|
||||
S4 rtp_avg_nclip_a4_v3s2_raw(V3_S2* v0, V3_S2* v1, V3_S2* v2, V3_S2* v3, S4* xy1, S4* xy2, S4* xy3, S4* xy4, S4* pp, S4* otz, S4* flag) asm("RotAverageNclip4");
|
||||
FI_ S4 rtp_avg_nclip_a4_v3s2(
|
||||
V3_S2* v0, V3_S2* v1, V3_S2* v2, V3_S2* v3,
|
||||
V2_S2* xy0, V2_S2* xy1, V2_S2* xy2, V2_S2* xy3,
|
||||
A2_S2* pp, S4* otz, S4* flag
|
||||
){
|
||||
return rtp_avg_nclip_a4_v3s2_raw(
|
||||
v0, v1, v2, v3,
|
||||
C_(S4*R_, xy0), C_(S4*R_, xy1), C_(S4*R_, xy2), C_(S4*R_, xy3),
|
||||
C_(S4*R_, pp), C_(S4*R_, otz), C_(S4*R_, flag)
|
||||
);
|
||||
}
|
||||
|
||||
void gte_matrix_set_rotation (MT3_S2S4* mat) asm("SetRotMatrix");
|
||||
void gte_matrix_set_translation(MT3_S2S4* mat) asm("SetTransMatrix");
|
||||
|
||||
// Einheit, Metrication to unit vector. "Normalization", not Orthogonal "Normal, Normalis". Directionalization.
|
||||
// RGA(Lengyel): Normalize the bulk of a zero-weight direction. This is not finite-point unitization (which forces w=1).
|
||||
S4 psy_normalize_v3s4(V3_S4* v0, V3_S4* v1) asm("VectorNormal");
|
||||
|
||||
// RGA(Lengyel): Apply the matrix expansion of a rigid transformation.
|
||||
// Motor antiproduct is equivalent for unitized points; LA form is what GTE consumes.
|
||||
V3_S4* mul_m3s2_v3s4(MT3_S2S4* m, V3_S4* v, V3_S4* result) asm("ApplyMatrixLV");
|
||||
|
||||
// RGA(Lengyel): Store the full translation column. The motor translator would store half this displacement in m.xyz.
|
||||
MT3_S2S4* trans_m3s2(MT3_S2S4* m, V3_S4* off) asm("TransMatrix");
|
||||
|
||||
MT3_S2S4* gte_comp_coord_m3s2(MT3_S2S4* m0, MT3_S2S4* m1, MT3_S2S4* result) asm("CompMatrixLV");
|
||||
|
||||
// RGA(Lengyel): Complement(Wedge(a,b)), i.e. the Euclidean 3D complement of the exterior product, stored as a V3_S4.
|
||||
// The underlying GTE OP is a specialized signed-16-bit D x IR command; the wedge interpretation is a 3D dual of the same 3 scalars.
|
||||
void cross_v3s4(V3_S4* v0, V3_S4* v1, V3_S4* result) asm("OuterProduct12");
|
||||
|
||||
@@ -1,21 +1,22 @@
|
||||
// tape_atom.metadata.h
|
||||
// word_count.metadata.h
|
||||
// Single source of truth for instruction-word counts.
|
||||
// Used by C (to define compile-time constants) AND Python (to count positions).
|
||||
//
|
||||
// Format: WORD_COUNT(MACRO_NAME, COUNT)
|
||||
// One line per macro that appears in your atom sources.
|
||||
//
|
||||
// This file is encoding-macros-only. The auto-generated component
|
||||
// macros (mac_X) live in duffle/gen/<dir>.macs.h (included separately
|
||||
// by the unity build). The unity build should include THIS file and
|
||||
// the .macs.h file in the same TU, with both wrapped (or the
|
||||
// include guard order handled) to avoid WORD_COUNT redeclaration.
|
||||
// This file is encoding-macros-only.
|
||||
// The auto-generated component macros (mac_X) live in the source directory's own gen/macs.h (per-directory aggregation; included separately by the unity build).
|
||||
// The unity build should include THIS file and the .macs.h file in the same TU, with both wrapped
|
||||
// (or the include guard order handled) to avoid WORD_COUNT redeclaration.
|
||||
//
|
||||
// To regenerate: hand-count the instructions in each macro definition.
|
||||
// (You'll only need to do this once per macro — they don't change often.)
|
||||
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
||||
|
||||
WORD_COUNT(nop, 1)
|
||||
WORD_COUNT(atom_label, 0)
|
||||
WORD_COUNT(atom_offset, 0)
|
||||
WORD_COUNT(load_upper_i, 1)
|
||||
WORD_COUNT(jump_reg, 1)
|
||||
WORD_COUNT(jump_link, 1)
|
||||
@@ -23,6 +24,7 @@ WORD_COUNT(call_reg, 1)
|
||||
WORD_COUNT(call_addr, 1)
|
||||
WORD_COUNT(branch_le_zero, 1)
|
||||
WORD_COUNT(branch_equal, 1)
|
||||
WORD_COUNT(branch_ne, 1)
|
||||
WORD_COUNT(add_ui, 1)
|
||||
WORD_COUNT(set_lt_u, 1)
|
||||
WORD_COUNT(set_lt_s, 1)
|
||||
@@ -30,7 +32,9 @@ WORD_COUNT(set_lt_si, 1)
|
||||
WORD_COUNT(set_lt_ui, 1)
|
||||
WORD_COUNT(load_word, 1)
|
||||
WORD_COUNT(load_half_u, 1)
|
||||
WORD_COUNT(load_byte_u, 1)
|
||||
WORD_COUNT(store_word, 1)
|
||||
WORD_COUNT(store_byte, 1)
|
||||
WORD_COUNT(add_ui_self, 1)
|
||||
WORD_COUNT(add_u_self, 1)
|
||||
WORD_COUNT(add_u, 1)
|
||||
@@ -38,6 +42,7 @@ WORD_COUNT(or_i, 1)
|
||||
WORD_COUNT(or_i_self, 1)
|
||||
WORD_COUNT(or_u, 1)
|
||||
WORD_COUNT(or_u_self, 1)
|
||||
WORD_COUNT(nor_u, 1)
|
||||
WORD_COUNT(shift_lleft, 1)
|
||||
WORD_COUNT(shift_lleft_self, 1)
|
||||
WORD_COUNT(shift_lright, 1)
|
||||
@@ -50,7 +55,17 @@ WORD_COUNT(gte_mv_to_ctrl_r, 1)
|
||||
WORD_COUNT(gte_sw, 1)
|
||||
WORD_COUNT(gte_cmdw_rtpt, 1)
|
||||
WORD_COUNT(gte_cmdw_nclip, 1)
|
||||
WORD_COUNT(gte_cmdw_op, 1)
|
||||
WORD_COUNT(gte_avg_sort_z3, 1)
|
||||
WORD_COUNT(gte_cmdw_sqr, 1)
|
||||
WORD_COUNT(gte_cmdw_gpf, 1)
|
||||
WORD_COUNT(shift_lleft_var, 1)
|
||||
WORD_COUNT(shift_aright_var, 1)
|
||||
WORD_COUNT(li_s, 1)
|
||||
WORD_COUNT(and_i, 1)
|
||||
WORD_COUNT(add_si, 1)
|
||||
WORD_COUNT(branch_lt_zero, 1)
|
||||
WORD_COUNT(sub_s, 1)
|
||||
WORD_COUNT(sub_u, 1)
|
||||
WORD_COUNT(nop2, 2)
|
||||
|
||||
@@ -17,19 +17,19 @@ enum {
|
||||
};
|
||||
|
||||
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
||||
typedef def_farray(OrderingTable_Buffer, 2);
|
||||
typedef Array_(OrderingTable_Buffer, 2);
|
||||
|
||||
typedef B1 PrimitiveBuffer[PrimitiveBuff_Len];
|
||||
typedef def_farray(PrimitiveBuffer, 2);
|
||||
typedef def_struct(PrimitiveArena) {
|
||||
typedef Array_(PrimitiveBuffer, 2);
|
||||
typedef Struct_(PrimitiveArena) {
|
||||
A2_PrimitiveBuffer buf;
|
||||
U4 used;
|
||||
};
|
||||
|
||||
#define Cube_num_verts 8
|
||||
typedef def_farray(V3_S2, Cube_num_verts);
|
||||
typedef Array_(V3_S2, Cube_num_verts);
|
||||
#define Cube_num_faces 6
|
||||
typedef def_farray(V4_S2, Cube_num_faces);
|
||||
typedef Array_(V4_S2, Cube_num_faces);
|
||||
void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
||||
memory_copy(verts, & (A8_V3_S2) {
|
||||
{ -128, -128, -128 },
|
||||
@@ -40,7 +40,7 @@ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
||||
{ 128, 128, -128 },
|
||||
{ 128, 128, 128 },
|
||||
{ -128, 128, 128 }
|
||||
}, size_of(A8_V3_S2) );
|
||||
}, S_(A8_V3_S2) );
|
||||
memory_copy(faces, & (A6_V4_S2) {
|
||||
{ 3, 2, 0, 1 },
|
||||
{ 0, 1, 4, 5 },
|
||||
@@ -48,10 +48,10 @@ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
||||
{ 1, 2, 5, 6 },
|
||||
{ 2, 3, 6, 7 },
|
||||
{ 3, 0, 7, 4 },
|
||||
}, size_of(A6_V4_S2) );
|
||||
}, S_(A6_V4_S2) );
|
||||
return;
|
||||
}
|
||||
typedef def_struct(Ent_Cube) {
|
||||
typedef Struct_(Ent_Cube) {
|
||||
V3_S4 accel;
|
||||
V3_S4 vel;
|
||||
V3_S4 pos;
|
||||
@@ -62,22 +62,22 @@ typedef def_struct(Ent_Cube) {
|
||||
};
|
||||
|
||||
#define Floor_num_verts 4
|
||||
typedef def_farray(V3_S2, Floor_num_verts);
|
||||
typedef Array_(V3_S2, Floor_num_verts);
|
||||
#define Floor_num_faces 2
|
||||
typedef def_farray(V3_S2, Floor_num_faces);
|
||||
typedef Array_(V3_S2, Floor_num_faces);
|
||||
void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
|
||||
memory_copy(verts, &(A4_V3_S2) {
|
||||
{ -900, 0, -900 },
|
||||
{ -900, 0, 900 },
|
||||
{ 900, 0, -900 },
|
||||
{ 900, 0, 900 },
|
||||
}, size_of(A8_V3_S2));
|
||||
}, S_(A8_V3_S2));
|
||||
memory_copy(faces, & (A2_V3_S2) {
|
||||
{ 0, 1, 2 },
|
||||
{ 1, 3, 2 },
|
||||
}, size_of(A2_V3_S2));
|
||||
}, S_(A2_V3_S2));
|
||||
};
|
||||
typedef def_struct(Ent_Floor) {
|
||||
typedef Struct_(Ent_Floor) {
|
||||
V3_S4 accel;
|
||||
V3_S4 pos;
|
||||
V3_S4 scale;
|
||||
@@ -86,7 +86,7 @@ typedef def_struct(Ent_Floor) {
|
||||
A2_V3_S2 faces;
|
||||
};
|
||||
|
||||
typedef def_struct(SMemory) {
|
||||
typedef Struct_(SMemory) {
|
||||
DoubleBuffer screen_buf;
|
||||
A2_OrderingTable_Buffer ordering_tbl;
|
||||
PrimitiveArena primitives;
|
||||
@@ -108,7 +108,7 @@ B1* prim__alloc(U4 type_width, Str8 type_name) {
|
||||
pa->used += type_width;
|
||||
return next;
|
||||
}
|
||||
#define prim_alloc(type) (type*)prim__alloc(size_of(type), txt( stringify(type)))
|
||||
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
|
||||
|
||||
void gp_screen_init_c11(DoubleBuffer* screen_buf, S2* active_buf_id)
|
||||
{
|
||||
|
||||
@@ -5,8 +5,8 @@
|
||||
# include "duffle/gp.h"
|
||||
#endif
|
||||
|
||||
typedef def_struct(DrawEnv_Packed) { U4 tag; U4 code[15]; };
|
||||
typedef def_struct(DrawEnv) {
|
||||
typedef Struct_(DrawEnv_Packed) { U4 tag; U4 code[15]; };
|
||||
typedef Struct_(DrawEnv) {
|
||||
Rect_S2 clip_area;
|
||||
A2_S2 drawing_offset;
|
||||
Rect_S2 texture_window;
|
||||
@@ -17,7 +17,7 @@ typedef def_struct(DrawEnv) {
|
||||
RGB8 initial_bg_color;
|
||||
DrawEnv_Packed dr_env; // reserved
|
||||
};
|
||||
typedef def_struct(DisplayEnv) {
|
||||
typedef Struct_(DisplayEnv) {
|
||||
Rect_S2 display_area;
|
||||
Rect_S2 screen;
|
||||
B1 vinterlace;
|
||||
@@ -25,9 +25,9 @@ typedef def_struct(DisplayEnv) {
|
||||
B1 pad0;
|
||||
B1 pad1;
|
||||
};
|
||||
typedef def_farray(DrawEnv, 2);
|
||||
typedef def_farray(DisplayEnv, 2);
|
||||
typedef def_struct(DoubleBuffer) {
|
||||
typedef Array_(DrawEnv, 2);
|
||||
typedef Array_(DisplayEnv, 2);
|
||||
typedef Struct_(DoubleBuffer) {
|
||||
A2_DrawEnv draw;
|
||||
A2_DisplayEnv display;
|
||||
};
|
||||
@@ -58,7 +58,7 @@ U4 vsync(U4 mode) __asm__("VSync");
|
||||
|
||||
void draw_orderingtbl(U4* buf) __asm__("DrawOTag");
|
||||
|
||||
typedef def_struct(PolyTag) {
|
||||
typedef Struct_(PolyTag) {
|
||||
U4 addr: 24;
|
||||
U4 len: 8;
|
||||
RGB8 color;
|
||||
@@ -106,7 +106,7 @@ typedef def_struct(PolyTag) {
|
||||
// #define setLineF4(p) set_len(p, 6), set_code(p, 0x4c),(p)->pad = 0x55555555
|
||||
// #define setLineG4(p) set_len(p, 9), set_code(p, 0x5c),(p)->pad = 0x55555555, (p)->p2 = 0, (p)->p3 = 0
|
||||
|
||||
typedef def_struct(Poly_F3) {
|
||||
typedef Struct_(Poly_F3) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
@@ -120,14 +120,14 @@ typedef def_struct(Poly_F3) {
|
||||
};
|
||||
};
|
||||
|
||||
typedef def_struct(Poly_G3) {
|
||||
typedef Struct_(Poly_G3) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2;
|
||||
};
|
||||
|
||||
typedef def_struct(Poly_F4) {
|
||||
typedef Struct_(Poly_F4) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
@@ -142,7 +142,7 @@ typedef def_struct(Poly_F4) {
|
||||
};
|
||||
};
|
||||
|
||||
typedef def_struct(Poly_G4) {
|
||||
typedef Struct_(Poly_G4) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
@@ -150,7 +150,7 @@ typedef def_struct(Poly_G4) {
|
||||
V2_S2 p3;
|
||||
};
|
||||
|
||||
typedef def_struct(Tile) {
|
||||
typedef Struct_(Tile) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
@@ -169,7 +169,7 @@ M3_S2* m3s2_scale (M3_S2* mat, V3_S4* vec) __asm__("ScaleMatrix");
|
||||
// Rotation, Translation, Perspective
|
||||
|
||||
S4 rtp_v3s2_raw(V3_S2* vec, S4* xy, S4* pp, S4* flag) __asm__("RotTransPers");
|
||||
FI_ S4 rtp_v3s2(V3_S2* vec, V2_S2* xy, A2_S2* pp, S4* flag) { return rtp_v3s2_raw(vec, cast(S4*R_, & xy->x), cast(S4*R_, pp), r_(flag)); }
|
||||
FI_ S4 rtp_v3s2(V3_S2* vec, V2_S2* xy, A2_S2* pp, S4* flag) { return rtp_v3s2_raw(vec, C_(S4*R_, & xy->x), C_(S4*R_, pp), r_(flag)); }
|
||||
|
||||
S4 rtp_avg_nclip_a3_v3s2_raw(V3_S2* v0, V3_S2* v1, V3_S2* v2, S4* xy1, S4* xy2, S4* xy3, S4* pp, S4* otz, S4* flag) __asm__("RotAverageNclip3");
|
||||
FI_ S4 rtp_avg_nclip_a3_v3s2(
|
||||
@@ -179,8 +179,8 @@ FI_ S4 rtp_avg_nclip_a3_v3s2(
|
||||
){
|
||||
return rtp_avg_nclip_a3_v3s2_raw(
|
||||
v0, v1, v2,
|
||||
cast(S4*R_, xy0), cast(S4*R_, xy1), cast(S4*R_, xy2),
|
||||
cast(S4*R_, pp), cast(S4*R_, otz), cast(S4*R_, flag)
|
||||
C_(S4*R_, xy0), C_(S4*R_, xy1), C_(S4*R_, xy2),
|
||||
C_(S4*R_, pp), C_(S4*R_, otz), C_(S4*R_, flag)
|
||||
);
|
||||
}
|
||||
|
||||
@@ -192,8 +192,8 @@ FI_ S4 rtp_avg_nclip_a4_v3s2(
|
||||
){
|
||||
return rtp_avg_nclip_a4_v3s2_raw(
|
||||
v0, v1, v2, v3,
|
||||
cast(S4*R_, xy0), cast(S4*R_, xy1), cast(S4*R_, xy2), cast(S4*R_, xy3),
|
||||
cast(S4*R_, pp), cast(S4*R_, otz), cast(S4*R_, flag)
|
||||
C_(S4*R_, xy0), C_(S4*R_, xy1), C_(S4*R_, xy2), C_(S4*R_, xy3),
|
||||
C_(S4*R_, pp), C_(S4*R_, otz), C_(S4*R_, flag)
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
@@ -1,193 +0,0 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# include "duffle/gen/duffle.macs.h"
|
||||
# include "duffle/gen/duffle.offsets.h"
|
||||
# include "duffle/atom_dsl.h"
|
||||
# include "duffle/lottes_tape.h"
|
||||
# include "tape_atom.metadata.h"
|
||||
# include "gen/gte_hello.offsets.h"
|
||||
# include "hello_gte.h"
|
||||
#endif
|
||||
|
||||
#pragma region MACs (Mips Atom components)
|
||||
|
||||
|
||||
|
||||
#pragma endregion MACs
|
||||
|
||||
#pragma region Baked Atoms
|
||||
|
||||
typedef Struct_(Binds_CubeTri) {
|
||||
U4 PrimCursor;
|
||||
V4_S2* FaceCursor;
|
||||
V3_S2* VertBase;
|
||||
U4* OtBase;
|
||||
};
|
||||
internal MipsAtom_(rbind_cube_g4_face) {
|
||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
||||
// Note(Ed): This entire thing is argument shuffle?
|
||||
// TODO(Ed): Eliminate
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
/* ============================================================================
|
||||
* cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
|
||||
* ============================================================================
|
||||
*
|
||||
* Reads 4 indices from R_FaceCur (V4_S2 = 8 bytes), loads 4 vertices into
|
||||
* the GTE, runs the PsyQ RotAverageNclip4 sequence, and renders a Poly_G4.
|
||||
*/
|
||||
atom_region (cube_g4_face, REGION_PRIM_ARENA)
|
||||
atom_group (cube_g4_face, GROUP_RENDER_PRIMS)
|
||||
atom_cadence (cube_g4_face, CADENCE_FRAME)
|
||||
atom_annot(cube_g4_face, phase_work,
|
||||
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
||||
atom_writes(R_PrimCursor, R_FaceCursor))
|
||||
internal
|
||||
MipsAtom_(cube_g4_face) {
|
||||
/* ── 1. Load 4 face indices from R_FaceCur (V4_S2 = 8 bytes) ───────── */
|
||||
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
||||
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
||||
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
||||
|
||||
/* ── 2. Load V0, V1, V2 into GTE (parallel to mac_load_tri_verts) ── */
|
||||
mac_load_tri_verts(R_T0, R_T1, R_T2),
|
||||
|
||||
/* ── 3. RTPT — transforms V0/V1/V2 → SXY0/SXY1/SXY2 + SZ1/SZ2/SZ3 ─── */
|
||||
nop2, gte_cmdw_rotate_translate_perspective_triple,
|
||||
|
||||
/* ── 4. NCLIP — backface culling on SXY0/SXY1/SXY2 (p0,p1,p2) ──────── */
|
||||
/* MUST be done BEFORE V3-RTPS overwrites SXY0 with p3. */
|
||||
nop2, gte_cmdw_nclip,
|
||||
|
||||
/* ── 5. Cull check: skip format/insert if MAC0 ≤ 0 (backface) ───────── */
|
||||
nop2, gte_mv_from_data_r(R_T0, C2_MAC0),
|
||||
nop, /* COP2 stall */
|
||||
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
|
||||
nop, /* BD slot */
|
||||
|
||||
/* ── 6. Format c0..c3 (color+code words) BEFORE V3-RTPS ─────────────── */
|
||||
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
||||
mac_format_g4_color(
|
||||
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
||||
/* c1 yellow */ 0xFF, 0xFF, 0x00,
|
||||
/* c2 cyan */ 0x00, 0xFF, 0xFF,
|
||||
/* c3 green */ 0x00, 0xFF, 0x00),
|
||||
|
||||
/* ── 7. Store p0..p2 BEFORE V3-RTPS overwrites SXY0 ─────────────────── */
|
||||
mac_gte_store_g4_p012_post_rtpt_pre_rtps(),
|
||||
|
||||
/* ── 8. Load V3 = verts[face->w] into V0 ─────────────────────────────── */
|
||||
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
||||
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
|
||||
/* ── 9. RTPS — transforms V0 (now V3) → SXY0 (p3) + SZ3 ─────────────── */
|
||||
nop2, gte_cmdw_rotate_translate_perspective_single,
|
||||
mac_gte_store_g4_p3_post_rtps(),
|
||||
|
||||
/* ── 10. AVSZ4 — average Z from SZ0/SZ1/SZ2/SZ3 ─────────────────────── */
|
||||
nop2, gte_cmdw_avg_sort_z4,
|
||||
nop2, gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||
|
||||
/* ── 11. Bounds check OTZ < OrderingTbl_Len ─────────────────────────── */
|
||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
|
||||
|
||||
/* ── 12. Insert into Ordering Table (length = 8 words for Poly_G4) ──── */
|
||||
mac_insert_ot_tag_g4(),
|
||||
|
||||
/* ── 13. Advance cursors & yield (both branch targets land here) ────── */
|
||||
atom_label(cube_g4_face_exit)
|
||||
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
|
||||
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_FloorTri) {
|
||||
U4 PrimCursor;
|
||||
V3_S2* FaceCursor;
|
||||
V3_S2* VertBase;
|
||||
U4* OtBase;
|
||||
};
|
||||
atom_region(rbind_floor_f3_face, REGION_PRIM_ARENA)
|
||||
atom_group(rbind_floor_f3_face, GROUP_RENDER_FLOOR)
|
||||
atom_cadence(rbind_floor_f3_face, CADENCE_FRAME)
|
||||
atom_annot(rbind_floor_f3_face, phase_bind
|
||||
, atom_reads()
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase))
|
||||
internal
|
||||
MipsAtom_(rbind_floor_f3_face) {
|
||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
atom_region( floor_f3_face, REGION_PRIM_ARENA)
|
||||
atom_group( floor_f3_face, GROUP_RENDER_FLOOR)
|
||||
atom_cadence(floor_f3_face, CADENCE_FRAME)
|
||||
atom_annot( floor_f3_face, phase_work,
|
||||
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
||||
atom_writes(R_PrimCursor, R_FaceCursor))
|
||||
internal
|
||||
MipsAtom_(floor_f3_face) {
|
||||
mac_load_tri_indices(R_T0, R_T1, R_T2),
|
||||
mac_load_tri_verts( R_T0, R_T1, R_T2),
|
||||
nop2, gte_cmdw_rotate_translate_perspective_triple,
|
||||
nop2, gte_cmdw_nclip,
|
||||
|
||||
/* Culling (Branch forward if Backface) */
|
||||
nop2, gte_mv_from_data_r(R_T0, C2_MAC0),
|
||||
nop,
|
||||
branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop,
|
||||
/* Format Primitive */
|
||||
// mac_format_f3_color(0x20FF, 0xFFFF), // works
|
||||
mac_format_f3_color(0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
|
||||
mac_gte_store_f3_post_rtpt(),
|
||||
|
||||
/* Calculate Depth */
|
||||
nop2, gte_avg_sort_z3,
|
||||
nop2, gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||
/* Bounds Check OTZ < 2048 (Branch forward to skip insertion) */
|
||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
|
||||
/* Insert into Ordering Table Linked List */
|
||||
mac_insert_ot_tag_f3(),
|
||||
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
|
||||
// Note(Ed): No bounds checking, should be checked before atom runs.
|
||||
|
||||
/* Advance Input Cursor & Yield (Both branch targets land here) */
|
||||
atom_label(floor_f3_face_exit)
|
||||
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
|
||||
atom_region( sync_primitive_arena, REGION_PRIM_ARENA)
|
||||
atom_group( sync_primitive_arena, GROUP_RENDER_FLOOR)
|
||||
atom_cadence(sync_primitive_arena, CADENCE_FRAME)
|
||||
atom_annot( sync_primitive_arena, phase_work,
|
||||
atom_reads( R_TapePtr, R_PrimCursor),
|
||||
atom_writes(R_TapePtr))
|
||||
internal MipsAtom_(sync_primitive_arena) {
|
||||
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
|
||||
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
|
||||
/* Calculate byte offset and store directly back to RAM */
|
||||
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
|
||||
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
#pragma endregion Baked Atoms
|
||||
@@ -0,0 +1,13 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
#pragma once
|
||||
#endif
|
||||
// Auto-generated by ps1_meta.lua (passes/auto_reg.lua) — DO NOT EDIT
|
||||
// Directory: C:\projects\Pikuma\ps1\code\hello_camera
|
||||
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.c
|
||||
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.h
|
||||
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.atom.c
|
||||
// Per-phase register allocations resolved by the lua pass.
|
||||
// R_<Sym>_Code = <chosen GPR's _Code constant> for every marker in this directory.
|
||||
|
||||
#define R_GpTmp_Code R_V0_Code
|
||||
|
||||
@@ -0,0 +1,41 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
#pragma once
|
||||
#endif
|
||||
// Auto-generated by ps1_meta.lua — DO NOT EDIT
|
||||
// Directory: C:\projects\Pikuma\ps1\code\hello_camera/
|
||||
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.c
|
||||
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.h
|
||||
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.atom.c
|
||||
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
|
||||
|
||||
#ifndef WORD_COUNT
|
||||
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
||||
#endif
|
||||
|
||||
#define mac_put_disp_env(reg_transfer, reg_base, port) \
|
||||
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port)
|
||||
WORD_COUNT(mac_put_disp_env, 5)
|
||||
|
||||
#define mac_put_draw_env(reg_transfer, reg_base, port) \
|
||||
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port) /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */ \
|
||||
, mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port) /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */ \
|
||||
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[1] TextureWindow (tw=(0,0)) */ \
|
||||
, mac_gcmd_push(enc_gp0_draw_area_tl_word(0, ScreenRes_Y), reg_transfer, reg_base, port) /* code[2] DrawArea top-left (clip.x=0, clip.y=ScreenRes_Y=240) */ \
|
||||
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) /* code[3] DrawArea bottom-right (clip.x+w=320, clip.y+h=480) */ \
|
||||
, mac_gcmd_push(gp0_word_set_draw_offset(), reg_transfer, reg_base, port) /* code[4] DrawOffset (ofs=(0,0)) — bare-cmd word; the GPU uses the current state machine. */ \
|
||||
, mac_gcmd_push(gp0_word_dr_env_mask(), reg_transfer, reg_base, port) /* code[5] Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit. */ \
|
||||
, mac_gcmd_push(gp0_word_dr_env_bg_color_cmd(1, 7, 7, 7), reg_transfer, reg_base, port) /* code[6] Initial-bg-color + auto-clear (isbg=1, r=7, g=7, b=7). */ \
|
||||
, mac_gcmd_push(gp0_word_dr_env_draw_mode(1), reg_transfer, reg_base, port) /* code[7] Re-assert DrawMode with isbg=1 (isbg-flag set; the 0xE1 cmd byte plus isbg only). */ /* code[8..10] Padding (NOP — GPU discards; the DR_ENV requires 16 words total). */ \
|
||||
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) /* code[11..12] TextureWindow bottom-right (tw.x+tw.w=0, tw.y+tw.h=0) — libpsyx emits twice. */ \
|
||||
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[13..14] Padding (NOP) — completes the 16-word packet. */ \
|
||||
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port)
|
||||
WORD_COUNT(mac_put_draw_env, 16)
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
|
||||
// Directory: C:\projects\Pikuma\ps1\code\hello_camera\
|
||||
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.c
|
||||
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.h
|
||||
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.atom.c
|
||||
#pragma once
|
||||
|
||||
#pragma region hello_camera
|
||||
|
||||
|
||||
// --- atom: pad_input_cube_rotation (60 words) ---
|
||||
|
||||
#define _atom_offset_dpad_left_exit_dpad_left 6
|
||||
#define _atom_offset_dpad_right_exit_dpad_right 6
|
||||
#define _atom_offset_dead_zone_low_check_dead_low_active 8
|
||||
#define _atom_offset_dead_zone_high_check_dead_high_active 15
|
||||
#define _atom_offset_dead_zone_skip_exit_stick 24
|
||||
#define _atom_offset_end_low_exit_stick 12
|
||||
|
||||
enum {
|
||||
atom_offset_dpad_left_exit_dpad_left = _atom_offset_dpad_left_exit_dpad_left,
|
||||
atom_offset_dpad_right_exit_dpad_right = _atom_offset_dpad_right_exit_dpad_right,
|
||||
atom_offset_dead_zone_low_check_dead_low_active = _atom_offset_dead_zone_low_check_dead_low_active,
|
||||
atom_offset_dead_zone_high_check_dead_high_active = _atom_offset_dead_zone_high_check_dead_high_active,
|
||||
atom_offset_dead_zone_skip_exit_stick = _atom_offset_dead_zone_skip_exit_stick,
|
||||
atom_offset_end_low_exit_stick = _atom_offset_end_low_exit_stick,
|
||||
};
|
||||
|
||||
// --- atom: pad_input_cam (39 words) ---
|
||||
|
||||
#define _atom_offset_left_x_exit_left_x 3
|
||||
#define _atom_offset_right_x_exit_right_x 3
|
||||
#define _atom_offset_up_y_exit_up_y 3
|
||||
#define _atom_offset_down_y_exit_down_y 3
|
||||
#define _atom_offset_cross_z_exit_cross_z 3
|
||||
#define _atom_offset_circle_z_exit_circle_z 3
|
||||
|
||||
enum {
|
||||
atom_offset_left_x_exit_left_x = _atom_offset_left_x_exit_left_x,
|
||||
atom_offset_right_x_exit_right_x = _atom_offset_right_x_exit_right_x,
|
||||
atom_offset_up_y_exit_up_y = _atom_offset_up_y_exit_up_y,
|
||||
atom_offset_down_y_exit_down_y = _atom_offset_down_y_exit_down_y,
|
||||
atom_offset_cross_z_exit_cross_z = _atom_offset_cross_z_exit_cross_z,
|
||||
atom_offset_circle_z_exit_circle_z = _atom_offset_circle_z_exit_circle_z,
|
||||
};
|
||||
|
||||
// --- atom: cube_g4_face (73 words) ---
|
||||
|
||||
#define _atom_offset_cull_cube_g4_face_exit 41
|
||||
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
|
||||
|
||||
enum {
|
||||
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
|
||||
atom_offset_bounds_chk_cube_g4_face_exit = _atom_offset_bounds_chk_cube_g4_face_exit,
|
||||
};
|
||||
|
||||
// --- atom: floor_f3_face (56 words) ---
|
||||
|
||||
#define _atom_offset_culling_floor_f3_face_exit 25
|
||||
#define _atom_offset_bounds_chk_floor_f3_face_exit 16
|
||||
|
||||
enum {
|
||||
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
|
||||
atom_offset_bounds_chk_floor_f3_face_exit = _atom_offset_bounds_chk_floor_f3_face_exit,
|
||||
};
|
||||
|
||||
#pragma endregion hello_camera
|
||||
|
||||
@@ -0,0 +1,643 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "duffle/gen/macs.h"
|
||||
# include "duffle/gen/offsets.h"
|
||||
# include "duffle/dsl.atom.h"
|
||||
# include "duffle/lottes_tape.h"
|
||||
# include "duffle/mips.h"
|
||||
# include "duffle/gte.h"
|
||||
# include "duffle/gp.h"
|
||||
# include "duffle/pad.h"
|
||||
# include "duffle/word_count.metadata.h"
|
||||
# include "duffle/psyq.h"
|
||||
# include "duffle/math.atom.h"
|
||||
# include "duffle/mips.atom.c"
|
||||
# include "duffle/gte.atom.c"
|
||||
# include "duffle/gp.atom.c"
|
||||
# include "duffle/psyq.atom.c"
|
||||
# include "gen/offsets.h"
|
||||
# include "gen/macs.h"
|
||||
# include "gen/auto_reg.h"
|
||||
# include "hello_camera.h"
|
||||
#endif
|
||||
|
||||
ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c);
|
||||
|
||||
#pragma region MACs (Mips Atom components)
|
||||
|
||||
FI_ Slice_MipsCode ac_put_disp_env(AtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
||||
MipsAtomComp_Proc_(ab, {
|
||||
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
|
||||
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
|
||||
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
||||
})
|
||||
|
||||
I_ Slice_MipsCode ac_put_draw_env(AtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
||||
MipsAtomComp_Proc_(ab, {
|
||||
/*
|
||||
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
|
||||
* References:
|
||||
* - libpsyx source: `toolchain/psyq-4_7/lib/libgpu.a` (binary, function `PutDrawEnv`)
|
||||
* - PSX-SPX doc: https://problemkaputt.de/psx-spx.htm#gputdrawingcommands
|
||||
* - PSYQ SDK: `setdrawenv` / `makelongdr_env` source
|
||||
* - NOCASH PSX spec: §"GP0(E1h) Draw Mode setting" through §"DR_ENV"
|
||||
*
|
||||
* The 16-word format is documented in the PSYQ SDK manual and on NOCASH's PSX-spec.txt. The libpsyx reference is at:
|
||||
* ./toolchain/psyq-4_7/lib/libgpu.a
|
||||
* (binary; the PutDrawEnv implementation builds the 16-word DR_ENV from the user's DRAWENV struct and emits it via GP0 GPU commands.)
|
||||
*
|
||||
* Word indices (libpsyx PutDrawEnv / SetDrawEnv order):
|
||||
* tag = (length << 24) | addr — 16-word packet (1 tag + 15 code)
|
||||
* code[0] = DrawMode (dfe=1, dtd=0, tpage=0) — must come first per libpsyx
|
||||
* code[1] = TextureWindow (tw=(0,0)) — bare-cmd word; GPU uses current state
|
||||
* code[2] = DrawArea top-left (clip.x=0, clip.y=240)
|
||||
* code[3] = DrawArea bottom-right (clip.x+w=320, clip.y+h=480)
|
||||
* code[4] = DrawOffset (ofs=(0,0)) — bare-cmd word
|
||||
* code[5] = Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit
|
||||
* code[6] = Initial-bg-color (isbg=1, r=7, g=7, b=7)
|
||||
* code[7] = DrawMode (isbg=1, tpage=0) — re-asserts DrawMode with isbg
|
||||
* code[8..10] = padding (NOP) — 3 words to fill the packet
|
||||
* code[11..12] = TextureWindow bottom-right — defaults to (0,0,0,0)
|
||||
* code[13..14] = padding (NOP) — completes the 16-word packet
|
||||
*/
|
||||
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port), /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */
|
||||
mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port), /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */
|
||||
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port), /* code[1] TextureWindow (tw=(0,0)) */
|
||||
mac_gcmd_push(enc_gp0_draw_area_tl_word(0, ScreenRes_Y), reg_transfer, reg_base, port), /* code[2] DrawArea top-left (clip.x=0, clip.y=ScreenRes_Y=240) */
|
||||
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port), /* code[3] DrawArea bottom-right (clip.x+w=320, clip.y+h=480) */
|
||||
|
||||
mac_gcmd_push(gp0_word_set_draw_offset(), reg_transfer, reg_base, port), /* code[4] DrawOffset (ofs=(0,0)) — bare-cmd word; the GPU uses the current state machine. */
|
||||
mac_gcmd_push(gp0_word_dr_env_mask(), reg_transfer, reg_base, port), /* code[5] Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit. */
|
||||
mac_gcmd_push(gp0_word_dr_env_bg_color_cmd(1, 7, 7, 7), reg_transfer, reg_base, port), /* code[6] Initial-bg-color + auto-clear (isbg=1, r=7, g=7, b=7). */
|
||||
mac_gcmd_push(gp0_word_dr_env_draw_mode(1), reg_transfer, reg_base, port), /* code[7] Re-assert DrawMode with isbg=1 (isbg-flag set; the 0xE1 cmd byte plus isbg only). */
|
||||
|
||||
/* code[8..10] Padding (NOP — GPU discards; the DR_ENV requires 16 words total). */
|
||||
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||
|
||||
/* code[11..12] TextureWindow bottom-right (tw.x+tw.w=0, tw.y+tw.h=0) — libpsyx emits twice. */
|
||||
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port),
|
||||
|
||||
/* code[13..14] Padding (NOP) — completes the 16-word packet. */
|
||||
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||
})
|
||||
|
||||
#pragma endregion MACs
|
||||
|
||||
#pragma region Atom Procs
|
||||
|
||||
#pragma region resolve_look_at
|
||||
/* ─── resolve_look_at bundle chain atoms ──────────────────────────── */
|
||||
|
||||
typedef AtomBundle_(resolve_look_at) { MipsAtom
|
||||
*input_and_sub,
|
||||
*normalize_fwd_uz,
|
||||
*cross_to_right,
|
||||
*normalize_right_ux,
|
||||
*cross_to_up,
|
||||
*normalize_up_uy,
|
||||
*populate_mt3s4s2;
|
||||
};
|
||||
|
||||
typedef Struct_(ResolveLookAtScratch) {
|
||||
V3_S4 fwd;
|
||||
V3_S4 uz;
|
||||
V3_S4 right;
|
||||
V3_S4 ux;
|
||||
V3_S4 up;
|
||||
V3_S4 uy;
|
||||
P3_S4 eye;
|
||||
P3_S4 target;
|
||||
V3_S4 up_in;
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_ResolveLookAtSub) {
|
||||
P3_S4* target;
|
||||
P3_S4* eye;
|
||||
V3_S4* up_in;
|
||||
};
|
||||
typedef Struct_(RegUse_resolve_look_at_input_and_sub) {
|
||||
Reg target_ptr;
|
||||
Reg eye_ptr;
|
||||
Reg up_in_ptr;
|
||||
union { Reg_(V3_S4) r012, up_in, eye; };
|
||||
union { Reg_(V3_S4) r345, target, fwd; };
|
||||
};
|
||||
/* Atom 0 in the bundle: input_and_sub. Stages C-side inputs into the scratchpad and computes fwd = target - eye. */
|
||||
internal MipsAtom* AtomBundleEntry_(resolve_look_at,input_and_sub)(AtomArena_R aa, RegUse_resolve_look_at_input_and_sub r)
|
||||
atom_info(atom_bind(Binds_ResolveLookAtSub)) MipsAtom_Proc_(aa, {
|
||||
load_word(r.target_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,target)),
|
||||
load_word(r.eye_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)),
|
||||
load_word(r.up_in_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)),
|
||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtSub)),
|
||||
|
||||
/* Stage up_in.x/y/z into the scratchpad. R_ScratchBase = R_SP = 0x1F800000. */
|
||||
mac_load_v3s4( r.up_in, r.up_in_ptr, 0), LdSlot_
|
||||
mac_store_v3s4(r.up_in, R_ScratchBase, O_(ResolveLookAtScratch,up_in)),
|
||||
|
||||
// Stage eye.x/y/z into the scratchpad (atom 6 reads these for the translation column).
|
||||
mac_load_v3s4( r.eye, r.eye_ptr, 0), LdSlot_
|
||||
mac_store_v3s4(r.eye, R_ScratchBase, O_(ResolveLookAtScratch,eye)),
|
||||
|
||||
/* Compute fwd = target - eye. */
|
||||
mac_load_v3s4( r.target, r.target_ptr, 0), LdSlot_
|
||||
mac_sub_v3s4_self(r.fwd, r.eye),
|
||||
mac_store_v3s4( r.fwd, R_ScratchBase, O_(ResolveLookAtScratch,fwd)),
|
||||
|
||||
mac_yield()
|
||||
})
|
||||
|
||||
typedef Struct_(Binds_ResolveLookAt_PopulateMT3S4S2) {
|
||||
MT3_S2S4* look_at; /* MT3_S2S4* — destination matrix address */
|
||||
};
|
||||
typedef Struct_(RegUse_resolve_look_at_populate_mt3s4s2) {
|
||||
Reg look_at;
|
||||
Reg eye; /* matrix_vector phase: load -eye */
|
||||
Reg_(V3_S4) row; /* populate phase: load ux/uy/uz */
|
||||
union { Reg r0, ux, vx; }; /* populate addr → matrix_vector v_x */
|
||||
union { Reg r1, uy, vy; }; /* populate uy → matrix_vector v_y */
|
||||
union { Reg r2, uz, vz; }; /* populate uz → matrix_vector v_z */
|
||||
};
|
||||
/* write look_at->m[][] from ux/uy/uz as packed S2 (populate),
|
||||
* ctc2 RT chain into C2[0..4] (matrix_vector), MVMVA RT*(-eye)>>12, store off
|
||||
* directly to look_at->t[] (trans_matrix).
|
||||
*
|
||||
* C11 ApplyMatrixLV semantics (gte.atom.c ac_apply_matrix_lv; libgte reference):
|
||||
* 1. ctc2 RT matrix (5 ctc2s to C2[0..4])
|
||||
* 2. lw -eye from memory
|
||||
* 3. S15 decomposition (eliminated here — the fused body takes the >>12 path
|
||||
* directly via mtc2 IR + MVMVA pass2, matching the libgte canonical output)
|
||||
* 4. mtc2 to IR1/2/3, nop2, MVMVA pass2 (sf=1, mx=0, v=3, cv=3)
|
||||
* 5. mfc2 MACs → off
|
||||
* 6. store off to look_at->t[] (skip scratch.eye intermediate)
|
||||
*/
|
||||
internal MipsAtom* AtomBundleEntry_(resolve_look_at,populate_mt3s4s2)(AtomArena_R aa, RegUse_resolve_look_at_populate_mt3s4s2 r)
|
||||
atom_info(atom_bind(Binds_ResolveLookAt_PopulateMT3S4S2)) MipsAtom_Proc_(aa, {
|
||||
/* --- Tape pop: look_at pointer --- */
|
||||
load_word(r.look_at, R_TapePtr, O_(Binds_ResolveLookAt_PopulateMT3S4S2,look_at)),
|
||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_ResolveLookAt_PopulateMT3S4S2)),
|
||||
|
||||
add_si(r.ux, R_ScratchBase, O_(ResolveLookAtScratch, ux)), LdSlot_
|
||||
add_si(r.uy, R_ScratchBase, O_(ResolveLookAtScratch, uy)),
|
||||
add_si(r.uz, R_ScratchBase, O_(ResolveLookAtScratch, uz)),
|
||||
add_si(r.eye, R_ScratchBase, O_(ResolveLookAtScratch, eye)),
|
||||
|
||||
/* write look_at->m[][] from ux/uy/uz as packed S2 */
|
||||
mac_load_v3s4(r.row, r.ux, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4, m[0])),
|
||||
mac_load_v3s4(r.row, r.uy, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4, m[1])),
|
||||
mac_load_v3s4(r.row, r.uz, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4, m[2])),
|
||||
|
||||
/* ctc2 RT chain + MVMVA RT * (-eye) >> 12 */
|
||||
/* C2[0] = (RT12<<16)|RT11 ← ctc2 RT11 from m[0][0..1]
|
||||
* C2[1] = (RT21<<16)|RT13 ← ctc2 RT12 from m[0][2..3]
|
||||
* C2[2] = (RT23<<16)|RT22 ← ctc2 RT13 from m[1][1..2]
|
||||
* C2[3] = (RT32<<16)|RT31 ← ctc2 RT21 from m[2][0..1]
|
||||
* C2[4] = (RT33<<16)|junk ← ctc2 RT22 from m[2][2] (half) */
|
||||
load_word( r.vx, r.look_at, O_(MT3_S2S4, m[0][0])), /* RT11|RT12 */ LdSlot_
|
||||
load_word( r.vy, r.look_at, O_(MT3_S2S4, m[0][2])), /* RT13|RT21 */ LdSlot_ gte_mv_to_ctrl_r(r.vx, gte_cr_RT11),
|
||||
load_word( r.vz, r.look_at, O_(MT3_S2S4, m[1][1])), /* RT22|RT23 */ LdSlot_ gte_mv_to_ctrl_r(r.vy, gte_cr_RT12),
|
||||
load_word( r.vx, r.look_at, O_(MT3_S2S4, m[2][0])), /* RT31|RT32 */ LdSlot_ gte_mv_to_ctrl_r(r.vz, gte_cr_RT13),
|
||||
load_half_u(r.vy, r.look_at, O_(MT3_S2S4, m[2][2])), /* RT33 */ LdSlot_ gte_mv_to_ctrl_r(r.vx, gte_cr_RT21),
|
||||
|
||||
GteDelay_ mac_load_word_v3(r.vx, r.vy, r.vz, r.eye, 0), LdSlot_
|
||||
mac_sub_s_v3(r.vx, r.vy, r.vz, R_0, R_0, R_0, r.vx, r.vy, r.vz),
|
||||
|
||||
gte_mv_to_data_r(r.vx, C2_IR1),
|
||||
gte_mv_to_data_r(r.vy, C2_IR2),
|
||||
gte_mv_to_data_r(r.vz, C2_IR3),
|
||||
GteDelay_ nop2,
|
||||
|
||||
/* MVMVA pass 2 — C11 ApplyMatrixLV command. sf=1, mx=0 (RT), v=3 (IR), cv=3. Reads RT × IR >> 12. */
|
||||
gte_cmdw_mvmva_c11_pass2, GteDelay_ load_word(R_AtomJmp, R_TapePtr, 0), // ac_yield: word 1
|
||||
mac_gte_mv_from_data_r_mac123(r.vx, r.vy, r.vz), GteDelay_ add_ui_self( R_TapePtr, S_(MipsCode)), // ac_yield: word 2
|
||||
|
||||
/* store off directly to look_at->t[] (skip scratch.eye intermediate) */
|
||||
mac_store_word_v3(r.vx, r.vy, r.vz, r.look_at, O_(MT3_S2S4, t)),
|
||||
|
||||
jump_reg(R_AtomJmp), BdSlot_ nop, // ac_yield: word 3-4
|
||||
})
|
||||
#pragma endregion resolve_look_at
|
||||
|
||||
#pragma endregion Atom Procs
|
||||
|
||||
#pragma region Baked Atoms
|
||||
|
||||
enum {
|
||||
R_ScreenX = R_T5 atom_reg atom_type(U2),
|
||||
R_ScreenY = R_T6 atom_reg atom_type(U2),
|
||||
R_ScreenBuf = R_T7 atom_reg, /* Caller-pinned: & smem.screen_buf */
|
||||
#define R_ScreenBuf_Code R_T7_Code
|
||||
};
|
||||
//screen_env_init. Mirrors the libpsyx's SetDefDispEnv + SetDefDrawEnv + the manual enable_auto_clear / initial_bg_color writes.
|
||||
internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
|
||||
, atom_reads(R_T0, R_ScreenX, R_ScreenY, R_ScreenBuf)
|
||||
, atom_writes(R_T0, R_ScreenX, R_ScreenY)
|
||||
) {
|
||||
/* display[0] = (0, 0, 320, 240); rest of struct zeroed. */
|
||||
add_ui(R_ScreenX, R_0, ScreenRes_X), add_ui(R_ScreenY, R_0, ScreenRes_Y),
|
||||
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area.width) + O_(DoubleBuffer,display[0])),
|
||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,display_area) + O_(DoubleBuffer,display[0])),
|
||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + O_(DoubleBuffer,display[0])),
|
||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + O_(DoubleBuffer,display[0])),
|
||||
|
||||
/* display[1] = (0, 240, 320, 240); rest of struct zeroed. */
|
||||
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area) + O_(DoubleBuffer,display[1])),
|
||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + O_(DoubleBuffer,display[1])),
|
||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + O_(DoubleBuffer,display[1])),
|
||||
|
||||
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + O_(DoubleBuffer,draw[0])), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
|
||||
mac_store_v2s2( R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + O_(DoubleBuffer,draw[0])), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
|
||||
|
||||
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + O_(DoubleBuffer,draw[1])),
|
||||
|
||||
/* draw[0].texture_window = (0, 0, 0, 0); two word-zeroes cover the full 8-byte tw field. */
|
||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + O_(DoubleBuffer,draw[0])),
|
||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + O_(DoubleBuffer,draw[0])),
|
||||
|
||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,drawing_offset[0].x) + O_(DoubleBuffer,draw[1])),
|
||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + O_(DoubleBuffer,draw[1])),
|
||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + O_(DoubleBuffer,draw[1])),
|
||||
|
||||
/* draw[0].texture_page = 10 (gp0_tpage_default). C11 SetDefDrawEnv at C11_only.elf:0x8001273C writes the same 0x0A. . */
|
||||
add_ui(R_T0, R_0, gp0_tpage_default),
|
||||
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + O_(DoubleBuffer,draw[0])),
|
||||
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + O_(DoubleBuffer,draw[1])),
|
||||
|
||||
/* draw[0] control bytes: flag_dither=1, flag_draw_on_display=1 (the dfe bit per psx-spx; libpsyx sets it via `SetDefDrawEnv`'s conditional at C11_only.elf:0x80012728), enable_auto_clear=1. Each byte is named;
|
||||
* the previous `store_word(R_0, ..., +20)` overwrote all four with zero. */
|
||||
add_ui(R_T0, R_0, 1),
|
||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + O_(DoubleBuffer,draw[0])),
|
||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + O_(DoubleBuffer,draw[0])),
|
||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + O_(DoubleBuffer,draw[0])),
|
||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + O_(DoubleBuffer,draw[1])),
|
||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + O_(DoubleBuffer,draw[1])),
|
||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + O_(DoubleBuffer,draw[1])),
|
||||
|
||||
/* draw[0].initial_bg_color = (r=7, g=7, b=7). */
|
||||
add_ui(R_T0, R_0, 7),
|
||||
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + O_(DoubleBuffer,draw[0])),
|
||||
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + O_(DoubleBuffer,draw[1])),
|
||||
|
||||
mac_yield(),
|
||||
};
|
||||
|
||||
enum {
|
||||
R_IO_BaseAddr = R_T4 atom_reg, /* Caller-pinned: IO_BASE_ADDR = 0x1F800000 */
|
||||
R_GP1_Offset = R_T2 atom_reg, /* Caller-pinned: GPIO_PORT1_OFFSET = 0x10 */
|
||||
atom_auto_reg(gp_screen_init, R_GpTmp), /* Auto-allocated scratch; resolved to a free pool GPR by the lua pass. C-preprocessor expands to R_GpTmp = R_GpTmp_Code with an atom_auto_reg trailing comment. */
|
||||
#define R_IO_BaseAddr_Code R_T4_Code
|
||||
#define R_GP1_Offset_Code R_T2_Code
|
||||
};
|
||||
internal MipsAtom_(gp_screen_init) atom_info(atom_phase(screen_init), atom_reads(R_IO_BaseAddr)) {
|
||||
store_word(R_0, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(00h) Reset */
|
||||
mac_gcmd_push(gp1_word_ResetCmdBuffer(), R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(01h) ClearFIFO; uses pinned R_ScreenX as the transfer reg. */
|
||||
mac_gcmd_push(gp1_word_AcknowledgeIRQ(), R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(02h) AckIRQ; uses pinned R_ScreenX as the transfer reg. */
|
||||
mac_gcmd_push(gp1_word_DisplayOn(), R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(03h) Display ON; uses pinned R_ScreenX as the transfer reg. */
|
||||
mac_gcmd_push(gp1_word_dma_to_gpu(), R_GpTmp, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(04h) DMADirection=2 (CPU->GPU). libpsyx's per-frame PutDrawEnv/DrawOTag use DMA2; without this the DMA queue never drains. Uses auto-allocated R_GpTmp. */
|
||||
mac_gcmd_push(gp1_word_StartDisplayArea(), R_GpTmp, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(05h) StartDisplayArea (X=0, Y=0); uses auto-allocated R_GpTmp. */
|
||||
|
||||
/* GP1: DisplayMode + Display Ranges. */
|
||||
mac_gcmd_push(gp1_word_display_mode_320x240_15bit_ntsc, R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||
mac_gcmd_push(gp1_word_horizontal_range_ntsc, R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||
mac_gcmd_push(gp1_word_vertical_range_ntsc, R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||
|
||||
/* GTE: SetGeomOffset (OFX, OFY) — ScreenRes_CenterX, ScreenRes_CenterY. */
|
||||
load_upper_i(R_ScreenX, ScreenRes_CenterX), gte_mv_to_ctrl_r(R_ScreenX, gte_cr_OFX_Code),
|
||||
load_upper_i(R_ScreenX, ScreenRes_CenterY), gte_mv_to_ctrl_r(R_ScreenX, gte_cr_OFY_Code),
|
||||
|
||||
/* GTE: SetGeomScreen (H) — CR26 (per PSX-SPX / libpsyx), value is the raw projection-plane distance, NOT shifted. */
|
||||
add_ui(R_ScreenX, R_0, ScreenZ), gte_mv_to_ctrl_r(R_ScreenX, gte_cr_H_Code),
|
||||
|
||||
/* GP1: DisplayEnable — bit 0 = 0 (Display ON). */
|
||||
mac_gcmd_push(gp1_word_DisplayOn(), R_GpTmp, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* Uses auto-allocated R_GpTmp. */
|
||||
mac_yield(),
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_PadApplyInput) {
|
||||
PadState* state;
|
||||
V3_S2* cube_rot;
|
||||
V3_S2* floor_rot;
|
||||
};
|
||||
enum {
|
||||
R_PadStateT5 = R_T5 atom_reg,
|
||||
R_CubeRot = R_T1 atom_reg,
|
||||
R_FloorRot = R_T2 atom_reg,
|
||||
};
|
||||
internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyInput)
|
||||
, atom_reads(R_T0, R_CubeRot, R_FloorRot, R_T3, R_T4, R_PadStateT5, R_TapePtr)
|
||||
, atom_writes( R_CubeRot, R_FloorRot)
|
||||
) {
|
||||
/* Pop Binds from tape (state, cube_rot, floor_rot) */
|
||||
load_word(R_PadStateT5, R_TapePtr, O_(Binds_PadApplyInput,state)),
|
||||
load_word(R_CubeRot, R_TapePtr, O_(Binds_PadApplyInput,cube_rot)),
|
||||
load_word(R_FloorRot, R_TapePtr, O_(Binds_PadApplyInput,floor_rot)),
|
||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_PadApplyInput)),
|
||||
|
||||
/* Load pad[0].buttons into R_T0. */
|
||||
load_word(R_T0, R_PadStateT5, O_(PadState,buttons)), LdSlot_ nop,
|
||||
// Note(Ed): Potential op with delay slot?
|
||||
|
||||
/* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */
|
||||
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)), BdSlot_
|
||||
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), LdSlot_
|
||||
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||
add_si( R_T4, R_T4, 30),
|
||||
add_si( R_T3, R_T3, 5),
|
||||
store_half(R_T4, R_CubeRot, O_(V3_S2,y)),
|
||||
store_half(R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||
atom_label(exit_dpad_left)
|
||||
|
||||
/* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */
|
||||
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)), BdSlot_
|
||||
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), LdSlot_
|
||||
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||
add_si( R_T4, R_T4, -30),
|
||||
add_si( R_T3, R_T3, -5),
|
||||
store_half(R_T4, R_CubeRot, O_(V3_S2,y)),
|
||||
store_half(R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||
atom_label(exit_dpad_right)
|
||||
|
||||
/* Analog left-stick X: dead zone 0x70..0x90.
|
||||
* Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */
|
||||
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), LdSlot_ //?
|
||||
|
||||
/* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly).
|
||||
* set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */
|
||||
add_ui(R_T4, R_0, PadDeadZone_HighBound), set_lt_u(R_T4, R_T3, R_T4), branch_ne(R_T4, R_0, atom_offset(dead_zone_low_check, dead_low_active)),
|
||||
add_ui(R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_low_active */
|
||||
|
||||
atom_label(dead_check_upper)
|
||||
/* left_x >= 0x70 → check upper bound. */
|
||||
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), /* reload */ LdSlot_ //?
|
||||
add_ui( R_T4, R_0, PadDeadZone_HighBound),
|
||||
|
||||
/* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */
|
||||
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)), BdSlot_
|
||||
add_ui( R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_high_active */
|
||||
jump_rel(atom_offset(dead_zone_skip, exit_stick)),
|
||||
BdSlot_ mac_yield_load(), LdSlot_
|
||||
|
||||
atom_label(dead_low_active)
|
||||
/* R_T3 = left_x (from line 632 lbu; not clobbered between dead_zone_low_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
||||
* The earlier `load_byte_u(R_T3, ...)` reload was redundant and introduced a load-use hazard on the next `sub_u`.
|
||||
* R_T4 = 0x80 from the BD-slot of `dead_zone_low_check`'s branch_ne. */
|
||||
sub_u( R_T3, R_T4, R_T3), /* R_T3 = 0x80 - left_x */
|
||||
/* delta = 0x80 - left_x (positive). */
|
||||
|
||||
/* R_T4 = cube_delta */
|
||||
shift_aright(R_T4, R_T3, 2),
|
||||
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), LdSlot_ nop,
|
||||
add_u( R_T0, R_T0, R_T4),
|
||||
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||
/* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap;
|
||||
* doesn't read R_T0; R_T4 settles by the subsequent add_u). */
|
||||
load_half( R_T0, R_FloorRot, O_(V3_S2,y)), LdSlot_
|
||||
shift_aright(R_T4, R_T3, 5),
|
||||
add_u( R_T0, R_T0, R_T4),
|
||||
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||
|
||||
jump_rel(atom_offset(end_low, exit_stick)),
|
||||
BdSlot_ mac_yield_load(), LdSlot_
|
||||
|
||||
atom_label(dead_high_active)
|
||||
/* R_T3 = left_x (from line 641 lbu in dead_check_upper; not clobbered between dead_zone_high_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
||||
* The earlier `load_byte_u(R_T3, ...)` reload was redundant and introduced a load-use hazard on the next `sub_u`.
|
||||
* R_T4 = 0x80 from the BD-slot of `dead_zone_high_check`'s branch_ne. */
|
||||
sub_u( R_T3, R_T4, R_T3),
|
||||
/* delta = 0x80 - left_x (signed negative). */
|
||||
|
||||
shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */
|
||||
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), LdSlot_ nop,
|
||||
add_u( R_T0, R_T0, R_T4),
|
||||
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||
|
||||
/* R_T4 = floor_delta (signed) — moved into the load-delay slot of the floor load below. */
|
||||
load_half( R_T0, R_FloorRot, O_(V3_S2,y)), LdSlot_
|
||||
shift_aright(R_T4, R_T3, 5),
|
||||
add_u( R_T0, R_T0, R_T4),
|
||||
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||
|
||||
atom_label(no_jump_fallthrough)
|
||||
mac_yield_load(), LdSlot_
|
||||
|
||||
atom_label(exit_stick)
|
||||
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the dead-zone/exit branch. */
|
||||
mac_yield_tail(),
|
||||
};
|
||||
|
||||
enum {
|
||||
R_Cam = R_T4 atom_reg,
|
||||
R_CamPadState = R_T5 atom_reg,
|
||||
};
|
||||
typedef Struct_(Binds_PadInputCam) {
|
||||
PadState* state;
|
||||
Camera* cam;
|
||||
};
|
||||
internal MipsAtom_(pad_input_cam) atom_info(atom_bind(Binds_PadInputCam)
|
||||
, atom_reads( R_Cam, R_CamPadState, R_TapePtr)
|
||||
, atom_writes(R_Cam)
|
||||
) {
|
||||
/* Bind pop: state → R_CamPadState (R_T5), cam → R_Cam (R_T4), advance R_TapePtr by 8. */
|
||||
load_word(R_CamPadState, R_TapePtr, O_(Binds_PadInputCam,state)),
|
||||
load_word(R_Cam, R_TapePtr, O_(Binds_PadInputCam,cam)),
|
||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_PadInputCam)),
|
||||
|
||||
/* Load pad[0].buttons into R_T0; nop fills the load-delay slot. */
|
||||
load_word(R_T0, R_CamPadState, O_(PadState,buttons)), LdSlot_
|
||||
load_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
||||
|
||||
// D-pad Left → cam.pos.x -= 50. and_i fulfills BD-slot for load on R_Cam.
|
||||
LdSlot_ and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(left_x, exit_left_x)), BdSlot_ nop,
|
||||
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
||||
atom_label(exit_left_x)
|
||||
|
||||
/* D-pad Right → cam.pos.x += 50. Reuses R_T1 from Left. */
|
||||
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(right_x, exit_right_x)), BdSlot_ nop,
|
||||
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
||||
atom_label(exit_right_x)
|
||||
|
||||
/* D-pad Up → cam.pos.y -= 50. Load pos.y BEFORE the andi. */
|
||||
load_word(R_T1, R_Cam, O_(Camera,pos.y)), LdSlot_
|
||||
and_i(R_T3, R_T0, Pad_Up), branch_le_zero(R_T3, atom_offset(up_y, exit_up_y)), BdSlot_ nop,
|
||||
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
|
||||
atom_label(exit_up_y)
|
||||
|
||||
/* D-pad Down → cam.pos.y += 50. Reuses R_T1 from Up. */
|
||||
and_i(R_T3, R_T0, Pad_Down), branch_le_zero(R_T3, atom_offset(down_y, exit_down_y)), BdSlot_ nop,
|
||||
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
|
||||
atom_label(exit_down_y)
|
||||
|
||||
/* D-pad Cross → cam.pos.z -= 50. Load pos.z BEFORE the andi. */
|
||||
load_word(R_T1, R_Cam, O_(Camera,pos.z)), LdSlot_
|
||||
and_i(R_T3, R_T0, Pad_Cross), branch_le_zero(R_T3, atom_offset(cross_z, exit_cross_z)), BdSlot_ load_word(R_AtomJmp, R_TapePtr, 0), LdSlot_ // ac_yield: word 1
|
||||
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
|
||||
atom_label(exit_cross_z)
|
||||
|
||||
/* D-pad Circle → cam.pos.z += 50. Reuses R_T1 from Cross. */
|
||||
and_i(R_T3, R_T0, Pad_Circle), branch_le_zero(R_T3, atom_offset(circle_z, exit_circle_z)), BdSlot_ add_ui_self(R_TapePtr, S_(MipsCode)), // ac_yield: word 2
|
||||
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
|
||||
atom_label(exit_circle_z)
|
||||
|
||||
jump_reg(R_AtomJmp), BdSlot_ nop // ac_yield: word 3-4
|
||||
};
|
||||
|
||||
enum {
|
||||
R_PrimCursor = R_T7 atom_reg atom_type(U4*), /* Output cursor (primitive buffer) */
|
||||
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2*), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
|
||||
R_VertBase = R_T5 atom_reg atom_type(V3_S2*), /* Base address of the vertex array */
|
||||
R_OtBase = R_T6 atom_reg atom_type(U4*), /* Base address of the Ordering Table */
|
||||
#define R_PrimCursor_Code R_T7_Code
|
||||
#define R_FaceCursor_Code R_T4_Code
|
||||
#define R_VertBase_Code R_T5_Code
|
||||
#define R_OtBase_Code R_T6_Code
|
||||
};
|
||||
typedef Struct_(Binds_CubeTri) {
|
||||
U4 PrimCursor;
|
||||
V4_S2* FaceCursor;
|
||||
V3_S2* VertBase;
|
||||
U4* OtBase;
|
||||
};
|
||||
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)
|
||||
, atom_reads(R_TapePtr)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
|
||||
){
|
||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
|
||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
|
||||
internal
|
||||
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
||||
atom_writes(R_PrimCursor, R_FaceCursor)
|
||||
){
|
||||
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
||||
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
||||
// load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
||||
|
||||
LdSlot_ mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
||||
GteDelay_ load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)), LdSlot_
|
||||
GteDelay_ load_word(R_AtomJmp, R_TapePtr, 0), LdSlot_ //ac_yield: word 2,
|
||||
gte_cmdw_rotate_translate_perspective_triple,
|
||||
gte_cmdw_nclip,
|
||||
|
||||
gte_mv_from_data_r(R_T0, C2_MAC0), GteDelay_ add_ui_self(R_TapePtr, S_(MipsCode)), // ac_yield: word 1
|
||||
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
|
||||
/* BD-slot: Write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
|
||||
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
|
||||
* harmless because the OT entry that points to this prim is created later. */
|
||||
BdSlot_ store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
||||
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
||||
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)), LdSlot_
|
||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
|
||||
mac_gte_store_g4_p012(R_PrimCursor),
|
||||
gte_cmdw_rotate_translate_perspective_single,
|
||||
mac_gte_store_g4_p3(R_PrimCursor),
|
||||
|
||||
gte_cmdw_avg_sort_z4,
|
||||
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), BdSlot_ nop,
|
||||
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_G4)),
|
||||
mac_format_g4_color(R_PrimCursor,
|
||||
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
||||
/* c1 yellow */ 0xFF, 0xFF, 0x00,
|
||||
/* c2 cyan */ 0x00, 0xFF, 0xFF,
|
||||
/* c3 green */ 0x00, 0xFF, 0x00),
|
||||
// end: branch(bounds_chk)
|
||||
// end: branch(cull)
|
||||
|
||||
atom_label(cube_g4_face_exit)
|
||||
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
|
||||
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
|
||||
jump_reg(R_AtomJmp), BdSlot_ nop // ac_yield: word 3-4
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_FloorTri) {
|
||||
U4 PrimCursor;
|
||||
V3_S2* FaceCursor;
|
||||
V3_S2* VertBase;
|
||||
U4* OtBase;
|
||||
};
|
||||
internal
|
||||
MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(floor_f3)
|
||||
, atom_reads(R_TapePtr)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
|
||||
){
|
||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
|
||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
// atom_dbg_skip
|
||||
internal
|
||||
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
||||
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor)
|
||||
) {
|
||||
mac_load_tri_indices(R_FaceCursor, R_T0, R_T1, R_T2),
|
||||
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2), GteDelay_ nop2,
|
||||
gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
|
||||
gte_cmdw_nclip,
|
||||
|
||||
/* Culling (Branch forward if Backface) */
|
||||
gte_mv_from_data_r(R_T0, C2_MAC0), GteDelay_ load_word(R_AtomJmp, R_TapePtr, 0), // ac_yield: word 1
|
||||
branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), BdSlot_ add_ui_self(R_TapePtr, S_(MipsCode)), // ac_yield: word 2
|
||||
/* Format Primitive */
|
||||
mac_gte_store_f3(R_PrimCursor),
|
||||
|
||||
/* Calculate Depth */
|
||||
gte_avg_sort_z3,
|
||||
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||
/* Bounds Check OTZ < 2048 (Branch forward to skip insertion) */
|
||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), BdSlot_ nop,
|
||||
mac_format_f3_color(R_PrimCursor, 0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
|
||||
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_F3)), /* Insert into Ordering Table Linked List */
|
||||
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
|
||||
// Note(Ed): No bounds checking, should be checked before atom runs.
|
||||
// end: branch(bounds_chk)
|
||||
// end: branch(culling)
|
||||
|
||||
/* Advance Input Cursor & Yield (Both branch targets land here) */
|
||||
atom_label(floor_f3_face_exit)
|
||||
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
|
||||
jump_reg(R_AtomJmp), BdSlot_ nop // ac_yield: word 3-4
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
|
||||
internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitiveArena)
|
||||
, atom_reads( R_TapePtr, R_PrimCursor)
|
||||
, atom_writes(R_TapePtr)
|
||||
){
|
||||
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
|
||||
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)), LdSlot_
|
||||
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
|
||||
/* Calculate byte offset and store directly back to RAM */
|
||||
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
|
||||
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
#pragma endregion Baked Atoms
|
||||
@@ -0,0 +1,436 @@
|
||||
#pragma region Vendors
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
// #include <assert.h>
|
||||
// #include "libgpu.h"
|
||||
// #include "libetc.h"
|
||||
// #include "libgte.h"
|
||||
#pragma endregion Vendors
|
||||
|
||||
#pragma region Duffle Headers
|
||||
# include "duffle/gen/macs.h"
|
||||
# include "duffle/gen/offsets.h"
|
||||
|
||||
#include "duffle/word_count.metadata.h"
|
||||
|
||||
#include "duffle/dsl.h"
|
||||
#include "duffle/memory.h"
|
||||
#include "duffle/math.h"
|
||||
|
||||
#include "duffle/gcc_asm.h"
|
||||
#include "duffle/mips.h"
|
||||
#include "duffle/gp.h"
|
||||
#include "duffle/gte.h"
|
||||
#include "duffle/pad.h"
|
||||
|
||||
#include "duffle/dsl.atom.h"
|
||||
#include "duffle/lottes_tape.h"
|
||||
|
||||
#include "duffle/bios.h"
|
||||
#include "duffle/psyq.h"
|
||||
#pragma endregion Duffle Headers
|
||||
|
||||
#pragma region Duffle TUs
|
||||
#include "duffle/pad.c"
|
||||
#include "duffle/math.atom.h"
|
||||
#include "duffle/mips.atom.c"
|
||||
#include "duffle/gte.atom.c"
|
||||
#include "duffle/gp.atom.c"
|
||||
#include "duffle/pad.atom.c"
|
||||
#include "duffle/psyq.atom.c"
|
||||
#pragma endregion Duffle TUs
|
||||
|
||||
#pragma region Hello Camera Headers
|
||||
# include "gen/macs.h"
|
||||
# include "gen/offsets.h"
|
||||
# include "gen/auto_reg.h"
|
||||
|
||||
#include "hello_camera.h"
|
||||
#pragma endregion Hello Camera Headers
|
||||
|
||||
#pragma region Hello Joypad TUs
|
||||
#include "hello_camera.atom.c"
|
||||
#pragma endregion Hello Joypad TUs
|
||||
|
||||
enum {
|
||||
MemTape_Len = 512,
|
||||
|
||||
ResolveLookAtArena_Words = 1024,
|
||||
ResolveLookAtArena_Size = ResolveLookAtArena_Words * S_(MipsCode),
|
||||
|
||||
CT_InitAtomMem_Words = Kilo_(4),
|
||||
CT_InitAtomMem_Size = CT_InitAtomMem_Words * S_(MipsCode),
|
||||
};
|
||||
typedef Struct_(SMemory) {
|
||||
PrimitiveArena primitives;
|
||||
A2_OrderingTable_Buffer ordering_tbl;
|
||||
DoubleBuffer screen_buf;
|
||||
S4 active_buf_id;
|
||||
|
||||
U4 MemTape[MemTape_Len];
|
||||
|
||||
MT3_S2S4 tform_world;
|
||||
MT3_S2S4 tform_view;
|
||||
|
||||
Camera cam;
|
||||
|
||||
Ent_Cube cube;
|
||||
Ent_Floor floor;
|
||||
|
||||
PadBiosRaw pad_raw[2];
|
||||
PadState pad[2];
|
||||
|
||||
U1 ct_init_atom_mem[CT_InitAtomMem_Size];
|
||||
MipsAtom* normalize_v3s4;
|
||||
MipsAtom* gte_cross_v3s4;
|
||||
|
||||
U1 resolve_look_at_mem[ResolveLookAtArena_Size];
|
||||
MipsAtom* resolve_look_at_bundle[AtomBundle_Len(resolve_look_at)];
|
||||
};
|
||||
global SMemory smem;
|
||||
extern SMemory smem;
|
||||
|
||||
#define pad0_btn_(btn) btn & smem.pad[0].buttons
|
||||
#define pad1_btn_(btn) btn & smem.pad[1].buttons
|
||||
|
||||
I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
|
||||
gknown PrimitiveArena* pa = & smem.primitives;
|
||||
gknown B1* buf = (B1*) r_(smem.primitives.buf)[smem.active_buf_id];
|
||||
assert(pa->used + type_width < PrimitiveBuff_Len);
|
||||
B1* next = buf + pa->used;
|
||||
pa->used += type_width;
|
||||
return next;
|
||||
}
|
||||
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
|
||||
|
||||
I_ void resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4* up_in) {
|
||||
// RGA(Lengyel): Build matrix expansion of a rigid transformation. Corresponding motor is not constructed; we write the LA form for GTE.
|
||||
// Preconditions: eye != target, up_in not collinear with (target - eye).
|
||||
V3_S4 right, up, forward;
|
||||
V3_S4 ux, uy, uz;
|
||||
V3_S4 pos, off;
|
||||
|
||||
forward = target[0]; sub_v3s4(& forward, eye[0]); // RGA(Lengyel): Affine point - point = zero-weight direction.
|
||||
psy_normalize_v3s4(& forward, & uz); // RGA(Lengyel): Normalize the direction bulk. Not finite-point unitization.
|
||||
|
||||
cross_v3s4(& uz, up_in, & right); psy_normalize_v3s4(& right, & ux); // RGA(Lengyel): Complement(Wedge(forward, up_in)) -> right axis.
|
||||
cross_v3s4(& uz, & ux, & up); psy_normalize_v3s4(& up, & uy); // RGA(Lengyel): Complement(Wedge(forward, right)) -> up axis.
|
||||
|
||||
// RGA(Lengyel): matrix expansion of the world-to-camera rotation (basis rows).
|
||||
look_at->m[0][0] = ux.x; look_at->m[0][1] = ux.y; look_at->m[0][2] = ux.z;
|
||||
look_at->m[1][0] = uy.x; look_at->m[1][1] = uy.y; look_at->m[1][2] = uy.z;
|
||||
look_at->m[2][0] = uz.x; look_at->m[2][1] = uz.y; look_at->m[2][2] = uz.z;
|
||||
|
||||
pos = eye[0]; mul_v3s4(& pos, v3s4(-1,-1,-1)); // RGA(Lengyel): -eye in world coordinates (spatial bulk only; implicit weight is dropped).
|
||||
|
||||
// RGA(Lengyel): R * (-eye) is the full matrix translation column.
|
||||
// Motor translator would store half this displacement in m.xyz; GTE consumes full column.
|
||||
mul_m3s2_v3s4(look_at, & pos, & off);
|
||||
trans_m3s2( look_at, & off);
|
||||
}
|
||||
FI_ void camera_look_at_c11(Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at_c11(& c->look_at, & c->pos, target, up_in); }
|
||||
|
||||
internal void compile_init_atoms(void) {
|
||||
AtomArena ab = atomarena_make(slice_ut_arr(smem.ct_init_atom_mem));
|
||||
RegFile rf = regfile(regfile_abi_mask);
|
||||
#define ralloc() regfile_alloc(& rf)
|
||||
#define ralloc_v3() { ralloc(), ralloc(), ralloc() }
|
||||
|
||||
smem.gte_cross_v3s4 = gte_cross_v3s4(& ab,
|
||||
RegUse_(gte_cross_v3s4) {
|
||||
.a = ralloc_v3(),
|
||||
.b = ralloc_v3(),
|
||||
.x = ralloc(),
|
||||
.y = ralloc(),
|
||||
.z = ralloc(),
|
||||
});
|
||||
regfile_reset(& rf);
|
||||
|
||||
smem.normalize_v3s4 = normalize_v3s4(& ab,
|
||||
RegUse_(normalize_v3s4) {
|
||||
.res = ralloc_v3(),
|
||||
.r0 = ralloc(),
|
||||
.r1 = ralloc(),
|
||||
.r2 = ralloc(),
|
||||
.r3 = ralloc(),
|
||||
.r4 = ralloc(),
|
||||
.r5 = ralloc(),
|
||||
});
|
||||
regfile_reset(& rf);
|
||||
|
||||
assert(ab.used <= CT_InitAtomMem_Size);
|
||||
#undef ralloc
|
||||
#undef ralloc_v3
|
||||
}
|
||||
|
||||
internal void compile_resolve_look_at(void) {
|
||||
AtomBundle_resolve_look_at_R bundle = C_(void*, smem.resolve_look_at_bundle);
|
||||
AtomArena ab = atomarena_make(slice_ut_arr(smem.resolve_look_at_mem));
|
||||
RegFile rf = regfile(regfile_abi_mask);
|
||||
#define ralloc() regfile_alloc(& rf)
|
||||
#define ralloc_v3() { ralloc(), ralloc(), ralloc() }
|
||||
bundle->input_and_sub = AtomBundleEntry_(resolve_look_at, input_and_sub)(& ab,
|
||||
RegUse_(resolve_look_at_input_and_sub) {
|
||||
.target_ptr = ralloc(),
|
||||
.eye_ptr = ralloc(),
|
||||
.up_in_ptr = ralloc(),
|
||||
.up_in = ralloc_v3(),
|
||||
.r012 = ralloc_v3(),
|
||||
.r345 = {ralloc(), R_AT, ralloc() },
|
||||
});
|
||||
regfile_reset(& rf);
|
||||
|
||||
bundle->normalize_fwd_uz = smem.normalize_v3s4;
|
||||
bundle->cross_to_right = smem.gte_cross_v3s4;
|
||||
bundle->normalize_right_ux = smem.normalize_v3s4;
|
||||
bundle->cross_to_up = smem.gte_cross_v3s4;
|
||||
bundle->normalize_up_uy = smem.normalize_v3s4;
|
||||
|
||||
bundle->populate_mt3s4s2 = AtomBundleEntry_(resolve_look_at,populate_mt3s4s2)(& ab,
|
||||
RegUse_(resolve_look_at_populate_mt3s4s2){
|
||||
.look_at = ralloc(),
|
||||
.eye = ralloc(),
|
||||
.row = ralloc_v3(),
|
||||
.r0 = ralloc(),
|
||||
.r1 = ralloc(),
|
||||
.r2 = ralloc(),
|
||||
});
|
||||
|
||||
assert(ab.used <= ResolveLookAtArena_Size); // Sanity check: arena didn't overflow.
|
||||
#undef ralloc
|
||||
}
|
||||
|
||||
// Emit the resolve_look_at bundle into the tape. Called once per frame from update().
|
||||
I_ void resolve_look_at(TapeBuilder_R tb, MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4* up_in) {
|
||||
/* Typed view of the scratchpad for field-address arithmetic. */
|
||||
ResolveLookAtScratch* sp = C_scratch(ResolveLookAtScratch*);
|
||||
AtomBundle_resolve_look_at_R bundle = C_(void*, smem.resolve_look_at_bundle);
|
||||
tb_emit(tb, bundle->input_and_sub); tb_bind_(tb, Binds_ResolveLookAtSub,
|
||||
.target = target,
|
||||
.eye = eye,
|
||||
.up_in = up_in,
|
||||
);
|
||||
tb_emit(tb, bundle->normalize_fwd_uz); tb_bind_(tb, Binds_normalize_v3s4,
|
||||
.src_offset = O_(ResolveLookAtScratch,fwd),
|
||||
.dst_offset = O_(ResolveLookAtScratch,uz),
|
||||
);
|
||||
tb_emit(tb, bundle->cross_to_right); tb_bind_(tb, Binds_gte_cross_v3s4,
|
||||
.src_a = & sp->uz,
|
||||
.src_b = & sp->up_in,
|
||||
.out = & sp->right,
|
||||
);
|
||||
tb_emit(tb, bundle->normalize_right_ux); tb_bind_(tb, Binds_normalize_v3s4,
|
||||
.src_offset = O_(ResolveLookAtScratch,right),
|
||||
.dst_offset = O_(ResolveLookAtScratch,ux),
|
||||
);
|
||||
tb_emit(tb, bundle->cross_to_up); tb_bind_(tb, Binds_gte_cross_v3s4,
|
||||
.src_a = & sp->uz,
|
||||
.src_b = & sp->ux,
|
||||
.out = & sp->up,
|
||||
);
|
||||
tb_emit(tb, bundle->normalize_up_uy); tb_bind_(tb, Binds_normalize_v3s4,
|
||||
.src_offset = O_(ResolveLookAtScratch,up),
|
||||
.dst_offset = O_(ResolveLookAtScratch,uy),
|
||||
);
|
||||
tb_emit(tb, bundle->populate_mt3s4s2); tb_bind_(tb, Binds_ResolveLookAt_PopulateMT3S4S2,
|
||||
.look_at = look_at,
|
||||
);
|
||||
}
|
||||
FI_ void camera_look_at(TapeBuilder_R tb, Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at(tb, & c->look_at, & c->pos, target, up_in); }
|
||||
|
||||
void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
{
|
||||
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
|
||||
|
||||
// Pad Input
|
||||
{
|
||||
tb.used = 0; tb_scope_run(& tb) {
|
||||
// Grab latest state from bios.
|
||||
tb_emit_(pad_bios_snapshot);
|
||||
tb_data(& tb, u4_(& smem.pad_raw[0]));
|
||||
tb_data(& tb, u4_(& smem.pad[0]));
|
||||
// tb_emit_(pad_bios_snapshot);
|
||||
// tb_data_(raw, & smem.pad_raw[1]);
|
||||
// tb_data_(state, & smem.pad[1]);
|
||||
|
||||
tb_emit_(pad_input_cam);
|
||||
tb_data(& tb, u4_(& smem.pad[0]));
|
||||
tb_data(& tb, u4_(& smem.cam));
|
||||
|
||||
// tb_emit_(pad_input_cube_rotation);
|
||||
// tb_data_(state, & smem.pad[0]);
|
||||
// tb_data_(cube_rot, & smem.cube.rot);
|
||||
// tb_data_(floor_rot, & smem.floor.rot);
|
||||
}
|
||||
}
|
||||
|
||||
orderingtbl_clear_reverse(ordering_buf, OrderingTbl_Len);
|
||||
|
||||
// Update the position based on acceleration and velocity
|
||||
gknown V3_S4_R pos = & smem.cube.pos;
|
||||
gknown V3_S4_R vel = & smem.cube.vel;
|
||||
gknown V3_S4_R acc = & smem.cube.accel;
|
||||
add_v3s4(vel, acc[0]);
|
||||
add_v3s4_fp(pos, vel[0]);
|
||||
|
||||
if (pos->y + 150 > smem.floor.pos.y) vel->y *= -1;
|
||||
|
||||
// Prep
|
||||
S4 nclip = 0;
|
||||
S4 orderingtbl_z = 0;
|
||||
A2_S2 p; //???
|
||||
S4 flag; //????
|
||||
|
||||
B4 use_c11_path = false;
|
||||
if (use_c11_path) {
|
||||
camera_look_at_c11(& smem.cam, & smem.cube.pos, & v3s4(0, -fp_one, 0));
|
||||
}
|
||||
if (use_c11_path == false)
|
||||
{
|
||||
tb.used = 0; tb_scope_run(& tb) {
|
||||
camera_look_at(& tb, & smem.cam, & smem.cube.pos, & v3s4(0, -fp_one, 0));
|
||||
}
|
||||
}
|
||||
|
||||
// Draw cube
|
||||
if (1)
|
||||
{
|
||||
mt3s2s4_rotation (& smem.cube.rot, & smem.tform_world);
|
||||
mt3s2s4_translation(& smem.tform_world, & smem.cube.pos);
|
||||
mt3s2s4_scale (& smem.tform_world, & smem.cube.scale);
|
||||
|
||||
// Combine world and look_at matrix.
|
||||
gte_comp_coord_m3s2(& smem.cam.look_at, & smem.tform_world, & smem.tform_view);
|
||||
gte_matrix_set_rotation (& smem.tform_view);
|
||||
gte_matrix_set_translation(& smem.tform_view);
|
||||
|
||||
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||
U4 prim_cursor = prim_base + pa->used;
|
||||
|
||||
tb.used = 0; tb_scope(& tb) {
|
||||
tb_emit(& tb, rbind_cube_g4_face);
|
||||
tb_data(& tb, prim_cursor);
|
||||
tb_data(& tb, u4_(smem.cube.faces));
|
||||
tb_data(& tb, u4_(smem.cube.verts));
|
||||
tb_data(& tb, u4_(ordering_buf));
|
||||
|
||||
for (U4 i = 0; i < Cube_num_faces; i++) {
|
||||
// Two triangles per quad face: (x,y,z) and (x,z,w)
|
||||
tb_emit(& tb, cube_g4_face);
|
||||
}
|
||||
|
||||
tb_emit(& tb, sync_primitive_arena);
|
||||
tb_data(& tb, u4_(& pa->used));
|
||||
tb_data(& tb, prim_base);
|
||||
}
|
||||
tape_run(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
|
||||
|
||||
// smem.cube.rot.y += 30;
|
||||
}
|
||||
// Draw floor
|
||||
if (1)
|
||||
{
|
||||
mt3s2s4_rotation (& smem.floor.rot, & smem.tform_world);
|
||||
mt3s2s4_translation(& smem.tform_world, & smem.floor.pos);
|
||||
mt3s2s4_scale (& smem.tform_world, & smem.floor.scale);
|
||||
|
||||
// Combine world and look_at matrix.
|
||||
gte_comp_coord_m3s2(& smem.cam.look_at, & smem.tform_world, & smem.tform_view);
|
||||
|
||||
gte_matrix_set_rotation (& smem.tform_view);
|
||||
gte_matrix_set_translation(& smem.tform_view);
|
||||
|
||||
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||
U4 prim_cursor = prim_base + pa->used;
|
||||
|
||||
// TODO(Ed): We should do a bounds check beforehand to confirm pa can hold all tris?
|
||||
// The tape atoms in-flight should not need to care.
|
||||
|
||||
// Prepare the tape. (Push protocol to tape)
|
||||
tb.used = 0; tb_scope(& tb) {
|
||||
// tb_emit(& tb, set_gte_mt3s2s4);
|
||||
// tb_data(& tb, u4_(& smem.tform_view));
|
||||
|
||||
tb_emit(& tb, rbind_floor_f3_face);
|
||||
// TODO(Ed): Just use a single context struct ref?
|
||||
tb_data(& tb, prim_cursor);
|
||||
tb_data(& tb, u4_(smem.floor.faces));
|
||||
tb_data(& tb, u4_(smem.floor.verts));
|
||||
tb_data(& tb, u4_(ordering_buf));
|
||||
for (U4 i = 0; i < Floor_num_faces; i++) {
|
||||
tb_emit(& tb, floor_f3_face);
|
||||
}
|
||||
// After floor_f3_face iterations complete, the primitive arena's used counter needs updating.
|
||||
tb_emit(& tb, sync_primitive_arena);
|
||||
tb_data(& tb, u4_(& pa->used));
|
||||
tb_data(& tb, prim_base);
|
||||
}
|
||||
tape_run(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
|
||||
|
||||
// C-side state (pa->used) has already been updated by the tape!
|
||||
// smem.floor.rot.y += 5;
|
||||
}
|
||||
}
|
||||
|
||||
void render(void) {
|
||||
}
|
||||
|
||||
void gp_display_frame(DoubleBuffer* screen_buf, S4* active_buf_id, U4* ordering_buf, PrimitiveArena* pa) {
|
||||
draw_sync(0);
|
||||
vsync(0);
|
||||
displayenv_put(& r_(screen_buf->display)[active_buf_id[0] ]);
|
||||
drawenv_put (& r_(screen_buf->draw) [active_buf_id[0] ]);
|
||||
{
|
||||
draw_orderingtbl(ordering_buf + OrderingTbl_Len - 1);
|
||||
pa->used = 0;
|
||||
}
|
||||
active_buf_id[0] = ! active_buf_id[0]; // Swap current buffer
|
||||
}
|
||||
|
||||
int main(void)
|
||||
{
|
||||
smem = (SMemory){0};
|
||||
// smem.primitives.used = 0;
|
||||
// smem.active_buf_id = 0;
|
||||
smem.cam.pos = v3s4(500, -1000, -1500);
|
||||
/*Persistent Entity Setup*/{
|
||||
ent_cube128_init(& smem.cube.verts, & smem.cube.faces); {
|
||||
Ent_Cube* cube = & smem.cube;
|
||||
cube->rot = v3s2(0, 0, 0);
|
||||
cube->scale = v3s4_fp_one();
|
||||
cube->accel = v3s4(0, 1, 0);
|
||||
cube->pos = v3s4(0, -400, 1800);
|
||||
}
|
||||
ent_floor_init(& smem.floor.verts, & smem.floor.faces); {
|
||||
Ent_Floor* floor = & smem.floor;
|
||||
floor->rot = v3s2(0, 0, 0);
|
||||
floor->pos = v3s4(0, 450, 1800);
|
||||
floor->scale = v3s4_fp_one();
|
||||
}
|
||||
}
|
||||
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape)); {
|
||||
reset_graph(0);
|
||||
/* Direct BIOS: poll both ports during VBlank. */
|
||||
pad_bios_init_start(& smem.pad_raw[0], & smem.pad_raw[1]);
|
||||
|
||||
compile_init_atoms();
|
||||
compile_resolve_look_at();
|
||||
|
||||
/* Pinned registers for the GPU init atom. */
|
||||
register U4* io_base_addr rgcc(R_IO_BaseAddr) = u4_r(IO_BASE_ADDR);
|
||||
register DoubleBuffer* screen_buf rgcc(R_ScreenBuf) = & smem.screen_buf;
|
||||
tb.used = 0; tb_scope_run(& tb) {
|
||||
tb_emit(& tb, screen_env_init);
|
||||
tb_emit(& tb, gp_screen_init);
|
||||
}
|
||||
}
|
||||
while (1) {
|
||||
gknown S4* active_buf_id = & smem.active_buf_id;
|
||||
gknown U4* ordering_buf = r_(smem.ordering_tbl)[active_buf_id[0]];
|
||||
gknown PrimitiveArena* pa = & smem.primitives;
|
||||
update(pa, ordering_buf);
|
||||
render();
|
||||
gp_display_frame(& smem.screen_buf, active_buf_id, ordering_buf, pa);
|
||||
};
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,102 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "duffle/dsl.h"
|
||||
# include "duffle/math.h"
|
||||
# include "duffle/gp.h"
|
||||
# include "duffle/pad.h"
|
||||
#endif
|
||||
|
||||
enum {
|
||||
// PrimitiveBuff_Len = 4096,
|
||||
// OrderingTbl_Len = 2048,
|
||||
PrimitiveBuff_Len = 131072,
|
||||
OrderingTbl_Len = 8192,
|
||||
};
|
||||
|
||||
enum {
|
||||
ScreenRes_X = 320,
|
||||
ScreenRes_Y = 240,
|
||||
ScreenZ = 320,
|
||||
ScreenRes_CenterX = (ScreenRes_X >> 1),
|
||||
ScreenRes_CenterY = (ScreenRes_Y >> 1),
|
||||
};
|
||||
|
||||
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
||||
typedef Array_(OrderingTable_Buffer, 2);
|
||||
|
||||
typedef B1 PrimitiveBuffer[PrimitiveBuff_Len];
|
||||
typedef Array_(PrimitiveBuffer, 2);
|
||||
typedef Struct_(PrimitiveArena) {
|
||||
A2_PrimitiveBuffer buf;
|
||||
U4 used;
|
||||
};
|
||||
|
||||
#define Cube_num_verts 8
|
||||
typedef Array_(V3_S2, Cube_num_verts);
|
||||
#define Cube_num_faces 6
|
||||
typedef Array_(V4_S2, Cube_num_faces);
|
||||
I_ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
||||
LP_ A8_V3_S2 baked_verts = (A8_V3_S2) {
|
||||
{ -128, -128, -128 },
|
||||
{ 128, -128, -128 },
|
||||
{ 128, -128, 128 },
|
||||
{ -128, -128, 128 },
|
||||
{ -128, 128, -128 },
|
||||
{ 128, 128, -128 },
|
||||
{ 128, 128, 128 },
|
||||
{ -128, 128, 128 }
|
||||
};
|
||||
LP_ A6_V4_S2 baked_faces = (A6_V4_S2) {
|
||||
{ 3, 2, 0, 1 },
|
||||
{ 0, 1, 4, 5 },
|
||||
{ 4, 5, 7, 6 },
|
||||
{ 1, 2, 5, 6 },
|
||||
{ 2, 3, 6, 7 },
|
||||
{ 3, 0, 7, 4 },
|
||||
};
|
||||
mem_copy(u4_(verts), u4_(& baked_verts), S_(A8_V3_S2) );
|
||||
mem_copy(u4_(faces), u4_(& baked_faces), S_(A6_V4_S2) );
|
||||
return;
|
||||
}
|
||||
typedef Struct_(Ent_Cube) {
|
||||
V3_S4 accel;
|
||||
V3_S4 vel;
|
||||
V3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
|
||||
V3_S4 scale;
|
||||
V3_S2 rot;
|
||||
A8_V3_S2 verts;
|
||||
A6_V4_S2 faces;
|
||||
};
|
||||
|
||||
#define Floor_num_verts 4
|
||||
typedef Array_(V3_S2, Floor_num_verts);
|
||||
#define Floor_num_faces 2
|
||||
typedef Array_(V3_S2, Floor_num_faces);
|
||||
I_ void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
|
||||
LP_ A4_V3_S2 baked_verts = (A4_V3_S2) {
|
||||
{ -900, 0, -900 },
|
||||
{ -900, 0, 900 },
|
||||
{ 900, 0, -900 },
|
||||
{ 900, 0, 900 },
|
||||
};
|
||||
LP_ A2_V3_S2 baked_faces = (A2_V3_S2) {
|
||||
{ 0, 1, 2 },
|
||||
{ 1, 3, 2 },
|
||||
};
|
||||
mem_copy(u4_(verts), u4_(& baked_verts), S_(A4_V3_S2));
|
||||
mem_copy(u4_(faces), u4_(& baked_faces), S_(A2_V3_S2));
|
||||
};
|
||||
typedef Struct_(Ent_Floor) {
|
||||
V3_S4 accel;
|
||||
V3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
|
||||
V3_S4 scale;
|
||||
V3_S2 rot;
|
||||
A4_V3_S2 verts;
|
||||
A2_V3_S2 faces;
|
||||
};
|
||||
|
||||
typedef Struct_(Camera) {
|
||||
P3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
|
||||
V3_S2 rot;
|
||||
MT3_S2S4 look_at;
|
||||
};
|
||||
@@ -5,20 +5,20 @@
|
||||
#pragma region hello_gte_tape
|
||||
|
||||
|
||||
// --- atom: cube_g4_face (87 words) ---
|
||||
// --- atom: cube_g4_face (77 words) ---
|
||||
|
||||
#define _atom_offset_cull_cube_g4_face_exit 48
|
||||
#define _atom_offset_bounds_chk_cube_g4_face_exit 12
|
||||
#define _atom_offset_cull_cube_g4_face_exit 42
|
||||
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
|
||||
|
||||
enum {
|
||||
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
|
||||
atom_offset_bounds_chk_cube_g4_face_exit = _atom_offset_bounds_chk_cube_g4_face_exit,
|
||||
};
|
||||
|
||||
// --- atom: floor_f3_face (66 words) ---
|
||||
// --- atom: floor_f3_face (58 words) ---
|
||||
|
||||
#define _atom_offset_culling_floor_f3_face_exit 29
|
||||
#define _atom_offset_bounds_chk_floor_f3_face_exit 13
|
||||
#define _atom_offset_culling_floor_f3_face_exit 25
|
||||
#define _atom_offset_bounds_chk_floor_f3_face_exit 16
|
||||
|
||||
enum {
|
||||
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
|
||||
@@ -0,0 +1,29 @@
|
||||
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
|
||||
// Source: C:\projects\Pikuma\ps1\code\hello_gte\hello_gte.tape.c
|
||||
#pragma once
|
||||
|
||||
#pragma region hello_gte.tape
|
||||
|
||||
|
||||
// --- atom: cube_g4_face (77 words) ---
|
||||
|
||||
#define _atom_offset_cull_cube_g4_face_exit 42
|
||||
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
|
||||
|
||||
enum {
|
||||
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
|
||||
atom_offset_bounds_chk_cube_g4_face_exit = _atom_offset_bounds_chk_cube_g4_face_exit,
|
||||
};
|
||||
|
||||
// --- atom: floor_f3_face (58 words) ---
|
||||
|
||||
#define _atom_offset_culling_floor_f3_face_exit 25
|
||||
#define _atom_offset_bounds_chk_floor_f3_face_exit 16
|
||||
|
||||
enum {
|
||||
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
|
||||
atom_offset_bounds_chk_floor_f3_face_exit = _atom_offset_bounds_chk_floor_f3_face_exit,
|
||||
};
|
||||
|
||||
#pragma endregion hello_gte.tape
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#include "stdio.h"
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include "assert.h"
|
||||
#include <assert.h>
|
||||
// #include "libgpu.h"
|
||||
// #include "libetc.h"
|
||||
// #include "libgte.h"
|
||||
@@ -14,16 +14,16 @@
|
||||
#include "duffle/gp.h"
|
||||
#include "duffle/gte.h"
|
||||
|
||||
# include "duffle/gen/duffle.macs.h"
|
||||
# include "duffle/gen/duffle.offsets.h"
|
||||
# include "duffle/gen/macs.h"
|
||||
# include "duffle/gen/offsets.h"
|
||||
#include "duffle/atom_dsl.h"
|
||||
#include "duffle/lottes_tape.h"
|
||||
#include "duffle/word_count.metadata.h"
|
||||
|
||||
# include "tape_atom.metadata.h"
|
||||
# include "gen/gte_hello.offsets.h"
|
||||
# include "gen/offsets.h"
|
||||
#include "hello_gte.h"
|
||||
|
||||
#include "hello_gte_tape.c"
|
||||
#include "hello_gte.tape.c"
|
||||
|
||||
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
||||
typedef Array_(OrderingTable_Buffer, 2);
|
||||
@@ -99,8 +99,13 @@ typedef Struct_(Ent_Floor) {
|
||||
A2_V3_S2 faces;
|
||||
};
|
||||
|
||||
enum { scratchpad_size = 1024, };
|
||||
enum {
|
||||
Scratchpad_Len = 1024,
|
||||
MemTape_Len = 512,
|
||||
};
|
||||
typedef Struct_(SMemory) {
|
||||
U4 MemTape[MemTape_Len];
|
||||
|
||||
DoubleBuffer screen_buf;
|
||||
A2_OrderingTable_Buffer ordering_tbl;
|
||||
PrimitiveArena primitives;
|
||||
@@ -117,7 +122,7 @@ global SMemory smem;
|
||||
extern SMemory smem;
|
||||
|
||||
// TODO(Ed):
|
||||
FI_ U4* spad_warm(MipsAtom atom) {
|
||||
FI_ U4* spad_warm(Slice_MipsCode atom) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
@@ -182,6 +187,7 @@ void gp_display_frame(DoubleBuffer* screen_buf, S4* active_buf_id, U4* ordering_
|
||||
void render(void) {
|
||||
}
|
||||
|
||||
GCC_OPTIMIZATION_DISABLE
|
||||
void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
{
|
||||
orderingtbl_clear_reverse(ordering_buf, OrderingTbl_Len);
|
||||
@@ -207,6 +213,8 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
A2_S2 p; //???
|
||||
S4 flag; //????
|
||||
|
||||
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
|
||||
|
||||
// Draw Cube
|
||||
if (0)
|
||||
{
|
||||
@@ -259,9 +267,8 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||
U4 prim_cursor = prim_base + pa->used;
|
||||
|
||||
LP_ U4 mem_temp_tape[512]; FArena tape_arena; farena_init(& tape_arena, slice_ut_arr(mem_temp_tape));
|
||||
TapeBuilder tb = tb_make_old(&tape_arena); tb_scope(& tb) {
|
||||
tb_emit(& tb, code_rbind_cube_g4_face);
|
||||
tb.used = 0; tb_scope(& tb) {
|
||||
tb_emit(& tb, rbind_cube_g4_face);
|
||||
tb_data(& tb, prim_cursor);
|
||||
tb_data(& tb, u4_(smem.cube.faces));
|
||||
tb_data(& tb, u4_(smem.cube.verts));
|
||||
@@ -269,10 +276,10 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
|
||||
for (U4 i = 0; i < Cube_num_faces; i++) {
|
||||
// Two triangles per quad face: (x,y,z) and (x,z,w)
|
||||
tb_emit(& tb, code_cube_g4_face);
|
||||
tb_emit(& tb, cube_g4_face);
|
||||
}
|
||||
|
||||
tb_emit(& tb, code_sync_primitive_arena);
|
||||
tb_emit(& tb, sync_primitive_arena);
|
||||
tb_data(& tb, u4_(& pa->used));
|
||||
tb_data(& tb, prim_base);
|
||||
}
|
||||
@@ -344,48 +351,40 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||
U4 prim_cursor = prim_base + pa->used;
|
||||
|
||||
// TODO(Ed): We should do a bounds check beforehand to confirm pa can hold all tris.
|
||||
// TODO(Ed): We should do a bounds check beforehand to confirm pa can hold all tris?
|
||||
// The tape atoms in-flight should not need to care.
|
||||
|
||||
// Prepare the tape. (Push protocol to tape)
|
||||
LP_ U4 mem_temp_tape[512];
|
||||
TapeBuilder tb = tb_make(slice_ut_arr(mem_temp_tape)); tb_scope(& tb) {
|
||||
tb_emit(& tb, code_set_gte_world);
|
||||
tb.used = 0; tb_scope(& tb) {
|
||||
tb_emit(& tb, set_gte_world);
|
||||
tb_data(& tb, u4_(& smem.tform_world));
|
||||
|
||||
tb_emit(& tb, code_rbind_floor_f3_face);
|
||||
tb_emit(& tb, rbind_floor_f3_face);
|
||||
// TODO(Ed): Just use a single context struct ref
|
||||
tb_data(& tb, prim_cursor);
|
||||
tb_data(& tb, u4_(smem.floor.faces));
|
||||
tb_data(& tb, u4_(smem.floor.verts));
|
||||
tb_data(& tb, u4_(ordering_buf));
|
||||
for (U4 i = 0; i < Floor_num_faces; i++) {
|
||||
tb_emit(& tb, code_floor_f3_face);
|
||||
tb_emit(& tb, floor_f3_face);
|
||||
}
|
||||
// After code_floor_f3_face iterations complete, the primitive arena's used counter needs updating.
|
||||
tb_emit(& tb, code_sync_primitive_arena);
|
||||
// After floor_f3_face iterations complete, the primitive arena's used counter needs updating.
|
||||
tb_emit(& tb, sync_primitive_arena);
|
||||
tb_data(& tb, u4_(& pa->used));
|
||||
tb_data(& tb, prim_base);
|
||||
}
|
||||
|
||||
tape_run(tb_slice(tb));// Fire off the tape.
|
||||
|
||||
// C-side state (pa->used) has already been updated by the tape!
|
||||
smem.floor.rot.y += 5;
|
||||
}
|
||||
// --- TAPE DIAGNOSTICS ---
|
||||
if (1)
|
||||
if (0)
|
||||
{
|
||||
LP_ U4 mem_temp_tape[512]; FArena tape_arena; farena_init(& tape_arena, slice_ut_arr(mem_temp_tape));
|
||||
TapeBuilder tb = tb_make_old(& tape_arena); tb_scope(& tb) {
|
||||
// Skip set_gte_world atom for diagnostics to isolate the triangle loop
|
||||
for (U4 i = 0; i < Floor_num_faces; i++) {
|
||||
// =======================================================
|
||||
// SWAP EMIT TO TEST DIFFERENT PARTS OF THE PIPELINE:
|
||||
// =======================================================
|
||||
// 1. code_diag_yield -> Tests Tape Engine jump logic
|
||||
// 2. code_diag_color -> Tests OT and Prim Arena memory
|
||||
// 3. code_diag_gte -> Tests Vertex arrays and GTE Math
|
||||
// tb_emit(& tb, code_diag_yield);
|
||||
// tb_emit(& tb, code_diag_color);
|
||||
// tb_emit(& tb, code_diag_gte);
|
||||
@@ -394,9 +393,9 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
B1* prim_cursor = (B1*)r_(pa->buf)[smem.active_buf_id] + pa->used;
|
||||
tape_run(tb_slice(tb));
|
||||
pa->used = (U4)prim_cursor - (U4)r_(pa->buf)[smem.active_buf_id];
|
||||
smem.floor.rot.y += 5;
|
||||
}
|
||||
}
|
||||
GCC_OPTIMIZATION_ENABLE
|
||||
|
||||
int main(void)
|
||||
{
|
||||
@@ -0,0 +1,218 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# include "duffle/gen/macs.h"
|
||||
# include "duffle/gen/offsets.h"
|
||||
# include "duffle/atom_dsl.h"
|
||||
# include "duffle/lottes_tape.h"
|
||||
# include "duffle/word_count.metadata.h"
|
||||
# include "gen/offsets.h"
|
||||
# include "hello_gte.h"
|
||||
#endif
|
||||
|
||||
#pragma region MACs (Mips Atom components)
|
||||
|
||||
|
||||
|
||||
#pragma endregion MACs
|
||||
|
||||
#pragma region Baked Atoms
|
||||
|
||||
/* DIAGNOSTIC 1: Pure tape loop test */
|
||||
internal MipsAtom_(diag_yield) { mac_yield() };
|
||||
|
||||
/* DIAGNOSTIC 2: Pure memory test (No GTE). Draws a fixed cyan triangle. */
|
||||
internal MipsAtom_(diag_color) {
|
||||
store_word( R_0, R_T7, 0),
|
||||
load_upper_i(R_AT, gp0_cmd_poly_f3 << 8 | 0xFF), /* High: MipsCode Poly_F3(0x20) + Color B:FF */
|
||||
or_i_self( R_AT, 0xFF00), /* Low: Color G:FF, R:00 (Cyan) */
|
||||
store_word( R_AT, R_T7, 4),
|
||||
|
||||
/* Fake coordinates - Swapped winding order to prevent GPU culling! */
|
||||
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 8), /* (16, 16) */
|
||||
load_upper_i(R_AT, 0x0050), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 12), /* (80, 16) */
|
||||
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0050), store_word(R_AT, R_T7, 16), /* (16, 80) */
|
||||
|
||||
add_ui( R_T1, R_0, 10),
|
||||
shift_lleft_self(R_T1, S_(U4)/2),
|
||||
add_u_self( R_T1, R_T6),
|
||||
|
||||
load_word( R_AT, R_T1, 0),
|
||||
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits),
|
||||
store_word( R_AT, R_T7, 0),
|
||||
shift_lleft(R_AT, R_T7, S_(PolyTag_len_bits)), shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
|
||||
or_u_self( R_AT, R_V0),
|
||||
store_word( R_AT, R_T1, 0),
|
||||
|
||||
add_ui(R_T7, R_T7, 20),
|
||||
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
/* DIAGNOSTIC 3: Pure GTE test (No Memory Writes) */
|
||||
internal MipsAtom_(diag_gte) {
|
||||
/* Load 3 indices */
|
||||
load_half_u(R_T0, R_T4, 0),
|
||||
load_half_u(R_T1, R_T4, 2),
|
||||
load_half_u(R_T2, R_T4, 4),
|
||||
|
||||
/* Load Vertices into GTE */
|
||||
shift_lleft( R_AT, R_T0, 3), add_u( R_AT, R_AT, R_T5),
|
||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
|
||||
shift_lleft( R_AT, R_T1, 3), add_u(R_AT, R_AT, R_T5),
|
||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
||||
|
||||
shift_lleft(R_AT, R_T2, 3), add_u(R_AT, R_AT, R_T5),
|
||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
||||
|
||||
/* Run Math */
|
||||
nop2, gte_cmdw_rtpt,
|
||||
nop2, gte_cmdw_nclip,
|
||||
nop2,
|
||||
|
||||
/* Advance Face Cursor and Yield */
|
||||
add_ui(R_T4, R_T4, 8),
|
||||
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_CubeTri) {
|
||||
U4 PrimCursor;
|
||||
V4_S2* FaceCursor;
|
||||
V3_S2* VertBase;
|
||||
U4* OtBase;
|
||||
};
|
||||
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)
|
||||
, atom_reads(R_TapePtr)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
){
|
||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
|
||||
internal
|
||||
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
||||
atom_writes(R_PrimCursor, R_FaceCursor)
|
||||
){
|
||||
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
||||
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
||||
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
||||
|
||||
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
|
||||
nop2, gte_cmdw_rotate_translate_perspective_triple, // required cpu -> gte delay slot
|
||||
gte_cmdw_nclip,
|
||||
|
||||
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
|
||||
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)), nop,
|
||||
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
||||
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
||||
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
|
||||
mac_gte_store_g4_p012(),
|
||||
gte_cmdw_rotate_translate_perspective_single,
|
||||
mac_gte_store_g4_p3(),
|
||||
|
||||
gte_cmdw_avg_sort_z4,
|
||||
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||
|
||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
|
||||
mac_insert_ot_tag_g4(),
|
||||
mac_format_g4_color(
|
||||
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
||||
/* c1 yellow */ 0xFF, 0xFF, 0x00,
|
||||
/* c2 cyan */ 0x00, 0xFF, 0xFF,
|
||||
/* c3 green */ 0x00, 0xFF, 0x00),
|
||||
// end: branch(bounds_chk)
|
||||
// end: branch(cull)
|
||||
|
||||
atom_label(cube_g4_face_exit)
|
||||
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
|
||||
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_FloorTri) {
|
||||
U4 PrimCursor;
|
||||
V3_S2* FaceCursor;
|
||||
V3_S2* VertBase;
|
||||
U4* OtBase;
|
||||
};
|
||||
internal
|
||||
MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(floor_f3)
|
||||
, atom_reads(R_TapePtr)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
){
|
||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
// atom_dbg_skip
|
||||
internal
|
||||
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
||||
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor)
|
||||
) {
|
||||
mac_load_tri_indices( R_T0, R_T1, R_T2),
|
||||
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
|
||||
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
|
||||
gte_cmdw_nclip,
|
||||
|
||||
/* Culling (Branch forward if Backface) */
|
||||
gte_mv_from_data_r(R_T0, C2_MAC0),
|
||||
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
|
||||
/* Format Primitive */
|
||||
mac_gte_store_f3(),
|
||||
|
||||
/* Calculate Depth */
|
||||
gte_avg_sort_z3,
|
||||
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||
/* Bounds Check OTZ < OrderingTbl_Len (Branch forward to skip insertion) */
|
||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
|
||||
mac_format_f3_color(0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
|
||||
mac_insert_ot_tag_f3(), /* Insert into Ordering Table Linked List */
|
||||
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
|
||||
// Note(Ed): No bounds checking, should be checked before atom runs.
|
||||
// end: branch(bounds_chk)
|
||||
// end: branch(culling)
|
||||
|
||||
/* Advance Input Cursor & Yield (Both branch targets land here) */
|
||||
atom_label(floor_f3_face_exit)
|
||||
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
|
||||
internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitiveArena)
|
||||
, atom_reads( R_TapePtr, R_PrimCursor)
|
||||
, atom_writes(R_TapePtr)
|
||||
){
|
||||
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
|
||||
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
|
||||
/* Calculate byte offset and store directly back to RAM */
|
||||
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
|
||||
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
#pragma endregion Baked Atoms
|
||||
@@ -0,0 +1,41 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
#pragma once
|
||||
#endif
|
||||
// Auto-generated by ps1_meta.lua — DO NOT EDIT
|
||||
// Directory: C:\projects\Pikuma\ps1\code\hello_joypad/
|
||||
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.c
|
||||
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.h
|
||||
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.atom.c
|
||||
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
|
||||
|
||||
#ifndef WORD_COUNT
|
||||
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
||||
#endif
|
||||
|
||||
#define mac_put_disp_env(reg_transfer, reg_base, port) \
|
||||
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port)
|
||||
WORD_COUNT(mac_put_disp_env, 5)
|
||||
|
||||
#define mac_put_draw_env(reg_transfer, reg_base, port) \
|
||||
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port) /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */ \
|
||||
, mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port) /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */ \
|
||||
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[1] TextureWindow (tw=(0,0)) */ \
|
||||
, mac_gcmd_push(enc_gp0_draw_area_tl_word(0, ScreenRes_Y), reg_transfer, reg_base, port) /* code[2] DrawArea top-left (clip.x=0, clip.y=ScreenRes_Y=240) */ \
|
||||
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) /* code[3] DrawArea bottom-right (clip.x+w=320, clip.y+h=480) */ \
|
||||
, mac_gcmd_push(gp0_word_set_draw_offset(), reg_transfer, reg_base, port) /* code[4] DrawOffset (ofs=(0,0)) — bare-cmd word; the GPU uses the current state machine. */ \
|
||||
, mac_gcmd_push(gp0_word_dr_env_mask(), reg_transfer, reg_base, port) /* code[5] Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit. */ \
|
||||
, mac_gcmd_push(gp0_word_dr_env_bg_color_cmd(1, 7, 7, 7), reg_transfer, reg_base, port) /* code[6] Initial-bg-color + auto-clear (isbg=1, r=7, g=7, b=7). */ \
|
||||
, mac_gcmd_push(gp0_word_dr_env_draw_mode(1), reg_transfer, reg_base, port) /* code[7] Re-assert DrawMode with isbg=1 (isbg-flag set; the 0xE1 cmd byte plus isbg only). */ /* code[8..10] Padding (NOP — GPU discards; the DR_ENV requires 16 words total). */ \
|
||||
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) /* code[11..12] TextureWindow bottom-right (tw.x+tw.w=0, tw.y+tw.h=0) — libpsyx emits twice. */ \
|
||||
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[13..14] Padding (NOP) — completes the 16-word packet. */ \
|
||||
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port)
|
||||
WORD_COUNT(mac_put_draw_env, 16)
|
||||
|
||||
@@ -0,0 +1,76 @@
|
||||
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
|
||||
// Directory: C:\projects\Pikuma\ps1\code\hello_joypad\
|
||||
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.c
|
||||
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.h
|
||||
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.atom.c
|
||||
#pragma once
|
||||
|
||||
#pragma region hello_joypad
|
||||
|
||||
|
||||
// --- atom: cube_g4_face (76 words) ---
|
||||
|
||||
#define _atom_offset_cull_cube_g4_face_exit 41
|
||||
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
|
||||
|
||||
enum {
|
||||
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
|
||||
atom_offset_bounds_chk_cube_g4_face_exit = _atom_offset_bounds_chk_cube_g4_face_exit,
|
||||
};
|
||||
|
||||
// --- atom: floor_f3_face (58 words) ---
|
||||
|
||||
#define _atom_offset_culling_floor_f3_face_exit 25
|
||||
#define _atom_offset_bounds_chk_floor_f3_face_exit 16
|
||||
|
||||
enum {
|
||||
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
|
||||
atom_offset_bounds_chk_floor_f3_face_exit = _atom_offset_bounds_chk_floor_f3_face_exit,
|
||||
};
|
||||
|
||||
// --- atom: pad_bios_snapshot (78 words) ---
|
||||
|
||||
#define _atom_offset_snap_root_skip_disconnected 8
|
||||
#define _atom_offset_disconnected_snap_end 61
|
||||
#define _atom_offset_case_2_id_dispatch 8
|
||||
#define _atom_offset_pending_snap_end 51
|
||||
#define _atom_offset_id_dispatch_try_analog_stick 11
|
||||
#define _atom_offset_id_dispatch_snap_end 38
|
||||
#define _atom_offset_try_analog_stick_try_analog_pad 12
|
||||
#define _atom_offset_analog_stick_snap_end 24
|
||||
#define _atom_offset_try_analog_pad_try_unsupported 11
|
||||
#define _atom_offset_analog_pad_snap_end 10
|
||||
|
||||
enum {
|
||||
atom_offset_snap_root_skip_disconnected = _atom_offset_snap_root_skip_disconnected,
|
||||
atom_offset_disconnected_snap_end = _atom_offset_disconnected_snap_end,
|
||||
atom_offset_case_2_id_dispatch = _atom_offset_case_2_id_dispatch,
|
||||
atom_offset_pending_snap_end = _atom_offset_pending_snap_end,
|
||||
atom_offset_id_dispatch_try_analog_stick = _atom_offset_id_dispatch_try_analog_stick,
|
||||
atom_offset_id_dispatch_snap_end = _atom_offset_id_dispatch_snap_end,
|
||||
atom_offset_try_analog_stick_try_analog_pad = _atom_offset_try_analog_stick_try_analog_pad,
|
||||
atom_offset_analog_stick_snap_end = _atom_offset_analog_stick_snap_end,
|
||||
atom_offset_try_analog_pad_try_unsupported = _atom_offset_try_analog_pad_try_unsupported,
|
||||
atom_offset_analog_pad_snap_end = _atom_offset_analog_pad_snap_end,
|
||||
};
|
||||
|
||||
// --- atom: pad_apply_input (60 words) ---
|
||||
|
||||
#define _atom_offset_dpad_left_exit_dpad_left 6
|
||||
#define _atom_offset_dpad_right_exit_dpad_right 6
|
||||
#define _atom_offset_dead_zone_low_check_dead_low_active 8
|
||||
#define _atom_offset_dead_zone_high_check_dead_high_active 15
|
||||
#define _atom_offset_dead_zone_skip_exit_stick 24
|
||||
#define _atom_offset_end_low_exit_stick 12
|
||||
|
||||
enum {
|
||||
atom_offset_dpad_left_exit_dpad_left = _atom_offset_dpad_left_exit_dpad_left,
|
||||
atom_offset_dpad_right_exit_dpad_right = _atom_offset_dpad_right_exit_dpad_right,
|
||||
atom_offset_dead_zone_low_check_dead_low_active = _atom_offset_dead_zone_low_check_dead_low_active,
|
||||
atom_offset_dead_zone_high_check_dead_high_active = _atom_offset_dead_zone_high_check_dead_high_active,
|
||||
atom_offset_dead_zone_skip_exit_stick = _atom_offset_dead_zone_skip_exit_stick,
|
||||
atom_offset_end_low_exit_stick = _atom_offset_end_low_exit_stick,
|
||||
};
|
||||
|
||||
#pragma endregion hello_joypad
|
||||
|
||||
@@ -0,0 +1,638 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "duffle/gen/macs.h"
|
||||
# include "duffle/gen/offsets.h"
|
||||
# include "duffle/dsl.atom.h"
|
||||
# include "duffle/lottes_tape.h"
|
||||
# include "duffle/mips.h"
|
||||
# include "duffle/gte.h"
|
||||
# include "duffle/gp.h"
|
||||
# include "duffle/pad.h"
|
||||
# include "duffle/word_count.metadata.h"
|
||||
# include "duffle/psyq.h"
|
||||
# include "duffle/math.atom.c"
|
||||
# include "duffle/mips.atom.c"
|
||||
# include "duffle/gte.atom.c"
|
||||
# include "duffle/gp.atom.c"
|
||||
# include "duffle/psyq.atom.c"
|
||||
# include "gen/offsets.h"
|
||||
# include "gen/macs.h"
|
||||
# include "hello_joypad.h"
|
||||
#endif
|
||||
|
||||
ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c);
|
||||
|
||||
#pragma region MACs (Mips Atom components)
|
||||
|
||||
FI_ Slice_MipsCode ac_put_disp_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
||||
MipsAtomComp_Proc_(ab, {
|
||||
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
|
||||
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
|
||||
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_put_draw_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
||||
MipsAtomComp_Proc_(ab, {
|
||||
/*
|
||||
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
|
||||
* References:
|
||||
* - libpsyx source: `toolchain/psyq-4_7/lib/libgpu.a` (binary, function `PutDrawEnv`)
|
||||
* - PSX-SPX doc: https://problemkaputt.de/psx-spx.htm#gputdrawingcommands
|
||||
* - PSYQ SDK: `setdrawenv` / `makelongdr_env` source
|
||||
* - NOCASH PSX spec: §"GP0(E1h) Draw Mode setting" through §"DR_ENV"
|
||||
*
|
||||
* The 16-word format is documented in the PSYQ SDK manual and on NOCASH's PSX-spec.txt. The libpsyx reference is at:
|
||||
* ./toolchain/psyq-4_7/lib/libgpu.a
|
||||
* (binary; the PutDrawEnv implementation builds the 16-word DR_ENV from the user's DRAWENV struct and emits it via GP0 GPU commands.)
|
||||
*
|
||||
* Word indices (libpsyx PutDrawEnv / SetDrawEnv order):
|
||||
* tag = (length << 24) | addr — 16-word packet (1 tag + 15 code)
|
||||
* code[0] = DrawMode (dfe=1, dtd=0, tpage=0) — must come first per libpsyx
|
||||
* code[1] = TextureWindow (tw=(0,0)) — bare-cmd word; GPU uses current state
|
||||
* code[2] = DrawArea top-left (clip.x=0, clip.y=240)
|
||||
* code[3] = DrawArea bottom-right (clip.x+w=320, clip.y+h=480)
|
||||
* code[4] = DrawOffset (ofs=(0,0)) — bare-cmd word
|
||||
* code[5] = Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit
|
||||
* code[6] = Initial-bg-color (isbg=1, r=7, g=7, b=7)
|
||||
* code[7] = DrawMode (isbg=1, tpage=0) — re-asserts DrawMode with isbg
|
||||
* code[8..10] = padding (NOP) — 3 words to fill the packet
|
||||
* code[11..12] = TextureWindow bottom-right — defaults to (0,0,0,0)
|
||||
* code[13..14] = padding (NOP) — completes the 16-word packet
|
||||
*/
|
||||
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port), /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */
|
||||
mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port), /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */
|
||||
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port), /* code[1] TextureWindow (tw=(0,0)) */
|
||||
mac_gcmd_push(enc_gp0_draw_area_tl_word(0, ScreenRes_Y), reg_transfer, reg_base, port), /* code[2] DrawArea top-left (clip.x=0, clip.y=ScreenRes_Y=240) */
|
||||
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port), /* code[3] DrawArea bottom-right (clip.x+w=320, clip.y+h=480) */
|
||||
|
||||
mac_gcmd_push(gp0_word_set_draw_offset(), reg_transfer, reg_base, port), /* code[4] DrawOffset (ofs=(0,0)) — bare-cmd word; the GPU uses the current state machine. */
|
||||
mac_gcmd_push(gp0_word_dr_env_mask(), reg_transfer, reg_base, port), /* code[5] Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit. */
|
||||
mac_gcmd_push(gp0_word_dr_env_bg_color_cmd(1, 7, 7, 7), reg_transfer, reg_base, port), /* code[6] Initial-bg-color + auto-clear (isbg=1, r=7, g=7, b=7). */
|
||||
mac_gcmd_push(gp0_word_dr_env_draw_mode(1), reg_transfer, reg_base, port), /* code[7] Re-assert DrawMode with isbg=1 (isbg-flag set; the 0xE1 cmd byte plus isbg only). */
|
||||
|
||||
/* code[8..10] Padding (NOP — GPU discards; the DR_ENV requires 16 words total). */
|
||||
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||
|
||||
/* code[11..12] TextureWindow bottom-right (tw.x+tw.w=0, tw.y+tw.h=0) — libpsyx emits twice. */
|
||||
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port),
|
||||
|
||||
/* code[13..14] Padding (NOP) — completes the 16-word packet. */
|
||||
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||
})
|
||||
|
||||
#pragma endregion MACs
|
||||
|
||||
#pragma region Baked Atoms
|
||||
|
||||
enum {
|
||||
R_ScreenX = R_T5 atom_reg atom_type(U2),
|
||||
R_ScreenY = R_T6 atom_reg atom_type(U2),
|
||||
R_ScreenBuf = R_T7 atom_reg, /* Caller-pinned: & smem.screen_buf */
|
||||
#define R_ScreenBuf_Code R_T7_Code
|
||||
};
|
||||
//screen_env_init. Mirrors the libpsyx's SetDefDispEnv + SetDefDrawEnv + the manual enable_auto_clear / initial_bg_color writes.
|
||||
internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
|
||||
, atom_reads(R_T0, R_ScreenX, R_ScreenY, R_ScreenBuf)
|
||||
, atom_writes(R_T0, R_ScreenX, R_ScreenY)
|
||||
) {
|
||||
/* display[0] = (0, 0, 320, 240); rest of struct zeroed. */
|
||||
add_ui(R_ScreenX, R_0, ScreenRes_X), add_ui(R_ScreenY, R_0, ScreenRes_Y),
|
||||
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area.width) + OA_(DoubleBuffer,display,0)),
|
||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,display_area) + OA_(DoubleBuffer,display,0)),
|
||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + OA_(DoubleBuffer,display,0)),
|
||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,0)),
|
||||
|
||||
/* display[1] = (0, 240, 320, 240); rest of struct zeroed. */
|
||||
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area) + OA_(DoubleBuffer,display,1)),
|
||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + OA_(DoubleBuffer,display,1)),
|
||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,1)),
|
||||
|
||||
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + OA_(DoubleBuffer,draw,0)), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
|
||||
mac_store_v2s2(R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + OA_(DoubleBuffer,draw,0)), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
|
||||
|
||||
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + OA_(DoubleBuffer,draw,1)),
|
||||
|
||||
/* draw[0].texture_window = (0, 0, 0, 0); two word-zeroes cover the full 8-byte tw field. */
|
||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + OA_(DoubleBuffer,draw,0)),
|
||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,0)),
|
||||
|
||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,drawing_offset[0].x) + OA_(DoubleBuffer,draw,1)),
|
||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + OA_(DoubleBuffer,draw,1)),
|
||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,1)),
|
||||
|
||||
/* draw[0].texture_page = 10 (gp0_tpage_default). C11 SetDefDrawEnv at C11_only.elf:0x8001273C writes the same 0x0A. . */
|
||||
add_ui(R_T0, R_0, gp0_tpage_default),
|
||||
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,0)),
|
||||
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,1)),
|
||||
|
||||
/* draw[0] control bytes: flag_dither=1, flag_draw_on_display=1 (the dfe bit per psx-spx; libpsyx sets it via `SetDefDrawEnv`'s conditional at C11_only.elf:0x80012728), enable_auto_clear=1. Each byte is named;
|
||||
* the previous `store_word(R_0, ..., +20)` overwrote all four with zero. */
|
||||
add_ui(R_T0, R_0, 1),
|
||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + OA_(DoubleBuffer,draw,0)),
|
||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + OA_(DoubleBuffer,draw,0)),
|
||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,0)),
|
||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + OA_(DoubleBuffer,draw,1)),
|
||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + OA_(DoubleBuffer,draw,1)),
|
||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,1)),
|
||||
|
||||
/* draw[0].initial_bg_color = (r=7, g=7, b=7). */
|
||||
add_ui(R_T0, R_0, 7),
|
||||
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,0)),
|
||||
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,1)),
|
||||
|
||||
mac_yield(),
|
||||
};
|
||||
|
||||
enum {
|
||||
R_IO_BaseAddr = R_T4 atom_reg, /* Caller-pinned: IO_BASE_ADDR = 0x1F800000 */
|
||||
#define R_IO_BaseAddr_Code R_T4_Code
|
||||
};
|
||||
internal MipsAtom_(gp_screen_init) atom_info(atom_phase(screen_init), atom_reads(R_IO_BaseAddr)) {
|
||||
store_word(R_0, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(00h) Reset */
|
||||
mac_gcmd_push(gp1_word_ResetCmdBuffer(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(01h) ClearFIFO */
|
||||
mac_gcmd_push(gp1_word_AcknowledgeIRQ(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(02h) AckIRQ */
|
||||
mac_gcmd_push(gp1_word_DisplayOn(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(03h) Display ON */
|
||||
mac_gcmd_push(gp1_word_dma_to_gpu(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(04h) DMADirection=2 (CPU→GPU). libpsyx's per-frame PutDrawEnv/DrawOTag use DMA2; without this the DMA queue never drains. */
|
||||
mac_gcmd_push(gp1_word_StartDisplayArea(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(05h) StartDisplayArea (X=0, Y=0) */
|
||||
|
||||
/* GP1: DisplayMode + Display Ranges */
|
||||
mac_gcmd_push(gp1_word_display_mode_320x240_15bit_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||
mac_gcmd_push(gp1_word_horizontal_range_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||
mac_gcmd_push(gp1_word_vertical_range_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||
|
||||
/* GTE: SetGeomOffset (OFX, OFY) — ScreenRes_CenterX, ScreenRes_CenterY. */
|
||||
load_upper_i(R_T5, ScreenRes_CenterX), gte_mv_to_ctrl_r(R_T5, gte_cr_OFX_Code),
|
||||
load_upper_i(R_T5, ScreenRes_CenterY), gte_mv_to_ctrl_r(R_T5, gte_cr_OFY_Code),
|
||||
|
||||
/* GTE: SetGeomScreen (H) — CR26 (per PSX-SPX / libpsyx), value is the raw projection-plane distance, NOT shifted. */
|
||||
add_ui(R_T5, R_0, ScreenZ), gte_mv_to_ctrl_r(R_T5, gte_cr_H_Code),
|
||||
|
||||
/* GP1: DisplayEnable — bit 0 = 0 (Display ON). */
|
||||
mac_gcmd_push(gp1_word_DisplayOn(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||
mac_yield(),
|
||||
};
|
||||
|
||||
enum {
|
||||
R_PrimCursor = R_T7 atom_reg atom_type(U4*), /* VRAM output cursor (primitive buffer) */
|
||||
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2*), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
|
||||
R_VertBase = R_T5 atom_reg atom_type(V3_S2*), /* Base address of the vertex array */
|
||||
R_OtBase = R_T6 atom_reg atom_type(U4*), /* Base address of the Ordering Table */
|
||||
#define R_PrimCursor_Code R_T7_Code
|
||||
#define R_FaceCursor_Code R_T4_Code
|
||||
#define R_VertBase_Code R_T5_Code
|
||||
#define R_OtBase_Code R_T6_Code
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_CubeTri) {
|
||||
U4 PrimCursor;
|
||||
V4_S2* FaceCursor;
|
||||
V3_S2* VertBase;
|
||||
U4* OtBase;
|
||||
};
|
||||
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)
|
||||
, atom_reads(R_TapePtr)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
|
||||
){
|
||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
|
||||
internal
|
||||
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
||||
atom_writes(R_PrimCursor, R_FaceCursor)
|
||||
){
|
||||
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
||||
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
||||
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
||||
|
||||
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
||||
nop2, gte_cmdw_rotate_translate_perspective_triple, // required cpu -> gte delay slot
|
||||
gte_cmdw_nclip,
|
||||
|
||||
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
|
||||
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
|
||||
/* BD-slot: write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
|
||||
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
|
||||
* harmless because the OT entry that points to this prim is created later, only on the body path. */
|
||||
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
||||
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
||||
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
|
||||
mac_gte_store_g4_p012(R_PrimCursor),
|
||||
gte_cmdw_rotate_translate_perspective_single,
|
||||
mac_gte_store_g4_p3(R_PrimCursor),
|
||||
|
||||
gte_cmdw_avg_sort_z4,
|
||||
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
|
||||
mac_insert_ot_tag_g4(R_OtBase, R_PrimCursor),
|
||||
mac_format_g4_color(R_PrimCursor,
|
||||
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
||||
/* c1 yellow */ 0xFF, 0xFF, 0x00,
|
||||
/* c2 cyan */ 0x00, 0xFF, 0xFF,
|
||||
/* c3 green */ 0x00, 0xFF, 0x00),
|
||||
// end: branch(bounds_chk)
|
||||
// end: branch(cull)
|
||||
|
||||
atom_label(cube_g4_face_exit)
|
||||
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
|
||||
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_FloorTri) {
|
||||
U4 PrimCursor;
|
||||
V3_S2* FaceCursor;
|
||||
V3_S2* VertBase;
|
||||
U4* OtBase;
|
||||
};
|
||||
internal
|
||||
MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(floor_f3)
|
||||
, atom_reads(R_TapePtr)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
|
||||
){
|
||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
// atom_dbg_skip
|
||||
internal
|
||||
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
||||
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor)
|
||||
) {
|
||||
mac_load_tri_indices(R_FaceCursor, R_T0, R_T1, R_T2),
|
||||
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
||||
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
|
||||
gte_cmdw_nclip,
|
||||
|
||||
/* Culling (Branch forward if Backface) */
|
||||
gte_mv_from_data_r(R_T0, C2_MAC0),
|
||||
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
|
||||
/* Format Primitive */
|
||||
mac_gte_store_f3(R_PrimCursor),
|
||||
|
||||
/* Calculate Depth */
|
||||
gte_avg_sort_z3,
|
||||
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||
/* Bounds Check OTZ < 2048 (Branch forward to skip insertion) */
|
||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
|
||||
mac_format_f3_color(R_PrimCursor, 0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
|
||||
mac_insert_ot_tag_f3(R_OtBase, R_PrimCursor), /* Insert into Ordering Table Linked List */
|
||||
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
|
||||
// Note(Ed): No bounds checking, should be checked before atom runs.
|
||||
// end: branch(bounds_chk)
|
||||
// end: branch(culling)
|
||||
|
||||
/* Advance Input Cursor & Yield (Both branch targets land here) */
|
||||
atom_label(floor_f3_face_exit)
|
||||
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
|
||||
internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitiveArena)
|
||||
, atom_reads( R_TapePtr, R_PrimCursor)
|
||||
, atom_writes(R_TapePtr)
|
||||
){
|
||||
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
|
||||
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
|
||||
/* Calculate byte offset and store directly back to RAM */
|
||||
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
|
||||
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
/* ----- pad_bios_snapshot -----
|
||||
* Per-frame snapshot of one BIOS pad buffer into PadState.
|
||||
* Decoder (branch ladder on raw[0] status + raw[1] id):
|
||||
* 1. raw[0] == 0xFF -> Disconnected (buttons=0, axes=0x80)
|
||||
* 2. raw[0]==0 && raw[1]==0 -> Pending (buttons=0, axes=0x80)
|
||||
* 3. raw[1] == 0x41 -> Digital (buttons normalized; axes=0x80)
|
||||
* 4. raw[1] == 0x53 -> AnalogStick (buttons normalized; axes from raw[4..7])
|
||||
* 5. raw[1] in 0x7x -> AnalogPad (buttons normalized; axes from raw[4..7])
|
||||
* 6. else -> Unsupported (buttons=0, axes=0x80)
|
||||
*
|
||||
* Buttons normalization: byte_swap16((~raw_buttons) & 0xFFFF).
|
||||
* raw_buttons = load_half_u(raw, 2) = raw[2] | (raw[3] << 8).
|
||||
* byte_swap16(x) = (x >> 8) | (x << 8); nor(x, R_0) = ~x. store_half truncates to 16 bits so the upper-16 mask is implicit in the store.
|
||||
*
|
||||
* Register use (atom-local; no wave-context touched):
|
||||
* R_T0 = raw base (kept throughout; axes loads read raw[4..7] from R_T0)
|
||||
* R_T1 = state base (kept throughout; all stores go through R_T1)
|
||||
* R_T2 = raw[0] status (alive across the disc/pending/id dispatch, then dead)
|
||||
* R_T3 = raw[1] id (alive across the id dispatch, then dead)
|
||||
* R_T4 = scratch (shifts, compares, immediate loads, store values)
|
||||
* R_T5 = scratch (parallel lui+ori for the 0x80808080 axes constant + byte-swap target)
|
||||
*/
|
||||
enum {
|
||||
R_PadRaw = R_T0 atom_reg atom_type(U1),
|
||||
R_PadState = R_T1 atom_reg,
|
||||
R_RawStatus = R_T2 atom_reg,
|
||||
R_RawId = R_T3 atom_reg,
|
||||
};
|
||||
typedef Struct_(Binds_PadBiosSnapshot) {
|
||||
PadBiosRaw* raw;
|
||||
PadState* state;
|
||||
};
|
||||
internal MipsAtom_(pad_bios_snapshot) atom_info(atom_bind(Binds_PadBiosSnapshot)
|
||||
, atom_reads( R_PadRaw, R_PadState, R_RawStatus, R_RawId, R_T4, R_T5, R_TapePtr)
|
||||
, atom_writes(R_PadRaw, R_PadState, R_RawStatus, R_RawId, R_T4, R_T5, R_TapePtr)
|
||||
) {
|
||||
/* === Bind consumption: T0 = raw, T1 = state, advance R_TapePtr by 8. */
|
||||
load_word(R_PadRaw, R_TapePtr, O_(Binds_PadBiosSnapshot,raw)),
|
||||
load_word(R_PadState, R_TapePtr, O_(Binds_PadBiosSnapshot,state)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_PadBiosSnapshot)),
|
||||
|
||||
/* === Read raw[0] (status) + raw[1] (id) */
|
||||
load_byte_u(R_RawStatus, R_PadRaw, 0),
|
||||
load_byte_u(R_RawId, R_PadRaw, 1),
|
||||
|
||||
atom_label(snap_root) /* === Case 1: Disconnected (status == 0xFF). */
|
||||
add_ui(R_T4, R_0, 0xFF), branch_ne(R_RawStatus, R_T4, atom_offset(snap_root, skip_disconnected)),
|
||||
/* BD-slot: pre-compute PadStatus_Disconnected. Branch reads R_T4=0xFF in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to pending/id_dispatch), R_T4 is overwritten by the next case body's add_ui — harmless. */
|
||||
|
||||
atom_label(disconnected) /* === Disconnected body. */
|
||||
/* R_T4 = PadStatus_Disconnected from snap_root BD-slot. */
|
||||
store_word(R_T4, R_PadState, O_(PadState,status)),
|
||||
store_half(R_0, R_PadState, O_(PadState,buttons)),
|
||||
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
|
||||
store_word( R_T4, R_PadState, O_(PadState,left_x)),
|
||||
store_byte( R_RawId, R_PadState, O_(PadState,id)),
|
||||
jump_rel(atom_offset(disconnected, snap_end)),
|
||||
/* BD-slot: load next atom's entry point (replaces the nop).
|
||||
* The unconditional branch always jumps to snap_end, where mac_yield_tail()
|
||||
* transfers control to R_AtomJmp without re-loading it. */
|
||||
mac_yield_load(),
|
||||
atom_label(skip_disconnected)
|
||||
|
||||
/* === Case 2: Pending (status == 0 && id == 0)
|
||||
* Combined check: if (status | id) != 0 then skip to id_dispatch.
|
||||
* Falls through to the Pending case only when both are zero. */
|
||||
or_u_self(R_RawStatus, R_RawId), branch_ne(R_RawStatus, R_0, atom_offset(case_2, id_dispatch)),
|
||||
/* BD-slot: pre-compute PadStatus_Pending. Branch reads R_RawStatus in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to id_dispatch), R_T4 is overwritten by the digital/analog body add_ui — harmless. */
|
||||
|
||||
atom_label(pending) /* === Pending body */
|
||||
/* R_T4 = PadStatus_Pending from case_2 BD-slot. */
|
||||
store_word(R_T4, R_PadState, O_(PadState,status)),
|
||||
store_half(R_0, R_PadState, O_(PadState,buttons)),
|
||||
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
|
||||
store_word( R_T4, R_PadState, O_(PadState,left_x)),
|
||||
store_byte( R_RawId, R_PadState, O_(PadState,id)),
|
||||
jump_rel(atom_offset(pending, snap_end)),
|
||||
mac_yield_load(),
|
||||
|
||||
atom_label(id_dispatch) /* === Case 3-6: ID dispatch */
|
||||
add_ui(R_T4, R_0, 0x41), branch_ne(R_RawId, R_T4, atom_offset(id_dispatch, try_analog_stick)),
|
||||
/* BD-slot: pre-compute PadStatus_Digital. Branch reads R_RawId in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to try_analog_stick), R_T4 is overwritten by the analog body add_ui. */
|
||||
|
||||
/* === Digital body (status, buttons normalize, axes=0x80, id, branch. */
|
||||
/* R_T4 = PadStatus_Digital from id_dispatch BD-slot. */
|
||||
store_word( R_T4, R_PadState, O_(PadState,status)),
|
||||
load_half_u(R_T4, R_PadRaw, 2 * S_(U1)),
|
||||
/* Fill R_T4's load-delay slot with the 0x80808080 axes constant into R_T5
|
||||
* (R_T5 is dead on this path; it's only consumed at the analog_pad range check). */
|
||||
load_upper_i(R_T5, 0x8080), or_i_self(R_T5, 0x8080),
|
||||
nor_u( R_T4, R_T4, R_0), /* raw_buttons is already in host bit order; no swap needed */
|
||||
store_half( R_T4, R_PadState, O_(PadState,buttons)),
|
||||
|
||||
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||
store_word( R_T5, R_PadState, O_(PadState,left_x)),
|
||||
add_ui( R_T4, R_0, 0x41),
|
||||
store_byte( R_T4, R_PadState, O_(PadState,id)),
|
||||
|
||||
jump_rel(atom_offset(id_dispatch, snap_end)),
|
||||
mac_yield_load(),
|
||||
|
||||
atom_label(try_analog_stick) /* === Case 4: AnalogStick (id == 0x53)*/
|
||||
add_ui(R_T4, R_0, 0x53), branch_ne(R_RawId, R_T4, atom_offset(try_analog_stick, try_analog_pad)),
|
||||
/* BD-slot: pre-compute PadStatus_AnalogStick. Branch reads R_RawId in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to try_analog_pad), R_T4 is overwritten by the analog_pad body add_ui. */
|
||||
|
||||
atom_label(analog_stick) /* === AnalogStick body
|
||||
* Axes are loaded as two halfwords: raw[6..7] → left_xy (sh at offset 8), raw[4..5] → right_xy (sh at offset 10).
|
||||
* R_T5 holds left_xy / id-value in turn (it's dead on this path — only consumed at the analog_pad range check). */
|
||||
/* R_T4 = PadStatus_AnalogStick from try_analog_stick BD-slot. */
|
||||
store_word( R_T4, R_PadState, O_(PadState,status)),
|
||||
load_half_u( R_T4, R_PadRaw, 2 * S_(U1)), /* R_T4 = raw_buttons */
|
||||
load_half_u( R_T5, R_PadRaw, 6 * S_(U1)), /* R_T5 = left_xy; fills R_T4's load-delay slot (doesn't read R_T4) */
|
||||
nor_u( R_T4, R_T4, R_0), /* R_T4 = ~raw_buttons */
|
||||
store_half( R_T4, R_PadState, O_(PadState,buttons)),
|
||||
load_half_u( R_T4, R_PadRaw, 4 * S_(U1)), /* R_T4 = right_xy; fills R_T5's load-delay slot */
|
||||
store_half( R_T5, R_PadState, O_(PadState,left_x)), /* R_T5 settled, store left_xy */
|
||||
store_half( R_T4, R_PadState, O_(PadState,right_x)),
|
||||
add_ui( R_T5, R_0, 0x53), /* R_T5 = id value (clobbers left_xy, already stored) */
|
||||
store_byte( R_T5, R_PadState, O_(PadState,id)),
|
||||
jump_rel(atom_offset(analog_stick, snap_end)),
|
||||
mac_yield_load(),
|
||||
|
||||
atom_label(try_analog_pad) /* === Case 5-6: AnalogPad (id & 0xF0 == 0x70) */
|
||||
and_i( R_T4, R_RawId, 0xF0),
|
||||
add_ui( R_T5, R_0, 0x70),
|
||||
branch_ne(R_T4, R_T5, atom_offset(try_analog_pad, try_unsupported)),
|
||||
/* BD-slot: pre-compute PadStatus_AnalogPad. Branch reads R_T4 in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to try_unsupported), R_T4 is overwritten by the unsupported body add_ui. */
|
||||
|
||||
atom_label(analog_pad) /* === AnalogPad body
|
||||
* Same shape as AnalogStick with AnalogPad status. R_T5 holds left_xy (it's dead on this path). */
|
||||
/* R_T4 = PadStatus_AnalogPad from try_analog_pad BD-slot. */
|
||||
store_word( R_T4, R_PadState, O_(PadState,status)),
|
||||
load_half_u(R_T4, R_PadRaw, 2 * S_(U1)), /* R_T4 = raw_buttons */
|
||||
load_half_u(R_T5, R_PadRaw, 6 * S_(U1)), /* R_T5 = left_xy; fills R_T4's load-delay slot */
|
||||
nor_u( R_T4, R_T4, R_0), /* R_T4 = ~raw_buttons */
|
||||
store_half( R_T4, R_PadState, O_(PadState,buttons)),
|
||||
load_half_u(R_T4, R_PadRaw, 4 * S_(U1)), /* R_T4 = right_xy; fills R_T5's load-delay slot */
|
||||
store_half( R_T5, R_PadState, O_(PadState,left_x)), /* R_T5 settled, store left_xy */
|
||||
store_half( R_T4, R_PadState, O_(PadState,right_x)),
|
||||
store_byte( R_RawId, R_PadState, O_(PadState,id)),
|
||||
|
||||
jump_rel(atom_offset(analog_pad, snap_end)),
|
||||
mac_yield_load(),
|
||||
|
||||
atom_label(try_unsupported) /* === Case 7: Unsupported — fall through from the AnalogPad range-check miss. */
|
||||
add_ui( R_T4, R_0, PadStatus_Unsupported),
|
||||
store_word(R_T4, R_PadState, O_(PadState,status)),
|
||||
store_half(R_0, R_PadState, O_(PadState,buttons)),
|
||||
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
|
||||
store_word( R_T4, R_PadState, O_(PadState,left_x)),
|
||||
add_ui( R_T4, R_0, 0xFF), /* 0xFF sentinel: "unknown id" */
|
||||
store_byte( R_T4, R_PadState, O_(PadState,id)),
|
||||
/* Fall through to snap_end. */
|
||||
|
||||
atom_label(no_jump_fallthrough)
|
||||
mac_yield_load(),
|
||||
|
||||
atom_label(snap_end)
|
||||
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the case-exit branch. */
|
||||
mac_yield_tail(),
|
||||
};
|
||||
|
||||
/* ----- pad_apply_input -----
|
||||
* Reads pad[0].buttons + pad[0].left_x;
|
||||
* Applies the input-semantics deltas to cube_rot.y + floor_rot.y:
|
||||
* - D-pad Left: cube_rot.y += 30, floor_rot.y += 5
|
||||
* - D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5
|
||||
* - Analog stick X (dead zone 0x70..0x90):
|
||||
* cube delta = (0x80 - left_x) >> 2 (range approx -32..+32)
|
||||
* floor delta = (0x80 - left_x) >> 5 (range approx -4..+4)
|
||||
* - D-pad + analog deltas add when used together.
|
||||
*
|
||||
* Convention:
|
||||
* pad_state = 0 means no buttons active.
|
||||
* The fail-safe zero-button value flows through unchanged, so a disconnected/fresh pad produces no rotation.
|
||||
* The branch_le_zero pattern below matches the existing pad_input_demo convention (atom body lines 248/257).
|
||||
*
|
||||
* Signed-delta trick:
|
||||
* load_byte_u zero-extends left_x to 32 bits; sub_u from 0x80 wraps to a SIGNED two's-complement value in the negative range;
|
||||
* shift_aright (sra) then correctly sign-extends the shift for both positive (left_x < 0x80) and negative (left_x > 0x80) cases.
|
||||
* Digital pads publish left_x = 0x80 → delta = 0 → no rotation, so the analog step is naturally a no-op for digital controllers.
|
||||
*/
|
||||
typedef Struct_(Binds_PadApplyInput) {
|
||||
PadState* state;
|
||||
V3_S2* cube_rot;
|
||||
V3_S2* floor_rot;
|
||||
};
|
||||
enum {
|
||||
R_PadStateT5 = R_T5 atom_reg,
|
||||
R_CubeRot = R_T1 atom_reg,
|
||||
R_FloorRot = R_T2 atom_reg,
|
||||
};
|
||||
internal MipsAtom_(pad_apply_input) atom_info(atom_bind(Binds_PadApplyInput)
|
||||
, atom_reads(R_T0, R_CubeRot, R_FloorRot, R_T3, R_T4, R_PadStateT5, R_TapePtr)
|
||||
, atom_writes( R_CubeRot, R_FloorRot)
|
||||
) {
|
||||
/* Pop Binds from tape (state, cube_rot, floor_rot) */
|
||||
load_word(R_PadStateT5, R_TapePtr, O_(Binds_PadApplyInput,state)),
|
||||
load_word(R_CubeRot, R_TapePtr, O_(Binds_PadApplyInput,cube_rot)),
|
||||
load_word(R_FloorRot, R_TapePtr, O_(Binds_PadApplyInput,floor_rot)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_PadApplyInput)),
|
||||
|
||||
/* Load pad[0].buttons into R_T0. */
|
||||
load_word(R_T0, R_PadStateT5, O_(PadState,buttons)), nop,
|
||||
// Note(Ed): Potential op with delay slot?
|
||||
|
||||
/* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */
|
||||
and_i(R_T3, R_T0, pad0_(Pad_Left)), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)),
|
||||
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
|
||||
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||
add_si( R_T4, R_T4, 30),
|
||||
add_si( R_T3, R_T3, 5),
|
||||
store_half(R_T4, R_CubeRot, O_(V3_S2,y)),
|
||||
store_half(R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||
atom_label(exit_dpad_left)
|
||||
|
||||
/* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */
|
||||
and_i(R_T3, R_T0, pad0_(Pad_Right)), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)),
|
||||
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
|
||||
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||
add_si( R_T4, R_T4, -30),
|
||||
add_si( R_T3, R_T3, -5),
|
||||
store_half(R_T4, R_CubeRot, O_(V3_S2,y)),
|
||||
store_half(R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||
atom_label(exit_dpad_right)
|
||||
|
||||
/* Analog left-stick X: dead zone 0x70..0x90.
|
||||
* Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */
|
||||
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left_x)),
|
||||
|
||||
/* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly).
|
||||
* set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */
|
||||
add_ui(R_T4, R_0, 0x70), set_lt_u(R_T4, R_T3, R_T4), branch_ne(R_T4, R_0, atom_offset(dead_zone_low_check, dead_low_active)),
|
||||
add_ui(R_T4, R_0, 0x80), /* BD-slot: pre-load 0x80 for dead_low_active */
|
||||
|
||||
atom_label(dead_check_upper)
|
||||
/* left_x >= 0x70 → check upper bound. */
|
||||
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left_x)), /* reload */
|
||||
add_ui( R_T4, R_0, 0x90),
|
||||
|
||||
/* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */
|
||||
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)),
|
||||
add_ui( R_T4, R_0, 0x80), /* BD-slot: pre-load 0x80 for dead_high_active */
|
||||
jump_rel(atom_offset(dead_zone_skip, exit_stick)),
|
||||
mac_yield_load(),
|
||||
|
||||
atom_label(dead_low_active)
|
||||
/* R_T3 = left_x (from line 632 lbu; not clobbered between dead_zone_low_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
||||
* The earlier `load_byte_u(R_T3, ...)` reload was redundant and introduced a load-use hazard on the next `sub_u`.
|
||||
* R_T4 = 0x80 from the BD-slot of `dead_zone_low_check`'s branch_ne. */
|
||||
sub_u( R_T3, R_T4, R_T3), /* R_T3 = 0x80 - left_x */
|
||||
/* delta = 0x80 - left_x (positive). */
|
||||
|
||||
/* R_T4 = cube_delta */
|
||||
shift_aright(R_T4, R_T3, 2),
|
||||
load_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||
nop,
|
||||
add_u( R_T0, R_T0, R_T4),
|
||||
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||
/* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap;
|
||||
* doesn't read R_T0; R_T4 settles by the subsequent add_u). */
|
||||
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||
shift_aright(R_T4, R_T3, 5),
|
||||
add_u( R_T0, R_T0, R_T4),
|
||||
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||
|
||||
jump_rel(atom_offset(end_low, exit_stick)),
|
||||
mac_yield_load(),
|
||||
|
||||
atom_label(dead_high_active)
|
||||
/* R_T3 = left_x (from line 641 lbu in dead_check_upper; not clobbered between dead_zone_high_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
||||
* The earlier `load_byte_u(R_T3, ...)` reload was redundant and introduced a load-use hazard on the next `sub_u`.
|
||||
* R_T4 = 0x80 from the BD-slot of `dead_zone_high_check`'s branch_ne. */
|
||||
sub_u( R_T3, R_T4, R_T3),
|
||||
/* delta = 0x80 - left_x (signed negative). */
|
||||
|
||||
shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */
|
||||
load_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||
nop,
|
||||
add_u( R_T0, R_T0, R_T4),
|
||||
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||
|
||||
/* R_T4 = floor_delta (signed) — moved into the load-delay slot of the floor load below. */
|
||||
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||
shift_aright(R_T4, R_T3, 5),
|
||||
add_u( R_T0, R_T0, R_T4),
|
||||
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||
|
||||
atom_label(no_jump_fallthrough)
|
||||
mac_yield_load(),
|
||||
|
||||
atom_label(exit_stick)
|
||||
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the dead-zone/exit branch. */
|
||||
mac_yield_tail(),
|
||||
};
|
||||
|
||||
#pragma endregion Baked Atoms
|
||||
@@ -0,0 +1,441 @@
|
||||
#pragma region Vendors
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <assert.h>
|
||||
// #include "libgpu.h"
|
||||
// #include "libetc.h"
|
||||
// #include "libgte.h"
|
||||
#pragma endregion Vendors
|
||||
|
||||
#pragma region Duffle Headers
|
||||
# include "duffle/gen/macs.h"
|
||||
# include "duffle/gen/offsets.h"
|
||||
|
||||
#include "duffle/word_count.metadata.h"
|
||||
|
||||
#include "duffle/dsl.h"
|
||||
#include "duffle/memory.h"
|
||||
#include "duffle/math.h"
|
||||
|
||||
#include "duffle/gcc_asm.h"
|
||||
#include "duffle/mips.h"
|
||||
#include "duffle/gp.h"
|
||||
#include "duffle/gte.h"
|
||||
#include "duffle/pad.h"
|
||||
|
||||
#include "duffle/dsl.atom.h"
|
||||
#include "duffle/lottes_tape.h"
|
||||
|
||||
#include "duffle/psyq.h"
|
||||
#pragma endregion Duffle Headers
|
||||
|
||||
#pragma region Duffle TUs
|
||||
#include "duffle/math.atom.c"
|
||||
#include "duffle/mips.atom.c"
|
||||
#include "duffle/gte.atom.c"
|
||||
#include "duffle/gp.atom.c"
|
||||
#include "duffle/psyq.atom.c"
|
||||
#pragma endregion Duffle TUs
|
||||
|
||||
#pragma region Joypade Headers
|
||||
# include "gen/macs.h"
|
||||
# include "gen/offsets.h"
|
||||
|
||||
#include "hello_joypad.h"
|
||||
#pragma region Joypad Headers
|
||||
|
||||
#pragma region Hello Joypad TUs
|
||||
#include "hello_joypad.atom.c"
|
||||
#pragma endregion Hello Joypad TUs
|
||||
|
||||
enum {
|
||||
Scratchpad_Len = 1024,
|
||||
MemTape_Len = 512,
|
||||
};
|
||||
typedef Struct_(SMemory) {
|
||||
PrimitiveArena primitives;
|
||||
A2_OrderingTable_Buffer ordering_tbl;
|
||||
DoubleBuffer screen_buf;
|
||||
S4 active_buf_id;
|
||||
|
||||
U4 MemTape[MemTape_Len];
|
||||
|
||||
M3_S2 tform_world;
|
||||
|
||||
Ent_Cube cube;
|
||||
Ent_Floor floor;
|
||||
|
||||
PadBiosRaw pad_raw[2];
|
||||
PadState pad[2];
|
||||
|
||||
U4_V scratchpad; // d-cache
|
||||
};
|
||||
global SMemory smem;
|
||||
extern SMemory smem;
|
||||
|
||||
I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
|
||||
gknown PrimitiveArena* pa = & smem.primitives;
|
||||
gknown B1* buf = (B1*) r_(smem.primitives.buf)[smem.active_buf_id];
|
||||
assert(pa->used + type_width < PrimitiveBuff_Len);
|
||||
B1* next = buf + pa->used;
|
||||
pa->used += type_width;
|
||||
return next;
|
||||
}
|
||||
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
|
||||
|
||||
/* Uses ONE 8-byte frame allocated via the compiler's standard prologue.
|
||||
* The 4 wasted-arg words for B(12h) InitPAD2 live at [SP+0..15] but are not explicitly allocated.
|
||||
* The compiler handles the MIPS O32 "wasted stack" convention for us by treating the B-call as a 4-arg call.
|
||||
*
|
||||
* The buffer pointers are passed as arguments so the compiler keeps them in callee-saved registers;
|
||||
* The B(12h) asm volatile block does NOT clobber those registers (it clobbers only the volatile GPRs + the B-table arg registers explicitly).
|
||||
* The C-level writes after the call re-load the pointers from their callee-saved homes.
|
||||
*
|
||||
* The clobber list for both B-calls names the full BIOS destroy set documented in kernelbios.md:167-174 (R1..R15, R24..R25, R31, HI/LO).
|
||||
* The kernel-ABI "volatile GPRs" subset is clb_system; the rest of the destroy set is enumerated explicitly here. */
|
||||
NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
|
||||
{
|
||||
/* Pin raw0 + raw1 to $a0 + $a1 via rgcc; the B(12h) call uses these directly.
|
||||
* The `(void)` casts mark them as unread after the call so the compiler doesn't need to move them back. */
|
||||
register PadBiosRaw* p0 rgcc(R_A0) = raw0;
|
||||
register PadBiosRaw* p1 rgcc(R_A1) = raw1;
|
||||
(void)p0; (void)p1;
|
||||
|
||||
// TODO(Ed): Properly annotate the raw values in the inline asm instructions.
|
||||
// Use enums.
|
||||
|
||||
/* B(12h) InitPAD2(raw0, 0x22, raw1, 0x22)
|
||||
* $a0 = raw0 (rgcc-bound; survives the sequence below)
|
||||
* $a1 = raw1 (preserved into $a2 before $a1 is overwritten)
|
||||
* $a2 = raw1 (moved from $a1; survives $a1's overwrite)
|
||||
* $a3 = 0x22 (immediate)
|
||||
* $t1 = 0x12 (function number)
|
||||
* $t2 = 0xB0 (BIOS B-table address) */
|
||||
asm volatile(
|
||||
asm_words(
|
||||
or_u( rarg_2, rarg_1, rdiscard), /* $a2 = $a1 = raw1 */
|
||||
add_ui( rarg_1, rdiscard, 0x22), /* $a1 = 0x22 */
|
||||
add_ui( rarg_3, rdiscard, 0x22), /* $a3 = 0x22 */
|
||||
add_ui( rtmp_1, rdiscard, 0x12), /* $t1 = 0x12 */
|
||||
add_ui( rtmp_2, rdiscard, 0xB0), /* $t2 = 0xB0 */
|
||||
call_reg(rtmp_2), /* jalr $t2, $ra */
|
||||
nop /* BD slot */
|
||||
)
|
||||
asm_rpins, r_use(p0), r_use(p1)
|
||||
asm_clobber:
|
||||
rlit(R_AT),
|
||||
rlit(R_V0), rlit(R_V1),
|
||||
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
||||
rlit(R_RA),
|
||||
clb_mem_drain
|
||||
);
|
||||
|
||||
/* The C-level writes re-load the pointers via the parameter names and write 0xFF to each
|
||||
* buffer's status byte to mark the initial-state hazard documented in kernelbios.md:1621-1624. */
|
||||
u1_v(raw0)[0] = 0xFF;
|
||||
u1_v(raw1)[0] = 0xFF;
|
||||
|
||||
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */
|
||||
asm volatile(
|
||||
asm_words(
|
||||
add_ui( rtmp_1, rdiscard, 0x13), /* $t1 = 0x13 */
|
||||
add_ui( rtmp_2, rdiscard, 0xB0), /* $t2 = 0xB0 (re-load) */
|
||||
call_reg(rtmp_2), /* jalr $t2, $ra */
|
||||
nop /* BD slot */
|
||||
)
|
||||
asm_clobber:
|
||||
rlit(R_AT),
|
||||
rlit(R_V0), rlit(R_V1),
|
||||
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
||||
rlit(R_RA),
|
||||
clb_mem_drain
|
||||
);
|
||||
}
|
||||
|
||||
GCC_OPTIMIZATION_DISABLE
|
||||
void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
{
|
||||
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
|
||||
|
||||
if (0) // Pad Input (dead — kept for the source-as-written record; references the deleted `pad_state` field)
|
||||
{
|
||||
(void)Pad_Left; (void)Pad_Right; /* suppress unused-token warnings */
|
||||
if (false) {
|
||||
smem.cube.rot.y += 30;
|
||||
smem.floor.rot.y += 5;
|
||||
}
|
||||
if (false) {
|
||||
smem.cube.rot.y -= 30;
|
||||
smem.floor.rot.y -= 5;
|
||||
}
|
||||
}
|
||||
if (1) // Pad Input (Tape version)
|
||||
{
|
||||
tb.used = 0; tb_scope_run(& tb) {
|
||||
/* BIOS-owned polling: per-frame snapshot of both ports. */
|
||||
tb_emit_(pad_bios_snapshot);
|
||||
tb_data_(raw, & smem.pad_raw[0]);
|
||||
tb_data_(state, & smem.pad[0]);
|
||||
tb_emit_(pad_bios_snapshot);
|
||||
tb_data_(raw, & smem.pad_raw[1]);
|
||||
tb_data_(state, & smem.pad[1]);
|
||||
/* Per-frame rotation apply: consume pad[0].buttons + pad[0].left_x */
|
||||
tb_emit_(pad_apply_input);
|
||||
tb_data_(state, & smem.pad[0]);
|
||||
tb_data_(cube_rot, & smem.cube.rot);
|
||||
tb_data_(floor_rot, & smem.floor.rot);
|
||||
}
|
||||
}
|
||||
|
||||
orderingtbl_clear_reverse(ordering_buf, OrderingTbl_Len);
|
||||
|
||||
// Update the position based on acceleration and velocity
|
||||
gknown V3_S4_R pos = & smem.cube.pos;
|
||||
gknown V3_S4_R vel = & smem.cube.vel;
|
||||
gknown V3_S4_R acc = & smem.cube.accel;
|
||||
add_v3s4(vel, acc[0]);
|
||||
add_v3s4_fp(pos, vel[0]);
|
||||
// vel->x += acc->x;
|
||||
// vel->y += acc->y;
|
||||
// vel->z += acc->z;
|
||||
// pos->x += vel->x;
|
||||
// pos->y += vel->y;
|
||||
// pos->z += vel->z;
|
||||
|
||||
if (pos->y + 150 > smem.floor.pos.y) vel->y *= -1;
|
||||
|
||||
// Prep
|
||||
S4 nclip = 0;
|
||||
S4 orderingtbl_z = 0;
|
||||
A2_S2 p; //???
|
||||
S4 flag; //????
|
||||
|
||||
|
||||
// Draw Cube
|
||||
if (0)
|
||||
{
|
||||
m3s2_rotation (& smem.cube.rot, & smem.tform_world);
|
||||
m3s2_translation(& smem.tform_world, & smem.cube.pos);
|
||||
m3s2_scale (& smem.tform_world, & smem.cube.scale);
|
||||
// gte_matrix_set_rotation (& smem.tform_world);
|
||||
gte_matrix_set_translation(& smem.tform_world);
|
||||
for (U4 face_id = 0; face_id < Cube_num_faces; face_id += 1)
|
||||
{
|
||||
Poly_G4* quad = prim_alloc(Poly_G4); set_poly_g4(quad);
|
||||
quad->c0 = rgb8(255, 0, 255);
|
||||
quad->c1 = rgb8(255, 255, 0);
|
||||
quad->c2 = rgb8( 0, 255, 255);
|
||||
quad->c3 = rgb8( 0, 255, 0);
|
||||
|
||||
V4_S2* face = & smem.cube.faces[face_id];
|
||||
V3_S2* p0 = & smem.cube.verts[face->x];
|
||||
V3_S2* p1 = & smem.cube.verts[face->y];
|
||||
V3_S2* p2 = & smem.cube.verts[face->z];
|
||||
V3_S2* p3 = & smem.cube.verts[face->w];
|
||||
|
||||
nclip = rtp_avg_nclip_a4_v3s2(
|
||||
p0, p1, p2, p3,
|
||||
& quad->p0, & quad->p1, & quad->p2, & quad->p3,
|
||||
& p, & orderingtbl_z, & flag
|
||||
);
|
||||
if (nclip <= 0) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if ((orderingtbl_z > 0) && (orderingtbl_z < OrderingTbl_Len)) {
|
||||
orderingtbl_add_primitive(ordering_buf[orderingtbl_z], quad);
|
||||
}
|
||||
}
|
||||
// smem.cube.rot.x += 6;
|
||||
// smem.cube.rot.y += 8;
|
||||
// smem.cube.rot.z += 12;
|
||||
smem.cube.rot.y += 30;
|
||||
}
|
||||
// Draw cube (tape method) - two triangles per face
|
||||
if (1)
|
||||
{
|
||||
m3s2_rotation (& smem.cube.rot, & smem.tform_world);
|
||||
m3s2_translation(& smem.tform_world, & smem.cube.pos);
|
||||
m3s2_scale (& smem.tform_world, & smem.cube.scale);
|
||||
gte_matrix_set_rotation (& smem.tform_world);
|
||||
gte_matrix_set_translation(& smem.tform_world);
|
||||
|
||||
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||
U4 prim_cursor = prim_base + pa->used;
|
||||
|
||||
tb.used = 0; tb_scope(& tb) {
|
||||
tb_emit(& tb, rbind_cube_g4_face);
|
||||
tb_data(& tb, prim_cursor);
|
||||
tb_data(& tb, u4_(smem.cube.faces));
|
||||
tb_data(& tb, u4_(smem.cube.verts));
|
||||
tb_data(& tb, u4_(ordering_buf));
|
||||
|
||||
for (U4 i = 0; i < Cube_num_faces; i++) {
|
||||
// Two triangles per quad face: (x,y,z) and (x,z,w)
|
||||
tb_emit(& tb, cube_g4_face);
|
||||
}
|
||||
|
||||
tb_emit(& tb, sync_primitive_arena);
|
||||
tb_data(& tb, u4_(& pa->used));
|
||||
tb_data(& tb, prim_base);
|
||||
}
|
||||
tape_run(tb_slice(tb));
|
||||
|
||||
// smem.cube.rot.y += 30;
|
||||
}
|
||||
// Draw Floor
|
||||
if (0)
|
||||
{
|
||||
m3s2_rotation (& smem.floor.rot, & smem.tform_world);
|
||||
m3s2_translation(& smem.tform_world, & smem.floor.pos);
|
||||
m3s2_scale (& smem.tform_world, & smem.floor.scale);
|
||||
gte_matrix_set_rotation (& smem.tform_world);
|
||||
gte_matrix_set_translation(& smem.tform_world);
|
||||
for (U4 face_id = 0; face_id < Floor_num_faces; face_id += 1)
|
||||
{
|
||||
Poly_F3* tri = prim_alloc(Poly_F3); set_poly_f3(tri);
|
||||
tri->color = rgb8(255, 255, 255);
|
||||
|
||||
V3_S2* face = & smem.floor.faces[face_id];
|
||||
register V3_S2* p0 rgcc(R_T4) = & smem.floor.verts[face->x];
|
||||
register V3_S2* p1 rgcc(R_T5) = & smem.floor.verts[face->y];
|
||||
register V3_S2* p2 rgcc(R_T6) = & smem.floor.verts[face->z];
|
||||
|
||||
gte_load_v0(p0, R_T4);
|
||||
/*
|
||||
asm volatile( ".word " "%0" ", %1" : :
|
||||
"i"(((op_lwc2 & OPCODE_MASK) << OPCODE_SHIFT) | ((R_T4 & REG_MASK) << RS_SHIFT) | ((gte_in_v0_xy & REG_MASK) << RT_SHIFT) | (0 & IMM_MASK)),
|
||||
"i"(((op_lwc2 & OPCODE_MASK) << OPCODE_SHIFT) | ((R_T4 & REG_MASK) << RS_SHIFT) | ((gte_in_v0_z & REG_MASK) << RT_SHIFT) | (GTE_Z_Offset & IMM_MASK)),
|
||||
"r"(p0) :
|
||||
"$2", "$8", "$9", "$31", "memory"
|
||||
);
|
||||
*/
|
||||
gte_load_v1(p1, R_T5);
|
||||
gte_load_v2(p2, R_T6);
|
||||
|
||||
gte_rtpt();
|
||||
gte_nclip();
|
||||
gte_stotz(& nclip);
|
||||
|
||||
// nclip = rtp_avg_nclip_a3_v3s2(p0, p1, p2
|
||||
// , & tri->p0, & tri->p1, & tri->p2
|
||||
// , & p, & orderingtbl_z, & flag
|
||||
// );
|
||||
// if (nclip <= 0) {
|
||||
// continue;
|
||||
// }
|
||||
|
||||
if (nclip > 0 ) {
|
||||
gte_stsxy3(& tri->p0, & tri->p1, & tri->p2);
|
||||
gte_avsz3();
|
||||
gte_stotz(& orderingtbl_z);
|
||||
|
||||
if ((orderingtbl_z > 0) && (orderingtbl_z < OrderingTbl_Len)) {
|
||||
orderingtbl_add_primitive(ordering_buf[orderingtbl_z], tri);
|
||||
}
|
||||
}
|
||||
}
|
||||
smem.floor.rot.y += 5;
|
||||
}
|
||||
// Draw floor tape method
|
||||
if (1)
|
||||
{
|
||||
m3s2_rotation (& smem.floor.rot, & smem.tform_world);
|
||||
m3s2_translation(& smem.tform_world, & smem.floor.pos);
|
||||
m3s2_scale (& smem.tform_world, & smem.floor.scale);
|
||||
|
||||
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||
U4 prim_cursor = prim_base + pa->used;
|
||||
|
||||
// TODO(Ed): We should do a bounds check beforehand to confirm pa can hold all tris?
|
||||
// The tape atoms in-flight should not need to care.
|
||||
|
||||
// Prepare the tape. (Push protocol to tape)
|
||||
tb.used = 0; tb_scope(& tb) {
|
||||
tb_emit(& tb, set_gte_world);
|
||||
tb_data(& tb, u4_(& smem.tform_world));
|
||||
|
||||
tb_emit(& tb, rbind_floor_f3_face);
|
||||
// TODO(Ed): Just use a single context struct ref
|
||||
tb_data(& tb, prim_cursor);
|
||||
tb_data(& tb, u4_(smem.floor.faces));
|
||||
tb_data(& tb, u4_(smem.floor.verts));
|
||||
tb_data(& tb, u4_(ordering_buf));
|
||||
for (U4 i = 0; i < Floor_num_faces; i++) {
|
||||
tb_emit(& tb, floor_f3_face);
|
||||
}
|
||||
// After floor_f3_face iterations complete, the primitive arena's used counter needs updating.
|
||||
tb_emit(& tb, sync_primitive_arena);
|
||||
tb_data(& tb, u4_(& pa->used));
|
||||
tb_data(& tb, prim_base);
|
||||
}
|
||||
tape_run(tb_slice(tb));// Fire off the tape.
|
||||
|
||||
// C-side state (pa->used) has already been updated by the tape!
|
||||
// smem.floor.rot.y += 5;
|
||||
}
|
||||
}
|
||||
GCC_OPTIMIZATION_ENABLE
|
||||
|
||||
void render(void) {
|
||||
}
|
||||
|
||||
void gp_display_frame(DoubleBuffer* screen_buf, S4* active_buf_id, U4* ordering_buf, PrimitiveArena* pa) {
|
||||
draw_sync(0);
|
||||
vsync(0);
|
||||
displayenv_put(& r_(screen_buf->display)[active_buf_id[0] ]);
|
||||
drawenv_put (& r_(screen_buf->draw) [active_buf_id[0] ]);
|
||||
{
|
||||
draw_orderingtbl(ordering_buf + OrderingTbl_Len - 1);
|
||||
pa->used = 0;
|
||||
}
|
||||
active_buf_id[0] = ! active_buf_id[0]; // Swap current buffer
|
||||
}
|
||||
|
||||
GCC_OPTIMIZATION_DISABLE
|
||||
int main(void)
|
||||
{
|
||||
smem = (SMemory){0};
|
||||
smem.scratchpad = C_(U4_V, 0x1F800000);
|
||||
// smem.primitives.used = 0;
|
||||
// smem.active_buf_id = 0;
|
||||
/*Persistent Entity Setup*/{
|
||||
ent_cube128_init(& smem.cube.verts, & smem.cube.faces); {
|
||||
Ent_Cube* cube = & smem.cube;
|
||||
cube->rot = v3s2(0, 0, 0);
|
||||
cube->scale = v3s4_fp_one();
|
||||
cube->accel = v3s4(0, 1, 0);
|
||||
cube->pos = v3s4(0, -400, 1800);
|
||||
}
|
||||
ent_floor_init(& smem.floor.verts, & smem.floor.faces); {
|
||||
Ent_Floor* floor = & smem.floor;
|
||||
floor->rot = v3s2(0, 0, 0);
|
||||
floor->pos = v3s4(0, 450, 1800);
|
||||
floor->scale = v3s4_fp_one();
|
||||
}
|
||||
}
|
||||
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape)); {
|
||||
reset_graph(0);
|
||||
/* Direct BIOS: poll both ports during VBlank. */
|
||||
pad_bios_init_start(& smem.pad_raw[0], & smem.pad_raw[1]);
|
||||
/* Pinned registers for the GPU init atom. */
|
||||
register U4* io_base_addr rgcc(R_IO_BaseAddr) = u4_r(IO_BASE_ADDR);
|
||||
register DoubleBuffer* screen_buf rgcc(R_ScreenBuf) = & smem.screen_buf;
|
||||
tb.used = 0; tb_scope_run(& tb) {
|
||||
tb_emit(& tb, screen_env_init);
|
||||
tb_emit(& tb, gp_screen_init);
|
||||
}
|
||||
}
|
||||
while (1) {
|
||||
gknown S4* active_buf_id = & smem.active_buf_id;
|
||||
gknown U4* ordering_buf = r_(smem.ordering_tbl)[active_buf_id[0]];
|
||||
gknown PrimitiveArena* pa = & smem.primitives;
|
||||
update(pa, ordering_buf);
|
||||
render();
|
||||
gp_display_frame(& smem.screen_buf, active_buf_id, ordering_buf, pa);
|
||||
};
|
||||
return 0;
|
||||
}
|
||||
GCC_OPTIMIZATION_ENABLE
|
||||
@@ -0,0 +1,102 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "duffle/dsl.h"
|
||||
# include "duffle/math.h"
|
||||
# include "duffle/gp.h"
|
||||
# include "duffle/pad.h"
|
||||
#endif
|
||||
|
||||
enum {
|
||||
// PrimitiveBuff_Len = 4096,
|
||||
// OrderingTbl_Len = 2048,
|
||||
PrimitiveBuff_Len = 131072,
|
||||
OrderingTbl_Len = 8192,
|
||||
};
|
||||
|
||||
enum {
|
||||
ScreenRes_X = 320,
|
||||
ScreenRes_Y = 240,
|
||||
ScreenZ = 320,
|
||||
ScreenRes_CenterX = (ScreenRes_X >> 1),
|
||||
ScreenRes_CenterY = (ScreenRes_Y >> 1),
|
||||
};
|
||||
|
||||
enum {
|
||||
fp_one = (1 << 12),
|
||||
};
|
||||
|
||||
#define v3s4_fp_one() v3s4(fp_one, fp_one, fp_one)
|
||||
|
||||
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
||||
typedef Array_(OrderingTable_Buffer, 2);
|
||||
|
||||
typedef B1 PrimitiveBuffer[PrimitiveBuff_Len];
|
||||
typedef Array_(PrimitiveBuffer, 2);
|
||||
typedef Struct_(PrimitiveArena) {
|
||||
A2_PrimitiveBuffer buf;
|
||||
U4 used;
|
||||
};
|
||||
|
||||
#define Cube_num_verts 8
|
||||
typedef Array_(V3_S2, Cube_num_verts);
|
||||
#define Cube_num_faces 6
|
||||
typedef Array_(V4_S2, Cube_num_faces);
|
||||
I_ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
||||
LP_ A8_V3_S2 baked_verts = (A8_V3_S2) {
|
||||
{ -128, -128, -128 },
|
||||
{ 128, -128, -128 },
|
||||
{ 128, -128, 128 },
|
||||
{ -128, -128, 128 },
|
||||
{ -128, 128, -128 },
|
||||
{ 128, 128, -128 },
|
||||
{ 128, 128, 128 },
|
||||
{ -128, 128, 128 }
|
||||
};
|
||||
LP_ A6_V4_S2 baked_faces = (A6_V4_S2) {
|
||||
{ 3, 2, 0, 1 },
|
||||
{ 0, 1, 4, 5 },
|
||||
{ 4, 5, 7, 6 },
|
||||
{ 1, 2, 5, 6 },
|
||||
{ 2, 3, 6, 7 },
|
||||
{ 3, 0, 7, 4 },
|
||||
};
|
||||
mem_copy(u4_(verts), u4_(& baked_verts), S_(A8_V3_S2) );
|
||||
mem_copy(u4_(faces), u4_(& baked_faces), S_(A6_V4_S2) );
|
||||
return;
|
||||
}
|
||||
typedef Struct_(Ent_Cube) {
|
||||
V3_S4 accel;
|
||||
V3_S4 vel;
|
||||
V3_S4 pos;
|
||||
V3_S4 scale;
|
||||
V3_S2 rot;
|
||||
A8_V3_S2 verts;
|
||||
A6_V4_S2 faces;
|
||||
};
|
||||
|
||||
#define Floor_num_verts 4
|
||||
typedef Array_(V3_S2, Floor_num_verts);
|
||||
#define Floor_num_faces 2
|
||||
typedef Array_(V3_S2, Floor_num_faces);
|
||||
I_ void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
|
||||
LP_ A4_V3_S2 baked_verts = (A4_V3_S2) {
|
||||
{ -900, 0, -900 },
|
||||
{ -900, 0, 900 },
|
||||
{ 900, 0, -900 },
|
||||
{ 900, 0, 900 },
|
||||
};
|
||||
LP_ A2_V3_S2 baked_faces = (A2_V3_S2) {
|
||||
{ 0, 1, 2 },
|
||||
{ 1, 3, 2 },
|
||||
};
|
||||
mem_copy(u4_(verts), u4_(& baked_verts), S_(A4_V3_S2));
|
||||
mem_copy(u4_(faces), u4_(& baked_faces), S_(A2_V3_S2));
|
||||
};
|
||||
typedef Struct_(Ent_Floor) {
|
||||
V3_S4 accel;
|
||||
V3_S4 pos;
|
||||
V3_S4 scale;
|
||||
V3_S2 rot;
|
||||
A4_V3_S2 verts;
|
||||
A2_V3_S2 faces;
|
||||
};
|
||||
@@ -0,0 +1,659 @@
|
||||
|
||||
#if 0 /* ac_pad_sio_write_pad_state — superseded by pad_bios_snapshot */
|
||||
|
||||
/* ============================================================
|
||||
* raw_sio_pad_poll_20260802 — superseded by bios_pad_buffer_snapshot_20260803.
|
||||
* The doomed raw-SIO production atoms (ac_pad_sio_write_pad_state,
|
||||
* pad_sio_init, pad_sio_step, pad_sio_diag_pin, pad_sio_diag_byte_exchange)
|
||||
* reference symbols that were removed from code/duffle/pad.h during
|
||||
* Phase 1. Each is wrapped in a narrow `#if 0` so the C compile skips
|
||||
* the body while the source-as-written text stays in place for the
|
||||
* Phase 5.1 deletion pass. The wrap is removed (and the bodies are
|
||||
* deleted) by Phase 5.1 of this track.
|
||||
* ============================================================ */
|
||||
|
||||
* Writes the per-port PadState in 5 instructions plus 4 store_word calls (status,
|
||||
* buttons, left_x/y/right_x/right_y packed, attempt). The provisional decode publishes
|
||||
* 0x0000FFFF buttons + centered axes on every path until response-byte decode lands.
|
||||
*
|
||||
* Args:
|
||||
* status_val - the PadSioStatus enum value to publish
|
||||
* state_ptr_reg - the PadState* base (R_PadState at the call site)
|
||||
* scratch_reg - scratch register for the value being stored (e.g., R_T0)
|
||||
*
|
||||
* Emits 9 instructions (status/buttons/axes/attempt stores plus the
|
||||
* two-instruction zero-extended buttons load).
|
||||
*/
|
||||
FI_ Slice_MipsCode ac_pad_sio_write_pad_state(MipsAtomBuilder_R ab, U4 status_val, U4 state_ptr_reg, U4 scratch_reg)
|
||||
MipsAtomComp_Proc_(ac_pad_sio_write_pad_state, ab, {
|
||||
add_ui(scratch_reg, R_0, status_val),
|
||||
store_word(scratch_reg, state_ptr_reg, O_(PadState,status)),
|
||||
/* FIX 2026-08-02: buttons = 0x0000FFFF = "no buttons pressed" in
|
||||
* libetc convention. Build it with LUI + ORI so addiu does not
|
||||
* sign-extend 0xFFFF to 0xFFFFFFFF. */
|
||||
load_upper_i(scratch_reg, 0x0000),
|
||||
or_i(scratch_reg, scratch_reg, 0xFFFF),
|
||||
store_word(scratch_reg, state_ptr_reg, O_(PadState,buttons)),
|
||||
add_ui(scratch_reg, R_0, 0x80808080),
|
||||
store_word(scratch_reg, state_ptr_reg, O_(PadState,left_x)),
|
||||
add_ui(scratch_reg, R_0, 0),
|
||||
store_word(scratch_reg, state_ptr_reg, O_(PadState,attempt))
|
||||
})
|
||||
#endif /* end ac_pad_sio_write_pad_state wrap */
|
||||
|
||||
/* ----- pad_sio_init -----
|
||||
* Boot-time SIO0 init. Caller pins R_T6 = sio_base_addr0.
|
||||
* Issues SIO CTRL=0x0040 (reset), MODE=0x000D, BAUD=0x0088.
|
||||
* (Phase 2 fills the body.)
|
||||
*/
|
||||
#if 0 /* pad_sio_init — superseded by pad_bios_init_start (Phase 1.3) */
|
||||
internal MipsAtom_(pad_sio_init) atom_info(atom_phase(pad_init)
|
||||
, atom_reads(R_T5, R_T6)
|
||||
, atom_writes(R_T5, R_T6)
|
||||
) {
|
||||
/* FIX 2026-08-02: explicitly load the KSEG1 base into R_T6 at the top of
|
||||
* the atom body. The rgcc(R_PadSioBase) binding in main() pins R_T6 = base
|
||||
* when main() runs, but $12 is caller-saved per the O32 ABI — when tape_run
|
||||
* is invoked, R_T6 is fair game. The atom body cannot rely on the value. */
|
||||
load_upper_i(R_T6, pad_IO_KSEG1_BASE >> 16), /* R_T6 high 16 = 0xBF80 */
|
||||
or_i(R_T6, R_T6, pad_IO_KSEG1_BASE & 0xFFFF), /* R_T6 = 0xBF800000 */
|
||||
|
||||
/* SIO CTRL = 0x0040 (reset) */
|
||||
add_ui(R_T5, R_0, pad_SIO_CTRL_RESET),
|
||||
store_half(R_T5, R_T6, pad_SIO_CTRL_OFFSET),
|
||||
/* SIO MODE = 0x000D (MUL1, 8-bit, no parity, idle-high) */
|
||||
add_ui(R_T5, R_0, pad_SIO_MODE_INIT),
|
||||
store_half(R_T5, R_T6, pad_SIO_MODE_OFFSET),
|
||||
/* SIO BAUD = 0x0088 (~250 kHz) */
|
||||
add_ui(R_T5, R_0, pad_SIO_BAUD_INIT),
|
||||
store_half(R_T5, R_T6, pad_SIO_BAUD_OFFSET),
|
||||
mac_yield(),
|
||||
};
|
||||
#endif /* end pad_sio_init wrap */
|
||||
|
||||
/* ----- pad_sio_step -----
|
||||
* Per-frame bounded raw-SIO transaction. Reads PadState pointers + SIO
|
||||
* base addresses from Binds_PadSioStep; writes per-port status +
|
||||
* buttons + axes into smem.pad[0..1].
|
||||
* Body shape (per spec §"Transaction model (per port, per pad_sio_step)"):
|
||||
* port 0: CTRL=CLEANUP → settle → CTRL=port-select → settle → exchange 5
|
||||
* bytes (addr + 0x42 0x00 0x00 0x00) → decode → write PadState[0]
|
||||
* → CTRL=CLEANUP.
|
||||
* port 1: swap scratch regs (sio_base_addr1 → R_PadSioBase, state1 →
|
||||
* R_PadState) → mirror port 0 sequence.
|
||||
*
|
||||
* Bounded-loop semantics: every countdown is wrapped in
|
||||
* add_ui_self(R_T1, -1) + branch_ne(R_T1, R_0, ...)
|
||||
* with a known maximum (pad_SIO_SETTLE_BEFORE_TX=1000, pad_SIO_SETTLE_AFTER_TX=2000,
|
||||
* pad_SIO_WAIT_BUDGET=4096). The static-analysis pass currently reports
|
||||
* has_loops = true; the follow-up metaprogram track that learns modeled-bounded
|
||||
* loops is out of scope here (per spec §"Risks").
|
||||
*
|
||||
* Scratch register strategy:
|
||||
* R_PadStatus = R_T4 — RESERVED for port-1 swap (holds state1)
|
||||
* R_PadCountdown = R_T5 — RESERVED for port-1 swap (holds sio_base_addr1)
|
||||
* R_T0 — byte-exchange value + STAT read (clobbered freely)
|
||||
* R_T1 — countdown budget (clobbered freely)
|
||||
* R_PadState = R_T7 — PadState* (preserved for PadState writes)
|
||||
* R_PadSioBase = R_T6 — SIO base (preserved through the port)
|
||||
*
|
||||
* Response decode (Task 3.1 teaching scope):
|
||||
* - status = PadSioStatus_Digital (hardcoded)
|
||||
* - buttons = 0xFFFF (no buttons pressed in the provisional libetc
|
||||
* convention; full response-byte decode is follow-up)
|
||||
* - axes = 0x80808080 (centered: left_x=0x80, left_y=0x80,
|
||||
* right_x=0x80, right_y=0x80)
|
||||
* - attempt = 0
|
||||
* - DualShock handshake (0x43 0x01 → 0x44 0x01 0x03 → 0x43 0x00) is
|
||||
* follow-up scope; the hardcoded digital decode is a placeholder.
|
||||
*
|
||||
* Both ports raise /CS (CTRL = pad_SIO_CTRL_CLEANUP) before exit. Both ports
|
||||
* treat response timeout as PadSioStatus_Disconnected per the spec §"Failure
|
||||
* handling" + the canonical per-port timeout semantics.
|
||||
*/
|
||||
#if 0 /* pad_sio_step — superseded by pad_bios_snapshot (Phase 2.1) */
|
||||
internal MipsAtom_(pad_sio_step) atom_info(atom_bind(Binds_PadSioStep)
|
||||
, atom_reads(R_TapePtr, R_PadSioBase, R_PadState, R_PadStatus, R_PadCountdown)
|
||||
, atom_writes(R_PadStatus, R_PadCountdown)
|
||||
) {
|
||||
/* FIX 2026-08-02: explicitly load KSEG1 base into R_PadSioBase (R_T6) at the
|
||||
* top. The rgcc() binding in main() does NOT survive the tape_run call
|
||||
* because R_T6 is caller-saved per the O32 ABI. The pad_sio_init atom
|
||||
* (also in the per-frame tape) reloads R_T6 separately. */
|
||||
load_upper_i(R_PadSioBase, pad_IO_KSEG1_BASE >> 16),
|
||||
or_i(R_PadSioBase, R_PadSioBase, pad_IO_KSEG1_BASE & 0xFFFF),
|
||||
|
||||
/* Pop Binds from tape (in Binds_PadSioStep declaration order) */
|
||||
load_word(R_PadState, R_TapePtr, O_(Binds_PadSioStep,state0)),
|
||||
load_word(R_PadStatus, R_TapePtr, O_(Binds_PadSioStep,state1)), /* reserved for port-1 swap */
|
||||
load_word(R_PadSioBase, R_TapePtr, O_(Binds_PadSioStep,sio_base_addr0)),
|
||||
load_word(R_PadCountdown, R_TapePtr, O_(Binds_PadSioStep,sio_base_addr1)), /* reserved for port-1 swap */
|
||||
add_ui_self(R_TapePtr, S_(Binds_PadSioStep)),
|
||||
|
||||
/* ============== PORT 0 TRANSACTION ============== */
|
||||
/* Use R_T0 (byte value / STAT read) + R_T1 (countdown) as scratch.
|
||||
* R_PadStatus (state1) + R_PadCountdown (sio_base_addr1) are preserved
|
||||
* through the port-0 body and swapped into R_PadSioBase + R_PadState
|
||||
* at atom_offset(port1_start, ...) below. */
|
||||
|
||||
/* 1. Cleanup: CTRL = 0x0010 (raise /CS, clear stale status) */
|
||||
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
|
||||
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||
/* Bounded by pad_SIO_SETTLE_BEFORE_TX = 1000 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_SETTLE_BEFORE_TX),
|
||||
atom_label(settle_pre_port0)
|
||||
nop, /* BD slot */
|
||||
add_ui_self(R_T1, -1),
|
||||
branch_ne(R_T1, R_0, atom_offset(settle_pre_port0, settle_pre_port0)),
|
||||
|
||||
/* 2. Port-select: CTRL = 0x0003 (TX enable + DTR /CS) for port 0 */
|
||||
add_ui(R_T0, R_0, pad_SIO_CTRL_TX_ENABLE),
|
||||
or_i(R_T0, R_T0, pad_SIO_CTRL_DTR_CS), /* set /CS line low */
|
||||
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||
/* Bounded by pad_SIO_SETTLE_AFTER_TX = 2000 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_SETTLE_AFTER_TX),
|
||||
atom_label(settle_post_port0)
|
||||
nop,
|
||||
add_ui_self(R_T1, -1),
|
||||
branch_ne(R_T1, R_0, atom_offset(settle_post_port0, settle_post_port0)),
|
||||
|
||||
/* 3. Address byte (0x01) — send + RX-ready wait + read response + RX-drain confirmation */
|
||||
add_ui(R_T0, R_0, pad_PROTO_ADDR),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ack0_port0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_ack0_port0, ack0_received_port0)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ack0_port0)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack0_port0, wait_ack0_port0)),
|
||||
/* RX timeout → mark disconnected; skip to port 1 */
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port0_from_ack0)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack0, port1_start)),
|
||||
|
||||
atom_label(ack0_received_port0)
|
||||
/* Read open-bus response byte 0 — discard per docs/psx-spx §controllersandmemorycards.md */
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* Confirm RX FIFO drained before sending byte 1. Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ackrel0_port0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_equal(R_T0, R_0, atom_offset(wait_ackrel0_port0, ack_released_port0)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ackrel0_port0)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel0_port0, wait_ackrel0_port0)),
|
||||
/* RX-drain timeout → disconnected; skip to port 1 */
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port0_from_ackrel0)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel0, port1_start)),
|
||||
|
||||
atom_label(ack_released_port0)
|
||||
|
||||
/* === Byte 1 (port 0): send 0x42 (cmd read) + RX-ready wait + read response + RX-drain confirmation === */
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T0, R_0, pad_PROTO_CMD_READ),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ack1_port0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_ack1_port0, ack1_received_port0)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ack1_port0)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack1_port0, wait_ack1_port0)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port0_from_ack1)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack1, port1_start)),
|
||||
|
||||
atom_label(ack1_received_port0)
|
||||
/* Read response ID byte — discarded for teaching scope (decode hardcoded). */
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* RX FIFO drain wait. Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ackrel1_port0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_equal(R_T0, R_0, atom_offset(wait_ackrel1_port0, ack_released1_port0)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ackrel1_port0)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel1_port0, wait_ackrel1_port0)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port0_from_ackrel1)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel1, port1_start)),
|
||||
|
||||
atom_label(ack_released1_port0)
|
||||
|
||||
/* === Byte 2 (port 0): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T0, R_0, 0x00),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ack2_port0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_ack2_port0, ack2_received_port0)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ack2_port0)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack2_port0, wait_ack2_port0)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port0_from_ack2)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack2, port1_start)),
|
||||
|
||||
atom_label(ack2_received_port0)
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ackrel2_port0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_equal(R_T0, R_0, atom_offset(wait_ackrel2_port0, ack_released2_port0)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ackrel2_port0)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel2_port0, wait_ackrel2_port0)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port0_from_ackrel2)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel2, port1_start)),
|
||||
|
||||
atom_label(ack_released2_port0)
|
||||
|
||||
/* === Byte 3 (port 0): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T0, R_0, 0x00),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ack3_port0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_ack3_port0, ack3_received_port0)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ack3_port0)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack3_port0, wait_ack3_port0)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port0_from_ack3)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack3, port1_start)),
|
||||
|
||||
atom_label(ack3_received_port0)
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ackrel3_port0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_equal(R_T0, R_0, atom_offset(wait_ackrel3_port0, ack_released3_port0)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ackrel3_port0)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel3_port0, wait_ackrel3_port0)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port0_from_ackrel3)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel3, port1_start)),
|
||||
|
||||
atom_label(ack_released3_port0)
|
||||
|
||||
/* === Byte 4 (FINAL, port 0): send 0x00 + RX-not-empty wait + read final byte === */
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T0, R_0, 0x00),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_rx4_port0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_rx4_port0, rx4_received_port0)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_rx4_port0)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_rx4_port0, wait_rx4_port0)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port0_from_rx4)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port0_from_rx4, port1_start)),
|
||||
|
||||
atom_label(rx4_received_port0)
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET), /* discard final byte */
|
||||
|
||||
/* === RESPONSE DECODE (hardcoded for teaching scope) ===
|
||||
* Per the plan §"Phase 3 task 3.1" + spec §"Architecture":
|
||||
* - Full decode (buttons/axes from response bytes) is follow-up scope.
|
||||
* - Teaching scope: hardcode digital poll response.
|
||||
* status = PadSioStatus_Digital
|
||||
* buttons = 0x0000FFFF (no buttons pressed — placeholder)
|
||||
* axes = 0x80808080 (left_x=0x80, left_y=0x80, right_x=0x80, right_y=0x80)
|
||||
* attempt = 0
|
||||
*/
|
||||
atom_label(decode_port0)
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Digital, R_PadState, R_T0),
|
||||
|
||||
/* /CS cleanup: raise /CS, clear stale status before exiting port 0. */
|
||||
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
|
||||
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||
|
||||
/* ============== PORT 1 SETUP ============== */
|
||||
/* Swap: R_PadCountdown holds sio_base_addr1; R_PadStatus holds state1. */
|
||||
atom_label(port1_start)
|
||||
add_u(R_PadSioBase, R_0, R_PadCountdown), /* sio_base_addr1 → R_PadSioBase */
|
||||
add_u(R_PadState, R_0, R_PadStatus), /* state1 → R_PadState */
|
||||
|
||||
/* ============== PORT 1 TRANSACTION (mirror of port 0) ============== */
|
||||
/* R_PadStatus + R_PadCountdown are no longer reserved (port 1 is the
|
||||
* last transaction); we still use R_T0/R_T1 as scratch to match port 0. */
|
||||
|
||||
/* 1. Cleanup: CTRL = 0x0010 (raise /CS, clear stale status) */
|
||||
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
|
||||
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||
/* Bounded by pad_SIO_SETTLE_BEFORE_TX = 1000 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_SETTLE_BEFORE_TX),
|
||||
atom_label(settle_pre_port1)
|
||||
nop,
|
||||
add_ui_self(R_T1, -1),
|
||||
branch_ne(R_T1, R_0, atom_offset(settle_pre_port1, settle_pre_port1)),
|
||||
|
||||
/* 2. Port-select: CTRL = 0x0003 | (1 << 13) (port 1 select) */
|
||||
add_ui(R_T0, R_0, pad_SIO_CTRL_TX_ENABLE),
|
||||
or_i(R_T0, R_T0, pad_SIO_CTRL_DTR_CS),
|
||||
or_i(R_T0, R_T0, 1 << 13), /* port 1 select bit (CTRL bit 13 = port select) */
|
||||
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||
/* Bounded by pad_SIO_SETTLE_AFTER_TX = 2000 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_SETTLE_AFTER_TX),
|
||||
atom_label(settle_post_port1)
|
||||
nop,
|
||||
add_ui_self(R_T1, -1),
|
||||
branch_ne(R_T1, R_0, atom_offset(settle_post_port1, settle_post_port1)),
|
||||
|
||||
/* 3. Address byte (0x01) — send + RX-ready wait + read response + RX-drain confirmation */
|
||||
add_ui(R_T0, R_0, pad_PROTO_ADDR),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ack0_port1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_ack0_port1, ack0_received_port1)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ack0_port1)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack0_port1, wait_ack0_port1)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port1_from_ack0)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack0, end_atom)),
|
||||
|
||||
atom_label(ack0_received_port1)
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ackrel0_port1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_equal(R_T0, R_0, atom_offset(wait_ackrel0_port1, ack_released_port1)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ackrel0_port1)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel0_port1, wait_ackrel0_port1)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port1_from_ackrel0)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel0, end_atom)),
|
||||
|
||||
atom_label(ack_released_port1)
|
||||
|
||||
/* === Byte 1 (port 1): send 0x42 (cmd read) + RX-ready wait + read response + RX-drain confirmation === */
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T0, R_0, pad_PROTO_CMD_READ),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ack1_port1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_ack1_port1, ack1_received_port1)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ack1_port1)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack1_port1, wait_ack1_port1)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port1_from_ack1)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack1, end_atom)),
|
||||
|
||||
atom_label(ack1_received_port1)
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ackrel1_port1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_equal(R_T0, R_0, atom_offset(wait_ackrel1_port1, ack_released1_port1)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ackrel1_port1)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel1_port1, wait_ackrel1_port1)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port1_from_ackrel1)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel1, end_atom)),
|
||||
|
||||
atom_label(ack_released1_port1)
|
||||
|
||||
/* === Byte 2 (port 1): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T0, R_0, 0x00),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ack2_port1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_ack2_port1, ack2_received_port1)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ack2_port1)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack2_port1, wait_ack2_port1)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port1_from_ack2)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack2, end_atom)),
|
||||
|
||||
atom_label(ack2_received_port1)
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ackrel2_port1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_equal(R_T0, R_0, atom_offset(wait_ackrel2_port1, ack_released2_port1)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ackrel2_port1)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel2_port1, wait_ackrel2_port1)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port1_from_ackrel2)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel2, end_atom)),
|
||||
|
||||
atom_label(ack_released2_port1)
|
||||
|
||||
/* === Byte 3 (port 1): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T0, R_0, 0x00),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ack3_port1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_ack3_port1, ack3_received_port1)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ack3_port1)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack3_port1, wait_ack3_port1)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port1_from_ack3)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack3, end_atom)),
|
||||
|
||||
atom_label(ack3_received_port1)
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ackrel3_port1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_equal(R_T0, R_0, atom_offset(wait_ackrel3_port1, ack_released3_port1)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ackrel3_port1)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel3_port1, wait_ackrel3_port1)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port1_from_ackrel3)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel3, end_atom)),
|
||||
|
||||
atom_label(ack_released3_port1)
|
||||
|
||||
/* === Byte 4 (FINAL, port 1): send 0x00 + RX-not-empty wait + read final byte === */
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T0, R_0, 0x00),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_rx4_port1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_rx4_port1, rx4_received_port1)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_rx4_port1)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_rx4_port1, wait_rx4_port1)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port1_from_rx4)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port1_from_rx4, end_atom)),
|
||||
|
||||
atom_label(rx4_received_port1)
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET), /* discard final byte */
|
||||
|
||||
/* === RESPONSE DECODE (port 1) === */
|
||||
atom_label(decode_port1)
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Digital, R_PadState, R_T0),
|
||||
|
||||
/* /CS cleanup: raise /CS, clear stale status before exiting port 1. */
|
||||
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
|
||||
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||
|
||||
atom_label(end_atom)
|
||||
mac_yield(),
|
||||
};
|
||||
#endif /* end pad_sio_step wrap */
|
||||
|
||||
/* ----- pad_sio_diag_pin -----
|
||||
* Per-frame diagnostic counter. The caller binds R_DiagPinScratch to
|
||||
* scratch_for_atom_diag_pin for temporary gdb verification.
|
||||
*/
|
||||
#if 0 /* pad_sio_diag_pin — superseded (raw-SIO phase removed) */
|
||||
internal MipsAtom_(pad_sio_diag_pin) atom_info(atom_phase(pad_init)
|
||||
, atom_reads(R_T0, R_T1, R_DiagPinScratch)
|
||||
, atom_writes(R_T0, R_T1, R_DiagPinScratch)
|
||||
) {
|
||||
/* FIX 2026-08-02: explicitly reload R_DiagPinScratch (R_T3 = $t3). Caller-saved
|
||||
* per O32 ABI; the rgcc binding in main() does not survive tape_run. */
|
||||
load_upper_i(R_DiagPinScratch, 0x8001),
|
||||
or_i(R_DiagPinScratch, R_DiagPinScratch, 0xC800),
|
||||
|
||||
/* High half = 0xD1A6; low half increments once per atom invocation. */
|
||||
load_word(R_T1, R_DiagPinScratch, 0),
|
||||
nop,
|
||||
add_ui(R_T1, R_T1, 1),
|
||||
and_i(R_T0, R_T1, 0xFFFF),
|
||||
load_upper_i(R_T1, 0xD1A6),
|
||||
or_i(R_T1, R_T1, 0),
|
||||
or_u(R_T1, R_T1, R_T0),
|
||||
store_word(R_T1, R_DiagPinScratch, 0),
|
||||
mac_yield(),
|
||||
};
|
||||
#endif /* end pad_sio_diag_pin wrap */
|
||||
|
||||
/* ----- pad_sio_diag_byte_exchange -----
|
||||
* Temporary two-byte wire probe: sends 0x01 and 0x42, then stores the
|
||||
* open-bus byte and response ID in scratch_for_atom_diag_pin.
|
||||
*/
|
||||
#if 0 /* pad_sio_diag_byte_exchange — superseded (raw-SIO phase removed) */
|
||||
internal MipsAtom_(pad_sio_diag_byte_exchange) atom_info(atom_phase(pad_init)
|
||||
, atom_reads(R_T0, R_T1, R_T2, R_PadSioBase, R_DiagPinScratch)
|
||||
, atom_writes(R_T0, R_T1, R_T2, R_PadSioBase, R_DiagPinScratch)
|
||||
) {
|
||||
/* FIX 2026-08-02: explicitly reload R_DiagPinScratch (R_T3 = $t3). Caller-saved
|
||||
* per O32 ABI; the rgcc binding in main() does not survive tape_run. */
|
||||
load_upper_i(R_DiagPinScratch, 0x8001),
|
||||
or_i(R_DiagPinScratch, R_DiagPinScratch, 0xC800),
|
||||
|
||||
/* FIX 2026-08-02: explicitly load KSEG1 base into R_PadSioBase (R_T6) at the
|
||||
* top. The rgcc() binding in main() does NOT survive the tape_run call
|
||||
* because R_T6 is caller-saved per the O32 ABI. */
|
||||
load_upper_i(R_PadSioBase, pad_IO_KSEG1_BASE >> 16),
|
||||
or_i(R_PadSioBase, R_PadSioBase, pad_IO_KSEG1_BASE & 0xFFFF),
|
||||
|
||||
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
|
||||
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||
add_ui(R_T0, R_0, pad_SIO_CTRL_TX_ENABLE),
|
||||
or_i(R_T0, R_T0, pad_SIO_CTRL_DTR_CS),
|
||||
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||
|
||||
add_ui(R_T0, R_0, pad_PROTO_ADDR),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(diag_wait_ack0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(diag_wait_ack0, diag_ack0_done)),
|
||||
add_ui_self(R_T1, -1),
|
||||
branch_ne(R_T1, R_0, atom_offset(diag_wait_ack0, diag_wait_ack0)),
|
||||
add_ui(R_T0, R_0, 0xDEADAC01),
|
||||
store_word(R_T0, R_DiagPinScratch, 0),
|
||||
branch_equal(R_0, R_0, atom_offset(diag_timeout_ack0, diag_timeout)),
|
||||
atom_label(diag_ack0_done)
|
||||
load_byte_u(R_T2, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T0, R_0, pad_PROTO_CMD_READ),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(diag_wait_ack1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(diag_wait_ack1, diag_ack1_done)),
|
||||
add_ui_self(R_T1, -1),
|
||||
branch_ne(R_T1, R_0, atom_offset(diag_wait_ack1, diag_wait_ack1)),
|
||||
add_ui(R_T0, R_0, 0xDEADAC02),
|
||||
store_word(R_T0, R_DiagPinScratch, 0),
|
||||
branch_equal(R_0, R_0, atom_offset(diag_timeout_ack1, diag_timeout)),
|
||||
atom_label(diag_ack1_done)
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
nop,
|
||||
shift_lleft(R_T0, R_T0, 8),
|
||||
or_u(R_T2, R_T2, R_T0),
|
||||
store_word(R_T2, R_DiagPinScratch, 0),
|
||||
atom_label(diag_success)
|
||||
branch_equal(R_0, R_0, atom_offset(diag_success, diag_done)),
|
||||
nop,
|
||||
atom_label(diag_timeout_ack0)
|
||||
add_ui(R_T0, R_0, 0xDEADAC01),
|
||||
store_word(R_T0, R_DiagPinScratch, 0),
|
||||
atom_label(diag_timeout_ack1)
|
||||
add_ui(R_T0, R_0, 0xDEADAC02),
|
||||
store_word(R_T0, R_DiagPinScratch, 0),
|
||||
atom_label(diag_timeout)
|
||||
add_ui(R_T0, R_0, 0xDEADACFF),
|
||||
store_word(R_T0, R_DiagPinScratch, 0),
|
||||
atom_label(diag_done)
|
||||
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
|
||||
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||
mac_yield(),
|
||||
};
|
||||
#endif /* end pad_sio_diag_byte_exchange wrap */
|
||||
+10625
File diff suppressed because one or more lines are too long
Binary file not shown.
|
After Width: | Height: | Size: 220 KiB |
@@ -6,18 +6,33 @@ A rest from the usual.
|
||||
|
||||
## Dependencies
|
||||
|
||||
I will be programming from a Windows 11 machine:
|
||||

|
||||
|
||||
```ps1
|
||||
# not really used yet for scripts (may never)
|
||||
scoop install lua
|
||||
```
|
||||
I will be programming from a Windows 11 machine (may eventually try this on the Steam Deck...):
|
||||
|
||||
[armips](https://github.com/Kingcom/armips)
|
||||
|
||||
* Supports doing bare-metal assembly for the ps1
|
||||
* `scoop install armips` or just clone and build..
|
||||
* Was used early in the course. Now I just use an macro asm dsl in C11.
|
||||
|
||||
[luajit-2.1](https://github.com/LuaJIT/LuaJIT.git)
|
||||
|
||||
```
|
||||
scoop install luajit
|
||||
```
|
||||
|
||||
* Used for lua scripts
|
||||
* Particularly, ps1_meta.lua which is a staged metaprogram pass for the custom C11 Assembly DSL used in this codebase.
|
||||
|
||||
[lpeg](https://github.com/roberto-ieru/LPeg.git)
|
||||
|
||||
* Lua is slow (even jitted) so this helps.
|
||||
|
||||
[lfs (LuaFileSystem)](https://github.com/lunarmodules/luafilesystem)
|
||||
|
||||
* Native directory enumeration + `mkdir` for the build scripts.
|
||||
* Used by `passes/word_count_eval.lua :: scan_dir` (native walk vs. `dir /b /s` subprocess,
|
||||
~2ms vs. ~56ms) and by `duffle.lua :: ensure_dir` + `to_absolute_path` (avoids
|
||||
`cmd.exe mkdir` + `cd` shell spawns, ~50ms each).
|
||||
|
||||
[pscx-redux](https://github.com/grumpycoders/pcsx-redux/): A collection of tools, research, hardware design, and libraries aiming at development and reverse engineering on the PlayStation 1.
|
||||
|
||||
@@ -57,3 +72,10 @@ scoop install lua
|
||||

|
||||

|
||||

|
||||

|
||||
|
||||
Win 11 machine:
|
||||
|
||||

|
||||
|
||||
Still haven't gotten around to trying this on linux...
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
-- gte_debug.lua — defensive version + prints error context.
|
||||
local ok, err = pcall(function()
|
||||
print("[debug] PCSX exists:", PCSX ~= nil)
|
||||
print("[debug] PCSX.WebServer exists:", PCSX and PCSX.WebServer ~= nil)
|
||||
print("[debug] PCSX.WebServer.Handlers exists:", PCSX and PCSX.WebServer and PCSX.WebServer.Handlers ~= nil)
|
||||
if not PCSX.WebServer then
|
||||
print("[debug] creating PCSX.WebServer...")
|
||||
PCSX.WebServer = {}
|
||||
end
|
||||
if not PCSX.WebServer.Handlers then
|
||||
print("[debug] creating PCSX.WebServer.Handlers...")
|
||||
PCSX.WebServer.Handlers = {}
|
||||
end
|
||||
print("[debug] type of Handlers:", type(PCSX.WebServer.Handlers))
|
||||
|
||||
PCSX.WebServer.Handlers.gte = function(req)
|
||||
local r = PCSX.getRegisters()
|
||||
local out = { "pc=0x" .. string.format("%x", r.pc) }
|
||||
for i = 0, 31 do
|
||||
out[#out + 1] = string.format("D[%d]=0x%08x C[%d]=0x%08x",
|
||||
i, r.CP2D.r[i], i, r.CP2C.r[i])
|
||||
end
|
||||
return table.concat(out, "\n")
|
||||
end
|
||||
print("[debug] handler registered")
|
||||
end)
|
||||
|
||||
if not ok then
|
||||
print("[debug] ERROR: " .. tostring(err))
|
||||
end
|
||||
@@ -0,0 +1,217 @@
|
||||
--- audit_lua_nesting.lua — Walk Lua source files and flag any block nesting deeper than 5 levels.
|
||||
---
|
||||
--- Usage:
|
||||
--- luajit scripts/audit_lua_nesting.lua scripts/duffle.lua scripts/ps1_meta.lua
|
||||
--- luajit scripts/audit_lua_nesting.lua scripts/passes/
|
||||
---
|
||||
--- Output: for each file, a list of {line, depth} entries where depth > 5.
|
||||
--- Returns exit code 1 if any violations found, 0 if clean.
|
||||
---
|
||||
--- **Implementation**: a hand-rolled depth tracker that counts:
|
||||
--- - `do`, `function`, `if`, `for`, `while`, `repeat` -> depth +1
|
||||
--- - `end`, `until` -> depth -1
|
||||
--- - `else`, `elseif` -> depth unchanged
|
||||
---
|
||||
--- **Caveats**: doesn't fully handle string/comment state (will miscount braces inside multi-line strings or block comments).
|
||||
--- For our metaprogram files (no embedded code generation), this is acceptable.
|
||||
|
||||
local M = {}
|
||||
|
||||
local BLOCK_OPEN = {
|
||||
["do"] = true,
|
||||
["function"] = true,
|
||||
["if"] = true,
|
||||
["for"] = true,
|
||||
["while"] = true,
|
||||
["repeat"] = true,
|
||||
}
|
||||
|
||||
local function is_block_close(token) return token == "end" or token == "until" end
|
||||
|
||||
-- (internal) Walk one source file and return a list of
|
||||
-- {line, depth, token} entries where depth > max_nesting.
|
||||
local function audit_file(path, max_nesting)
|
||||
local f = io.open(path, "r")
|
||||
if not f then error("Cannot open " .. path) end
|
||||
local content = f:read("*a")
|
||||
f:close()
|
||||
|
||||
local violations = {}
|
||||
local depth = 0
|
||||
local line = 1
|
||||
local pos = 1
|
||||
local src_len = #content
|
||||
local token_idx = 0
|
||||
|
||||
local function read_ident_at(start_pos)
|
||||
local ident_start = start_pos
|
||||
if ident_start > src_len then return nil end
|
||||
local first_ch = content:sub(ident_start, ident_start)
|
||||
if not (first_ch:match("[%a_]")) then return nil end
|
||||
local scan = start_pos + 1
|
||||
while scan <= src_len do
|
||||
local ch = content:sub(scan, scan)
|
||||
if not (ch:match("[%w_]")) then break end
|
||||
scan = scan + 1
|
||||
end
|
||||
return content:sub(ident_start, scan - 1), scan
|
||||
end
|
||||
|
||||
-- Skip past a string literal or comment starting at `start_pos`.
|
||||
-- Returns the position just past the construct, or nil if `start_pos`
|
||||
-- is not the start of a string/comment.
|
||||
local function skip_string_or_comment(start_pos)
|
||||
local ch = content:sub(start_pos, start_pos)
|
||||
if ch == '"' or ch == "'" then
|
||||
local scan = start_pos + 1
|
||||
while scan <= src_len do
|
||||
local c = content:sub(scan, scan)
|
||||
if c == "\\" then scan = scan + 2
|
||||
elseif c == ch then return scan + 1
|
||||
else scan = scan + 1
|
||||
end
|
||||
end
|
||||
return src_len + 1
|
||||
elseif ch == "-" and content:sub(start_pos + 1, start_pos + 1) == "-" then
|
||||
local scan = start_pos + 2
|
||||
if content:sub(scan, scan + 1) == "[[" and content:sub(scan + 2, scan + 3) == "[" then
|
||||
-- Long bracket comment [==[ ... ]==]
|
||||
scan = scan + 2
|
||||
local eq = ""
|
||||
while content:sub(scan, scan) == "=" do
|
||||
eq = eq .. "="
|
||||
scan = scan + 1
|
||||
end
|
||||
local close_marker = "]" .. eq .. "]"
|
||||
local close_pos = content:find(close_marker, scan, true)
|
||||
if close_pos then
|
||||
return close_pos + #close_marker
|
||||
else
|
||||
return src_len + 1
|
||||
end
|
||||
else
|
||||
while scan <= src_len and content:sub(scan, scan) ~= "\n" do scan = scan + 1 end
|
||||
return scan + 1
|
||||
end
|
||||
elseif ch == "[" and content:sub(start_pos + 1, start_pos + 1) == "[" then
|
||||
local scan = start_pos + 2
|
||||
local eq = ""
|
||||
while content:sub(scan, scan) == "=" do
|
||||
eq = eq .. "="
|
||||
scan = scan + 1
|
||||
end
|
||||
local close_marker = "]" .. eq .. "]"
|
||||
local close_pos = content:find(close_marker, scan, true)
|
||||
if close_pos then
|
||||
return close_pos + #close_marker
|
||||
else
|
||||
return src_len + 1
|
||||
end
|
||||
end
|
||||
return nil
|
||||
end
|
||||
|
||||
while pos <= src_len do
|
||||
local ch = content:sub(pos, pos)
|
||||
if ch == "\n" then line = line + 1 end
|
||||
|
||||
local skip_to = skip_string_or_comment(pos)
|
||||
if skip_to then
|
||||
for scan = pos, skip_to - 1 do
|
||||
if content:sub(scan, scan) == "\n" then line = line + 1 end
|
||||
end
|
||||
pos = skip_to
|
||||
elseif ch:match("[%a_]") then
|
||||
local tok, next_pos = read_ident_at(pos)
|
||||
token_idx = token_idx + 1
|
||||
if BLOCK_OPEN[tok] then
|
||||
depth = depth + 1
|
||||
if depth > max_nesting then
|
||||
violations[#violations + 1] = {
|
||||
line = line,
|
||||
depth = depth,
|
||||
token = tok,
|
||||
}
|
||||
end
|
||||
elseif is_block_close(tok) then
|
||||
depth = depth - 1
|
||||
end
|
||||
pos = next_pos
|
||||
else
|
||||
pos = pos + 1
|
||||
end
|
||||
end
|
||||
|
||||
return violations
|
||||
end
|
||||
|
||||
--- Audit one file. Returns nil if clean, else a list of violations.
|
||||
--- @param path string
|
||||
--- @param max_nesting integer -- default 5
|
||||
--- @return table|nil
|
||||
function M.audit(path, max_nesting)
|
||||
local violations = audit_file(path, max_nesting or 5)
|
||||
if #violations == 0 then return nil end
|
||||
return violations
|
||||
end
|
||||
|
||||
-- Module CLI.
|
||||
if arg and arg[1] then
|
||||
local max_nesting = 5
|
||||
local files = {}
|
||||
for arg_idx = 1, #arg do
|
||||
if arg[arg_idx] == "--max" and arg[arg_idx + 1] then
|
||||
max_nesting = tonumber(arg[arg_idx + 1]) or 5
|
||||
else
|
||||
files[#files + 1] = arg[arg_idx]
|
||||
end
|
||||
end
|
||||
|
||||
-- Accept either a directory or a file path. Directory args are
|
||||
-- expanded via lfs.dir (native, no subprocess).
|
||||
local lfs = require("lfs")
|
||||
local function is_dir(p)
|
||||
return lfs.attributes(p, "mode") == "directory"
|
||||
end
|
||||
local function list_lua(dir)
|
||||
local out = {}
|
||||
if not is_dir(dir) then return out end
|
||||
for entry in lfs.dir(dir) do
|
||||
if entry:match("%.lua$") then
|
||||
out[#out + 1] = dir .. "/" .. entry
|
||||
end
|
||||
end
|
||||
return out
|
||||
end
|
||||
|
||||
local to_check = {}
|
||||
for _, f in ipairs(files) do
|
||||
if is_dir(f) then
|
||||
for _, sub in ipairs(list_lua(f)) do to_check[#to_check + 1] = sub end
|
||||
else
|
||||
to_check[#to_check + 1] = f
|
||||
end
|
||||
end
|
||||
|
||||
local total_violations = 0
|
||||
for _, f in ipairs(to_check) do
|
||||
local v = M.audit(f, max_nesting)
|
||||
if v then
|
||||
io.write(string.format("\n%s\n", f))
|
||||
for _, x in ipairs(v) do
|
||||
io.write(string.format(" line %d: depth %d (after '%s')\n", x.line, x.depth, x.token))
|
||||
end
|
||||
total_violations = total_violations + #v
|
||||
end
|
||||
end
|
||||
|
||||
if total_violations == 0 then
|
||||
io.write("OK: no files exceed max nesting of " .. max_nesting .. "\n")
|
||||
os.exit(0)
|
||||
else
|
||||
io.write(string.format("\n%d nesting violation(s) found.\n", total_violations))
|
||||
os.exit(1)
|
||||
end
|
||||
end
|
||||
|
||||
return M
|
||||
+232
-88
@@ -81,19 +81,6 @@ $path_psyq = join-path $path_toolchain 'psyq-4_7'
|
||||
$path_psyq_iwyu = join-path $path_toolchain 'psyq_iwyu'
|
||||
$path_psyq_imyu_inc = join-path $path_psyq_iwyu 'include'
|
||||
|
||||
function Get-SourceFiles { param([Parameter(Mandatory=$true)] [string[]]$paths, [Parameter(Mandatory=$true)] [string[]]$extensions)
|
||||
$files = @()
|
||||
foreach ($p in $paths) {
|
||||
if (-not (test-path $p)) { continue }
|
||||
foreach ($ext in $extensions) {
|
||||
Get-ChildItem -Path $p -File -Recurse -Filter "*$ext" -ErrorAction SilentlyContinue | ForEach-Object {
|
||||
$files += $_.FullName
|
||||
}
|
||||
}
|
||||
}
|
||||
return ($files | Sort-Object -Unique)
|
||||
}
|
||||
|
||||
function assemble-unit { param(
|
||||
[string] $unit,
|
||||
[string] $link_module,
|
||||
@@ -153,7 +140,7 @@ function compile-unit { param(
|
||||
$f_arch_no_shared,
|
||||
$f_arch_no_stack_prot
|
||||
)
|
||||
# $compile_args += $f_std_c23
|
||||
$compile_args += $f_std_c11
|
||||
$compile_args += ($f_include + $path_psyq_imyu_inc)
|
||||
$compile_args += ($f_include + $path_nugget)
|
||||
|
||||
@@ -193,29 +180,15 @@ function link-modules { param([string[]]$link_modules, [string] $elf, [string[]
|
||||
$link_args += ($f_link_pass_through_prefix + $f_link_mapfile + $map)
|
||||
|
||||
$link_args += ($f_link_pass_through_prefix + $f_link_start_group)
|
||||
# 16 removed entries (c2, card, cd, comb, ds, gs, gun, hmd, math, mcrd, mcx, press, sio, snd, spu, tap)
|
||||
# had LOAD lines in the map but ZERO .o files pulled in — they were unused.
|
||||
# 5 kept libraries (api, c, etc, gpu, gte) are required by the C-side calls in hello_joypad.c (reset_graph, draw_sync, vsync, etc.).
|
||||
$libraries = @(
|
||||
"api",
|
||||
"c",
|
||||
"c2",
|
||||
"card",
|
||||
"cd",
|
||||
"comb",
|
||||
"ds",
|
||||
"etc",
|
||||
"gpu",
|
||||
"gs",
|
||||
"gte",
|
||||
"gun",
|
||||
"hmd",
|
||||
"math",
|
||||
"mcrd",
|
||||
"mcx",
|
||||
"pad",
|
||||
"press",
|
||||
"sio",
|
||||
"snd",
|
||||
"spu",
|
||||
"tap"
|
||||
"gte"
|
||||
)
|
||||
foreach ($lib in $libraries) {
|
||||
$link_args += ($f_link_lib + $lib)
|
||||
@@ -243,6 +216,117 @@ function make-binary { param([string]$elf, [string]$exe)
|
||||
if ($LASTEXITCODE -ne 0) { Write-Error "Objcopy failed. Aborting."; exit 1 }
|
||||
}
|
||||
|
||||
function ps1-meta { param(
|
||||
[string]$unity_root,
|
||||
[string[]]$sources,
|
||||
[Parameter(Mandatory=$true)][string]$metadata,
|
||||
[string]$out_root = (join-path $path_build 'gen'),
|
||||
[string[]]$passes = @('--pre-link'),
|
||||
[string[]]$extra_args = @()
|
||||
)
|
||||
# `--unity-root` and `--source` are mutually exclusive. Exactly one of `$unity_root` / `$sources` must be supplied; the other must be absent.
|
||||
if ($null -ne $unity_root -and $unity_root -ne '')
|
||||
{
|
||||
if ($null -ne $sources -and $sources.Count -gt 0) {
|
||||
write-error 'ps1-meta: -unity_root and -sources are mutually exclusive'
|
||||
exit 2
|
||||
}
|
||||
}
|
||||
elseif ($null -eq $sources -or $sources.Count -eq 0) {
|
||||
write-error 'ps1-meta: either -unity_root <file> or -sources <file...> is required'
|
||||
exit 2
|
||||
}
|
||||
|
||||
$script = join-path $path_scripts 'ps1_meta.lua'
|
||||
$input_summary = if ($null -ne $unity_root -and $unity_root -ne '') {
|
||||
"unity=$unity_root"
|
||||
}
|
||||
else {
|
||||
"$($sources.Count) source(s)"
|
||||
}
|
||||
write-host "ps1-meta $input_summary, passes=$($passes -join ',')" ` -ForegroundColor Magenta
|
||||
|
||||
$arg_list = @($passes) + @('--metadata', $metadata) + @('--out-root', $out_root) + @($extra_args)
|
||||
if ($null -ne $unity_root -and $unity_root -ne '') {
|
||||
$arg_list += @('--unity-root', $unity_root)
|
||||
}
|
||||
else {
|
||||
foreach ($s in $sources) { $arg_list += @('--source', $s) }
|
||||
}
|
||||
& luajit $script @arg_list
|
||||
if ($LASTEXITCODE -ne 0) {
|
||||
write-error "ps1-meta failed (exit $LASTEXITCODE). Aborting."
|
||||
exit $LASTEXITCODE
|
||||
}
|
||||
}
|
||||
|
||||
function inject-dwarf { param(
|
||||
[string]$elf,
|
||||
[string]$path_gen
|
||||
)
|
||||
$base_name = [System.IO.Path]::GetFileNameWithoutExtension($elf)
|
||||
$path_dwarf_line_bin = join-path $path_gen "$base_name.dwarf_line.bin"
|
||||
$path_dwarf_aranges_bin = join-path $path_gen "$base_name.dwarf_aranges.bin"
|
||||
$path_dwarf_rnglists_bin = join-path $path_gen "$base_name.dwarf_rnglists.bin"
|
||||
$path_dwarf_info_bin = join-path $path_gen "$base_name.dwarf_info.bin"
|
||||
$path_dwarf_abbrev_bin = join-path $path_gen "$base_name.dwarf_abbrev.bin"
|
||||
$path_dwarf_str_bin = join-path $path_gen "$base_name.dwarf_str.bin"
|
||||
$path_dwarf_loc_bin = join-path $path_gen "$base_name.dwarf_loc.bin"
|
||||
$path_dwarf_loclists_bin = join-path $path_gen "$base_name.dwarf_loclists.bin"
|
||||
$path_inject_elf = join-path $path_build "$base_name.dwarf-injected.elf"
|
||||
|
||||
if (-not (Test-Path $path_dwarf_line_bin)) { return }
|
||||
if (-not (Test-Path $path_dwarf_aranges_bin)) { return }
|
||||
if (-not (Test-Path $path_dwarf_rnglists_bin)) { return }
|
||||
|
||||
Write-Host "[build] DWARF-injecting $elf -> $path_inject_elf"
|
||||
Copy-Item -LiteralPath $elf -Destination $path_inject_elf -Force
|
||||
|
||||
# Objcopy call 1: 3x --update-section for the PC-mapping tables (line, aranges, rnglists).
|
||||
$objcopy_args_dwarf_pc = @(
|
||||
"--update-section=.debug_line=$path_dwarf_line_bin",
|
||||
"--update-section=.debug_aranges=$path_dwarf_aranges_bin",
|
||||
"--update-section=.debug_rnglists=$path_dwarf_rnglists_bin"
|
||||
)
|
||||
& $Objcopy @objcopy_args_dwarf_pc $path_inject_elf 2>&1 | Out-Null
|
||||
if ($LASTEXITCODE -ne 0) {
|
||||
Write-Warning "[build] objcopy dwarf-pc splice failed (exit $LASTEXITCODE); removing $path_inject_elf"
|
||||
Remove-Item -LiteralPath $path_inject_elf -ErrorAction SilentlyContinue
|
||||
return
|
||||
}
|
||||
|
||||
# Objcopy call 2: 3x --update-section + 2x --add-section for the debug-data tables (info, abbrev, str, loc, loclists).
|
||||
$objcopy_args_dwarf_info = @(
|
||||
"--update-section=.debug_info=$path_dwarf_info_bin",
|
||||
"--update-section=.debug_abbrev=$path_dwarf_abbrev_bin",
|
||||
"--update-section=.debug_str=$path_dwarf_str_bin",
|
||||
"--add-section=.debug_loc=$path_dwarf_loc_bin",
|
||||
"--add-section=.debug_loclists=$path_dwarf_loclists_bin"
|
||||
)
|
||||
& $Objcopy @objcopy_args_dwarf_info $path_inject_elf 2>&1 | Out-Null
|
||||
if ($LASTEXITCODE -ne 0) {
|
||||
Write-Warning "[build] objcopy dwarf-info splice failed (exit $LASTEXITCODE); removing $path_inject_elf"
|
||||
Remove-Item -LiteralPath $path_inject_elf -ErrorAction SilentlyContinue
|
||||
return
|
||||
}
|
||||
|
||||
# Baked atoms execute from RAM but are emitted as C data arrays, so their ELF sections lack SHF_EXECINSTR.
|
||||
# GDB discards line rows for non-code sections. Mark only the debug-copy sections executable.
|
||||
# The original ELF and PS-EXE remain byte/flag unchanged.
|
||||
& $Objcopy `
|
||||
--set-section-flags ".rodata=alloc,load,readonly,code,contents" `
|
||||
--set-section-flags ".data=alloc,load,data,code,contents" `
|
||||
$path_inject_elf 2>&1 | Out-Null
|
||||
if ($LASTEXITCODE -ne 0) {
|
||||
Write-Warning "[build] atom-section flag update failed (exit $LASTEXITCODE); removing $path_inject_elf"
|
||||
Remove-Item -LiteralPath $path_inject_elf -ErrorAction SilentlyContinue
|
||||
}
|
||||
else {
|
||||
Write-Host "[build] DWARF-injected ELF: $path_inject_elf"
|
||||
}
|
||||
}
|
||||
# inject-dwarf
|
||||
|
||||
function build-hello_psyqo {
|
||||
$includes += @()
|
||||
|
||||
@@ -317,57 +401,16 @@ function build-graphis_hello {
|
||||
}
|
||||
# build-graphis_hello
|
||||
|
||||
# ps1-meta orchestrator. Replaces generate-TapeAtomOffsets +
|
||||
# generate-TapeAtomAnnotations with a single invocation. Dispatches
|
||||
# the 6 passes (word-counts / components / annotation / offsets /
|
||||
# static-analysis / report) in dependency-topological order.
|
||||
|
||||
function any-stale {
|
||||
param([Parameter(Mandatory=$true)][string[]]$sources,
|
||||
[Parameter(Mandatory=$true)][string]$metadata,
|
||||
[Parameter(Mandatory=$true)][string]$out_root)
|
||||
if (-not (test-path $out_root)) { return $true }
|
||||
$out_mtime = (get-item $out_root).LastWriteTimeUtc
|
||||
$src_mtime = ($sources | ForEach-Object { (get-item $_).LastWriteTimeUtc } | Measure-Object -Maximum).Maximum
|
||||
$meta_mtime = (get-item $metadata).LastWriteTimeUtc
|
||||
return ($src_mtime -gt $out_mtime) -or ($meta_mtime -gt $out_mtime)
|
||||
}
|
||||
|
||||
function ps1-meta {
|
||||
param(
|
||||
[Parameter(Mandatory=$true)][string[]]$sources,
|
||||
[Parameter(Mandatory=$true)][string]$metadata,
|
||||
[string]$out_root = (join-path $path_build 'gen'),
|
||||
[string[]]$passes = @('--all')
|
||||
)
|
||||
$script = join-path $path_scripts 'ps1_meta.lua'
|
||||
write-host "ps1-meta $($sources.Count) source(s), passes=$($passes -join ',')" `
|
||||
-ForegroundColor Magenta
|
||||
$arg_list = @($passes) + @('--metadata', $metadata) + @('--out-root', $out_root)
|
||||
foreach ($s in $sources) { $arg_list += @('--source', $s) }
|
||||
& luajit $script @arg_list
|
||||
if ($LASTEXITCODE -ne 0) {
|
||||
write-error "ps1-meta failed (exit $LASTEXITCODE). Aborting."
|
||||
exit $LASTEXITCODE
|
||||
}
|
||||
}
|
||||
|
||||
function build-gte_hello {
|
||||
function build-hello_gte {
|
||||
$includes += @()
|
||||
|
||||
$path_module = join-path $path_code 'gte_hello'
|
||||
$path_module = join-path $path_code 'hello_gte'
|
||||
$path_duffle = join-path $path_code 'duffle'
|
||||
$path_atom_metadata = join-path $path_module 'tape_atom.metadata.h'
|
||||
$path_atom_metadata = join-path $path_duffle 'word_count.metadata.h'
|
||||
$path_build_gen = join-path $path_build 'gen'
|
||||
|
||||
$source_dirs = @($path_duffle, $path_module)
|
||||
$atom_sources = Get-SourceFiles -paths $source_dirs -extensions @('.h', '.c')
|
||||
|
||||
if (any-stale -sources $atom_sources -metadata $path_atom_metadata -out_root (join-path $path_build 'gen')) {
|
||||
ps1-meta -sources $atom_sources -metadata $path_atom_metadata -out_root (join-path $path_build 'gen')
|
||||
} else {
|
||||
write-host "ps1-meta all $($atom_sources.Count) source(s) up-to-date" `
|
||||
-ForegroundColor DarkGray
|
||||
}
|
||||
$src_c = join-path $path_module 'hello_gte.c'
|
||||
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen
|
||||
|
||||
$assemble_args = @()
|
||||
$assemble_args += $f_debug
|
||||
@@ -383,7 +426,6 @@ function build-gte_hello {
|
||||
|
||||
# assemble-unit $src_asm $module_asm $includes $assemble_args
|
||||
|
||||
$src_c = join-path $path_module 'hello_gte.c'
|
||||
$module_c = join-path $path_build 'hello_gte_c.o'
|
||||
|
||||
$compile_args = @()
|
||||
@@ -407,23 +449,125 @@ function build-gte_hello {
|
||||
)
|
||||
link-modules $link_modules $elf $link_args
|
||||
make-binary $elf $exe
|
||||
}
|
||||
build-gte_hello
|
||||
|
||||
# Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start).
|
||||
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
|
||||
|
||||
inject-dwarf $elf $path_build_gen
|
||||
}
|
||||
# build-hello_gte
|
||||
|
||||
function build-hello_joypad {
|
||||
$includes += @()
|
||||
|
||||
$path_module = join-path $path_code 'hello_joypad'
|
||||
$path_duffle = join-path $path_code 'duffle'
|
||||
$path_atom_metadata = join-path $path_duffle 'word_count.metadata.h'
|
||||
$path_build_gen = join-path $path_build 'gen'
|
||||
|
||||
$src_c = join-path $path_module 'hello_joypad.c'
|
||||
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen
|
||||
|
||||
$assemble_args = @()
|
||||
$assemble_args += $f_debug
|
||||
$assemble_args += $f_optimize_none
|
||||
$assemble_args += ($f_include + $path_code)
|
||||
|
||||
$src_asm_crt = join-path $path_nugget_common 'crt0/crt0.s'
|
||||
$module_asm_crt = join-path $path_build 'crt0.o'
|
||||
assemble-unit $src_asm_crt $module_asm_crt $includes $assemble_args
|
||||
|
||||
$module_c = join-path $path_build 'hello_joypad_c.o'
|
||||
|
||||
$compile_args = @()
|
||||
$compile_args += $f_debug
|
||||
$compile_args += $f_optimize_none
|
||||
# $compile_args += $f_optimize_intrinsics
|
||||
# $compile_args += $f_optimize_size
|
||||
# $compile_args += $f_optimize_debug
|
||||
$compile_args += ($f_include + $path_code)
|
||||
compile-unit $src_c $module_c $includes $compile_args
|
||||
|
||||
$elf = join-path $path_build 'hello_joypad.elf'
|
||||
$exe = join-path $path_build 'hello_joypad.ps-exe'
|
||||
|
||||
$link_args = @()
|
||||
$link_args += $f_debug
|
||||
# $link_args += $f_optimize_size
|
||||
$link_modules = @(
|
||||
$module_asm_crt,
|
||||
$module_c
|
||||
)
|
||||
link-modules $link_modules $elf $link_args
|
||||
make-binary $elf $exe
|
||||
|
||||
# Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start).
|
||||
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
|
||||
|
||||
inject-dwarf $elf $path_build_gen
|
||||
}
|
||||
# build-hello_joypad
|
||||
|
||||
function build-hello_camera {
|
||||
$includes += @()
|
||||
|
||||
$path_module = join-path $path_code 'hello_camera'
|
||||
$path_duffle = join-path $path_code 'duffle'
|
||||
$path_atom_metadata = join-path $path_duffle 'word_count.metadata.h'
|
||||
$path_build_gen = join-path $path_build 'gen'
|
||||
|
||||
$src_c = join-path $path_module 'hello_camera.c'
|
||||
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--pre-link')
|
||||
|
||||
$assemble_args = @()
|
||||
$assemble_args += $f_debug
|
||||
$assemble_args += $f_optimize_none
|
||||
$assemble_args += ($f_include + $path_code)
|
||||
|
||||
$src_asm_crt = join-path $path_nugget_common 'crt0/crt0.s'
|
||||
$module_asm_crt = join-path $path_build 'crt0.o'
|
||||
assemble-unit $src_asm_crt $module_asm_crt $includes $assemble_args
|
||||
|
||||
$module_c = join-path $path_build 'hello_camera_c.o'
|
||||
|
||||
$compile_args = @()
|
||||
$compile_args += $f_debug
|
||||
$compile_args += ($f_define + 'BUILD_DEBUG')
|
||||
$compile_args += $f_optimize_none
|
||||
# $compile_args += $f_optimize_intrinsics
|
||||
# $compile_args += $f_optimize_size
|
||||
# $compile_args += $f_optimize_debug
|
||||
$compile_args += ($f_include + $path_code)
|
||||
compile-unit $src_c $module_c $includes $compile_args
|
||||
|
||||
$elf = join-path $path_build 'hello_camera.elf'
|
||||
$exe = join-path $path_build 'hello_camera.ps-exe'
|
||||
|
||||
$link_args = @()
|
||||
$link_args += $f_debug
|
||||
# $link_args += $f_optimize_size
|
||||
$link_modules = @(
|
||||
$module_asm_crt,
|
||||
$module_c
|
||||
)
|
||||
link-modules $link_modules $elf $link_args
|
||||
make-binary $elf $exe
|
||||
|
||||
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
|
||||
|
||||
inject-dwarf $elf $path_build_gen
|
||||
}
|
||||
build-hello_camera
|
||||
|
||||
# NO idea if this works yet...
|
||||
function Send-ToEmulator { param(
|
||||
[string]$exePath
|
||||
)
|
||||
function Send-ToEmulator { param( [string]$exePath )
|
||||
$uri = "http://localhost:8080/api/v1/load-exec"
|
||||
|
||||
# Absolute path is safest for the emulator web server
|
||||
$absolutePath = [System.IO.Path]::GetFullPath($exePath)
|
||||
|
||||
# Create JSON payload pointing to your compiled .ps-exe
|
||||
$body = @{
|
||||
filename = $absolutePath
|
||||
} | ConvertTo-Json
|
||||
$body = @{ filename = $absolutePath } | ConvertTo-Json
|
||||
|
||||
Write-Host "Pushing hot-reload to PCSX-Redux..." -ForegroundColor Magenta
|
||||
try {
|
||||
|
||||
+77
-543
@@ -1,557 +1,91 @@
|
||||
-- duffle.lua
|
||||
--
|
||||
-- Shared primitives + domain tables for the tape-atom metaprograms.
|
||||
-- Both `tape_atom_annotation_pass.lua` and `tape_atom.offset_gen.meta.lua`
|
||||
-- `require("duffle")` for these.
|
||||
--
|
||||
-- 5.3-compatible Lua (no 5.4/5.5-only features):
|
||||
-- - no <close> / <toclose>
|
||||
-- - no continue keyword
|
||||
-- - no string.dump improvements
|
||||
-- - LuaJIT 5.1+extensions model is the primary target
|
||||
--
|
||||
-- No :match / :gmatch regex use anywhere; all delimiter-splitting is
|
||||
-- hand-rolled or via LPeg (the regex-free PEG library).
|
||||
--
|
||||
-- Phase 3: the hot lexer primitives are LPeg-backed where it pays off.
|
||||
-- The hand-rolled variants remain for callers that need a fallback.
|
||||
--- duffle.lua — facade over duffle_scan / duffle_isa / duffle_emit.
|
||||
|
||||
local M = {}
|
||||
--- @class DuffleExport
|
||||
--- bag: open module-export keys from duffle_scan / duffle_isa / duffle_emit
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Section 0: LPeg patterns (compiled once at module load)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
--
|
||||
-- LPeg is a PEG library (no regex). All patterns below are first-class
|
||||
-- pattern values; they're cheap to build and reuse.
|
||||
--
|
||||
-- Note: lpeg is required lazily because Lua 5.5 may not have it on its
|
||||
-- cpath at the same location as LuaJIT. We attempt the require and fall
|
||||
-- back to the hand-rolled implementations if it fails.
|
||||
local scan = require("duffle_scan") ---@type DuffleExport
|
||||
local isa = require("duffle_isa") ---@type DuffleExport
|
||||
local emit = require("duffle_emit") ---@type DuffleExport
|
||||
local M = {} ---@type DuffleExport
|
||||
|
||||
local lpeg_ok, lpeg = pcall(require, "lpeg")
|
||||
local lpeg_lib = nil
|
||||
local lpeg_alpha_pat, lpeg_alnum_pat, lpeg_ident_pat
|
||||
local lpeg_str_or_cmt_pat, lpeg_ws_and_cmt_pat
|
||||
local lpeg_scan_to_target_pat -- generic "anything but target or balanced group" matcher
|
||||
--- @alias Path string
|
||||
--- @alias LineNum integer
|
||||
--- @alias ByteOff integer
|
||||
--- @alias MacroName string
|
||||
--- @alias AtomName string
|
||||
--- @alias Severity string
|
||||
|
||||
if lpeg_ok then
|
||||
lpeg_lib = lpeg
|
||||
local P, S, R = lpeg.P, lpeg.S, lpeg.R
|
||||
--- @class SourceFile
|
||||
--- @field path Path
|
||||
--- @field text string
|
||||
--- @field dir string
|
||||
--- @field basename string
|
||||
--- @field scan SourceScan|nil
|
||||
|
||||
-- Character class patterns
|
||||
local alpha_pat = R("AZ", "az") + P("_")
|
||||
local digit_pat = R("09")
|
||||
lpeg_alnum_pat = alpha_pat + digit_pat
|
||||
--- @class CorpusView
|
||||
--- @field register_alias_registry table<string, AliasEntry>
|
||||
--- @field type_name_registry table<string, TypeNameEntry>
|
||||
--- @field atom_views table<AtomName, AtomViewEntry>
|
||||
--- @field atom_ctxs table<AtomName, AtomCtxEntry>
|
||||
--- @field atom_phases table<string, AtomPhaseGroup>
|
||||
--- @field binds_by_name table<string, BindsEntry>
|
||||
--- @field atoms_by_name table<AtomName, AtomEntry>
|
||||
--- @field atom_infos AtomInfoEntry[]
|
||||
--- @field components table<string, Component>
|
||||
--- @field component_atom_infos AtomInfoEntry[]|nil
|
||||
--- @field tape_chains table<string, TapeChain>|nil
|
||||
--- @field source_order SourceFile[]
|
||||
--- @field collisions CorpusCollision[]
|
||||
|
||||
-- Identifier: alpha followed by zero+ alnum. Capture as a string.
|
||||
lpeg_alpha_pat = alpha_pat
|
||||
lpeg_ident_pat = lpeg.C(alpha_pat * lpeg_alnum_pat^0)
|
||||
|
||||
-- String literal: "..." with backslash escapes.
|
||||
local lpeg_str_pat = P('"') * (P(1) - S('"\\') + P('\\') * P(1))^0 * P('"')
|
||||
-- Char literal: '...' with backslash escapes.
|
||||
local lpeg_chr_pat = P("'") * (P(1) - S("'\\") + P('\\') * P(1))^0 * P("'")
|
||||
-- Line comment: // ... to end-of-line.
|
||||
local lpeg_line_cmt_pat = P("//") * (P(1) - S("\n"))^0
|
||||
-- Block comment: /* ... */ (no nesting per C standard).
|
||||
local lpeg_block_cmt_pat = P("/*") * (P(1) - P("*/"))^0 * P("*/")
|
||||
-- String or comment (any of the four forms).
|
||||
lpeg_str_or_cmt_pat = lpeg_str_pat + lpeg_chr_pat + lpeg_line_cmt_pat + lpeg_block_cmt_pat
|
||||
|
||||
-- Whitespace + comment skipper: zero+ (whitespace run | string | comment).
|
||||
local ws_pat = S(" \t\n\r\v\f")
|
||||
lpeg_ws_and_cmt_pat = (ws_pat + lpeg_str_or_cmt_pat)^0
|
||||
|
||||
-- Generic "skip until target, but step over balanced groups" matcher.
|
||||
-- Used by scan_to_char for non-ident / non-bracket chars.
|
||||
-- We accept any single char except the target.
|
||||
-- The balanced-group stepping is handled by the caller (via read_balanced).
|
||||
lpeg_scan_to_target_pat = function(target)
|
||||
return (P(1) - P(target))^0
|
||||
end
|
||||
--- @param src DuffleExport
|
||||
--- @param label string
|
||||
--- @return nil
|
||||
local function merge(src, label)
|
||||
for k, v in pairs(src) do ---@type string, any
|
||||
if M[k] ~= nil and M[k] ~= v then
|
||||
error("duffle facade name collision on " .. tostring(k) .. " from " .. label, 0)
|
||||
end
|
||||
M[k] = v
|
||||
end
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Section 1: character classification (byte-based for hot loops)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
--
|
||||
-- Two APIs:
|
||||
-- is_space(c), is_alpha(c), etc. — accept a single-char STRING (legacy)
|
||||
-- is_space_byte(b), is_alpha_byte(b), etc. — accept a single-byte INTEGER
|
||||
--
|
||||
-- The byte-based versions are 5-10x faster in tight loops because they
|
||||
-- avoid the string allocation per s:sub(i, i) call.
|
||||
merge(scan, "duffle_scan")
|
||||
merge(isa, "duffle_isa")
|
||||
merge(emit, "duffle_emit")
|
||||
|
||||
-- Whitespace characters per C locale.
|
||||
function M.is_space_byte(b)
|
||||
return b == 32 or b == 9 or b == 10 or b == 13 or b == 11 or b == 12
|
||||
--- @param ctx PassCtx
|
||||
--- @return CorpusView
|
||||
function M.corpus_view(ctx)
|
||||
local corpus = ctx and ctx.shared and ctx.shared.corpus ---@type Corpus
|
||||
if not corpus then error("requires ctx.shared.corpus", 0) end
|
||||
return {
|
||||
register_alias_registry = corpus.register_alias_registry or {},
|
||||
type_name_registry = corpus.type_name_registry or {},
|
||||
atom_views = corpus.atom_views or {},
|
||||
atom_ctxs = corpus.atom_ctxs or {},
|
||||
atom_phases = corpus.atom_phases or {},
|
||||
binds_by_name = corpus.binds_by_name or {},
|
||||
atoms_by_name = corpus.atoms_by_name or {},
|
||||
atom_infos = corpus.atom_infos or {},
|
||||
components = corpus.components or {},
|
||||
component_atom_infos = corpus.component_atom_infos or {},
|
||||
tape_chains = corpus.tape_chains or {},
|
||||
source_order = corpus.source_order or {},
|
||||
collisions = corpus.collisions or {},
|
||||
}
|
||||
end
|
||||
|
||||
-- Letters (a-z, A-Z) and underscore.
|
||||
function M.is_alpha_byte(b)
|
||||
if not b then return false end
|
||||
if b >= 97 and b <= 122 then return true end -- 'a'..'z'
|
||||
if b >= 65 and b <= 90 then return true end -- 'A'..'Z'
|
||||
return b == 95 -- '_'
|
||||
--- @param rules CheckRule[]
|
||||
--- @param phase string
|
||||
--- @param item AtomEntry|SourceFile
|
||||
--- @param pipe_ctx PassScratch
|
||||
--- @param findings Finding[]
|
||||
--- @return nil
|
||||
function M.run_check_rules(rules, phase, item, pipe_ctx, findings)
|
||||
for _, rule in ipairs(rules) do ---@type integer, CheckRule
|
||||
local fn = rule[phase] ---@type (fun(item: AtomEntry|SourceFile, pipe_ctx: PassScratch, findings: Finding[]): nil)|nil
|
||||
if fn then fn(item, pipe_ctx, findings) end
|
||||
end
|
||||
end
|
||||
|
||||
-- Single digit.
|
||||
function M.is_digit_byte(b) return b and b >= 48 and b <= 57 end
|
||||
|
||||
-- Letter OR digit OR underscore.
|
||||
function M.is_alnum_byte(b) return M.is_alpha_byte(b) or M.is_digit_byte(b) end
|
||||
|
||||
-- String-based wrappers (kept for callers that already have a single-char
|
||||
-- string; the byte versions are what the hot loops should call).
|
||||
function M.is_space(c)
|
||||
if type(c) == "number" then return M.is_space_byte(c) end
|
||||
return c == " " or c == "\t" or c == "\n" or c == "\r" or c == "\v" or c == "\f"
|
||||
end
|
||||
function M.is_alpha(c)
|
||||
if type(c) == "number" then return M.is_alpha_byte(c) end
|
||||
if not c or #c == 0 then return false end
|
||||
if c >= "a" and c <= "z" then return true end
|
||||
if c >= "A" and c <= "Z" then return true end
|
||||
return c == "_"
|
||||
end
|
||||
function M.is_digit(c)
|
||||
if type(c) == "number" then return M.is_digit_byte(c) end
|
||||
return c and c >= "0" and c <= "9"
|
||||
end
|
||||
function M.is_alnum(c) return M.is_alpha(c) or M.is_digit(c) end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Section 2: string primitives
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Trim leading and trailing whitespace from a string.
|
||||
function M.trim(s)
|
||||
local a = 1
|
||||
while a <= #s and M.is_space_byte(s:byte(a)) do a = a + 1 end
|
||||
local b = #s
|
||||
while b >= a and M.is_space_byte(s:byte(b)) do b = b - 1 end
|
||||
return s:sub(a, b)
|
||||
end
|
||||
|
||||
-- Linear-search for a single-byte target in a string.
|
||||
-- (Phase 3 retained this for places where LPeg is overkill.)
|
||||
function M.find_byte(haystack, target, start)
|
||||
for i = start or 1, #haystack do
|
||||
if haystack:byte(i) == target then return i end
|
||||
end
|
||||
return nil
|
||||
end
|
||||
|
||||
-- Returns the directory portion of a path.
|
||||
function M.dirname(path)
|
||||
local last_sep = 0
|
||||
for i = 1, #path do
|
||||
if path:byte(i) == 47 or path:byte(i) == 92 then last_sep = i end -- '/' or '\\'
|
||||
end
|
||||
if last_sep == 0 then return "." end
|
||||
return path:sub(1, last_sep - 1)
|
||||
end
|
||||
|
||||
-- Returns the basename of a path, with the file extension stripped.
|
||||
function M.basename_no_ext(path)
|
||||
local last_sep = 0
|
||||
for i = 1, #path do
|
||||
if path:byte(i) == 47 or path:byte(i) == 92 then last_sep = i end
|
||||
end
|
||||
local a = last_sep + 1
|
||||
local last_dot = #path + 1
|
||||
for i = #path, a, -1 do
|
||||
if path:byte(i) == 46 then last_dot = i; break end -- '.'
|
||||
end
|
||||
return path:sub(a, last_dot - 1)
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Section 3: I/O primitives
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
function M.read_file(path)
|
||||
local f = io.open(path, "r")
|
||||
if not f then error("Cannot open " .. path) end
|
||||
local content = f:read("*a")
|
||||
f:close()
|
||||
return content
|
||||
end
|
||||
|
||||
function M.write_file(path, content)
|
||||
local f = io.open(path, "w")
|
||||
if not f then error("Cannot write " .. path) end
|
||||
f:write(content)
|
||||
f:close()
|
||||
end
|
||||
|
||||
function M.ensure_dir(path)
|
||||
local is_win = package.config:sub(1, 1) == "\\"
|
||||
os.execute(is_win and ('if not exist "' .. path .. '" mkdir "' .. path .. '"')
|
||||
or ('mkdir -p "' .. path .. '" 2>/dev/null'))
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Section 4: C-language scanner primitives
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- LPeg-backed skipper when LPeg is available, hand-rolled fallback otherwise.
|
||||
-- Returns position just past the construct, or `i` unchanged if no
|
||||
-- string/comment starts at position i.
|
||||
function M.skip_str_or_cmt(s, i)
|
||||
if lpeg_ok then
|
||||
local new_pos = lpeg.match(lpeg_str_or_cmt_pat, s, i)
|
||||
if new_pos then return new_pos end
|
||||
return i
|
||||
end
|
||||
-- Hand-rolled fallback (kept for builds where LPeg isn't available).
|
||||
local c = s:byte(i)
|
||||
if c == 34 or c == 39 then -- '"' or '\''
|
||||
i = i + 1
|
||||
while i <= #s do
|
||||
local b = s:byte(i)
|
||||
if b == 92 then i = i + 2 -- '\\'
|
||||
elseif b == c then return i + 1
|
||||
else i = i + 1 end
|
||||
end
|
||||
return #s + 1
|
||||
elseif c == 47 then -- '/'
|
||||
local nx = s:byte(i + 1)
|
||||
if nx == 47 then -- '//'
|
||||
while i <= #s and s:byte(i) ~= 10 do i = i + 1 end
|
||||
return i
|
||||
elseif nx == 42 then -- '/*'
|
||||
i = i + 2
|
||||
while i <= #s - 1 do
|
||||
if s:byte(i) == 42 and s:byte(i + 1) == 47 then -- '*/'
|
||||
return i + 2
|
||||
end
|
||||
i = i + 1
|
||||
end
|
||||
return #s + 1
|
||||
end
|
||||
end
|
||||
return i
|
||||
end
|
||||
|
||||
-- Skip whitespace AND C-style comments starting at position i.
|
||||
-- LPeg-backed when available; ~5-10x faster than the hand-rolled version.
|
||||
function M.skip_ws_and_cmt(s, i)
|
||||
if lpeg_ok then
|
||||
local new_pos = lpeg.match(lpeg_ws_and_cmt_pat, s, i)
|
||||
if new_pos then return new_pos end
|
||||
return i
|
||||
end
|
||||
-- Hand-rolled fallback.
|
||||
local len = #s
|
||||
while i <= len do
|
||||
if M.is_space_byte(s:byte(i)) then
|
||||
i = i + 1
|
||||
else
|
||||
local nx = M.skip_str_or_cmt(s, i)
|
||||
if nx > i then i = nx else break end
|
||||
end
|
||||
end
|
||||
return i
|
||||
end
|
||||
|
||||
-- Read a C-style identifier (alpha followed by zero+ alnum) starting at
|
||||
-- position i. Returns the identifier string + the position just past it,
|
||||
-- or nil + i if no identifier starts here.
|
||||
function M.read_ident(s, i)
|
||||
if lpeg_ok then
|
||||
local result = lpeg.match(lpeg_ident_pat, s, i)
|
||||
if result then return result, i + #result end
|
||||
return nil, i
|
||||
end
|
||||
-- Hand-rolled fallback.
|
||||
if not M.is_alpha_byte(s:byte(i)) then return nil, i end
|
||||
local a = i
|
||||
i = i + 1
|
||||
while i <= #s and M.is_alnum_byte(s:byte(i)) do i = i + 1 end
|
||||
return s:sub(a, i - 1), i
|
||||
end
|
||||
|
||||
-- Read a balanced-delimited group (parens, braces, or brackets) starting
|
||||
-- at position i. Returns the inner content (between the delimiters) +
|
||||
-- the position just past the closing delimiter, or nil + i if `s[i]`
|
||||
-- isn't `open_char`.
|
||||
--
|
||||
-- (Hand-rolled; the depth counting makes pure LPeg awkward here.)
|
||||
function M.read_balanced(s, open_char, close_char, i)
|
||||
local open_byte = open_char:byte()
|
||||
if s:byte(i) ~= open_byte then return nil, i end
|
||||
i = i + 1
|
||||
local len = #s
|
||||
local depth = 1
|
||||
local a = i
|
||||
while i <= len and depth > 0 do
|
||||
local c = s:byte(i)
|
||||
if c == open_byte then
|
||||
depth = depth + 1
|
||||
i = i + 1
|
||||
elseif c == close_char:byte() then
|
||||
depth = depth - 1
|
||||
if depth == 0 then break end
|
||||
i = i + 1
|
||||
else
|
||||
local nx = M.skip_str_or_cmt(s, i)
|
||||
if nx > i then i = nx else i = i + 1 end
|
||||
end
|
||||
end
|
||||
return s:sub(a, i - 1), i + 1
|
||||
end
|
||||
|
||||
-- Convenience specializations of read_balanced.
|
||||
M.read_parens = function(s, i) return M.read_balanced(s, "(", ")", i) end
|
||||
M.read_braces = function(s, i) return M.read_balanced(s, "{", "}", i) end
|
||||
M.read_brackets = function(s, i) return M.read_balanced(s, "[", "]", i) end
|
||||
|
||||
-- Scan forward from position `start` until we find a specific single byte
|
||||
-- `target`, transparently stepping over balanced parens/braces/brackets.
|
||||
-- Returns the position of `target`, or nil if not found.
|
||||
function M.scan_to_char(s, target, start)
|
||||
local target_byte = target:byte()
|
||||
local i = start
|
||||
while i <= #s do
|
||||
local c = s:byte(i)
|
||||
if c == target_byte then return i end
|
||||
if c == 40 then local _, a = M.read_balanced(s, "(", ")", i); i = a -- '('
|
||||
elseif c == 123 then local _, a = M.read_balanced(s, "{", "}", i); i = a -- '{'
|
||||
elseif c == 91 then local _, a = M.read_balanced(s, "[", "]", i); i = a -- '['
|
||||
else
|
||||
local nx = M.skip_str_or_cmt(s, i)
|
||||
if nx > i then i = nx else i = i + 1 end
|
||||
end
|
||||
end
|
||||
return nil
|
||||
end
|
||||
|
||||
-- Split a brace-body into top-level comma-separated tokens. Honors nested
|
||||
-- parens/braces/brackets and skips strings/comments.
|
||||
--
|
||||
-- FIX (2026-07-09): split at top-level NEWLINES and SEMICOLONS too, AND
|
||||
-- emit a token break after a top-level comment/string. Previous behavior
|
||||
-- glued the macro call after a comment into the same token, so
|
||||
-- `word_count_of_token` only saw the leading ident (often nil after
|
||||
-- stripping the comment), undercounting the body. See Phase 1 of the
|
||||
-- branch-offset regression investigation. Pure-comment / pure-string
|
||||
-- chunks (which now appear between real statements) are filtered out so
|
||||
-- they contribute 0 words instead of 1.
|
||||
function M.split_top_level_commas(body)
|
||||
local tokens = {}
|
||||
local i = 1
|
||||
local len = #body
|
||||
local token_start = 1
|
||||
|
||||
-- True iff `chunk` contains any non-whitespace, non-comment, non-string
|
||||
-- content (i.e., real token material). Walks through ws + comments
|
||||
-- individually so a chunk like " /* trailing */ shift_lleft(...)"
|
||||
-- is correctly classified as having real content (the macro call).
|
||||
local function has_real_content(chunk)
|
||||
local k = 1
|
||||
local klen = #chunk
|
||||
while k <= klen do
|
||||
if M.is_space(chunk:sub(k, k)) then
|
||||
k = k + 1
|
||||
else
|
||||
local nx = M.skip_str_or_cmt(chunk, k)
|
||||
if nx > k then
|
||||
k = nx -- skipped a comment or string
|
||||
else
|
||||
return true -- found real content
|
||||
end
|
||||
end
|
||||
end
|
||||
return false
|
||||
end
|
||||
|
||||
local function emit(end_pos)
|
||||
if end_pos >= token_start then
|
||||
local chunk = body:sub(token_start, end_pos)
|
||||
if M.trim(chunk) ~= "" then
|
||||
if has_real_content(chunk) then
|
||||
tokens[#tokens + 1] = chunk
|
||||
elseif #tokens > 0 then
|
||||
-- Pure comment/string chunk at top level (no
|
||||
-- preceding instruction content within this chunk).
|
||||
-- APPEND it to the LAST token so emit-context
|
||||
-- callers (components.lua build_component_lines)
|
||||
-- can convert `// trailing comment` to `/* */`
|
||||
-- and emit it with the macro body. For word
|
||||
-- counting, count_token_words only inspects the
|
||||
-- leading ident, so a trailing comment doesn't
|
||||
-- affect the count.
|
||||
--
|
||||
-- This is the second-half fix to commit 98e27c2:
|
||||
-- the first fix correctly broke top-level comments
|
||||
-- off from the NEXT statement (fixing macro-call
|
||||
-- word counts); this fix preserves them on the
|
||||
-- PREVIOUS statement (restoring the comments in
|
||||
-- the emitted .macs.h output).
|
||||
tokens[#tokens] = tokens[#tokens] .. chunk
|
||||
end
|
||||
end
|
||||
token_start = end_pos + 1
|
||||
end
|
||||
end
|
||||
|
||||
while i <= len do
|
||||
local c = body:byte(i)
|
||||
if c == 40 then -- '('
|
||||
local _, a = M.read_parens(body, i); i = a
|
||||
elseif c == 123 then -- '{'
|
||||
local _, a = M.read_braces(body, i); i = a
|
||||
elseif c == 91 then -- '['
|
||||
local _, a = M.read_brackets(body, i); i = a
|
||||
elseif c == 44 then -- ','
|
||||
emit(i - 1)
|
||||
i = i + 1
|
||||
token_start = i
|
||||
elseif c == 59 then -- ';'
|
||||
emit(i - 1)
|
||||
i = i + 1
|
||||
token_start = i
|
||||
elseif c == 10 then -- '\n'
|
||||
emit(i - 1)
|
||||
i = i + 1
|
||||
token_start = i
|
||||
else
|
||||
local nx = M.skip_str_or_cmt(body, i)
|
||||
if nx > i then
|
||||
-- Skipped a comment or string at top level: emit token break.
|
||||
i = nx
|
||||
emit(i - 1)
|
||||
else
|
||||
i = i + 1
|
||||
end
|
||||
end
|
||||
end
|
||||
emit(len)
|
||||
return tokens
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Section 5: load_word_counts
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
function M.load_word_counts(metadata_path)
|
||||
local counts = {}
|
||||
local content = M.read_file(metadata_path)
|
||||
local len = #content
|
||||
local i = 1
|
||||
local prefix = "WORD_COUNT("
|
||||
while i <= len do
|
||||
local nl = M.find_byte(content, 10, i) -- '\n'
|
||||
local line_end = nl or (len + 1)
|
||||
local line = content:sub(i, line_end - 1)
|
||||
local trimmed = M.trim(line)
|
||||
if trimmed:sub(1, #prefix) == prefix and trimmed:sub(-1) == ")" then
|
||||
local inner = trimmed:sub(#prefix + 1, #trimmed - 1)
|
||||
local comma = M.find_byte(inner, 44, 1) -- ','
|
||||
if comma then
|
||||
counts[M.trim(inner:sub(1, comma - 1))] =
|
||||
tonumber(M.trim(inner:sub(comma + 1)))
|
||||
end
|
||||
end
|
||||
i = line_end + 1
|
||||
end
|
||||
return counts
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Section 6: LineIndex (perf fix — replaces the per-call rescan line_of)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
function M.LineIndex(source)
|
||||
local positions = {}
|
||||
local n = 0
|
||||
for i = 1, #source do
|
||||
if source:byte(i) == 10 then -- '\n'
|
||||
n = n + 1
|
||||
positions[n] = i
|
||||
end
|
||||
end
|
||||
local function line_of(pos)
|
||||
local lo, hi = 1, n
|
||||
while lo <= hi do
|
||||
local mid = math.floor((lo + hi) / 2)
|
||||
if positions[mid] <= pos then
|
||||
lo = mid + 1
|
||||
else
|
||||
hi = mid - 1
|
||||
end
|
||||
end
|
||||
return hi + 1
|
||||
end
|
||||
return line_of
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Section 7: domain tables
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
M.WAVE_CONTEXT_REGS = {
|
||||
["R_PrimCursor"] = { alias = "R_T7", size = 4, role = "output cursor (prim arena)" },
|
||||
["R_FaceCursor"] = { alias = "R_T4", size = 4, role = "input cursor (face array)" },
|
||||
["R_VertBase"] = { alias = "R_T5", size = 4, role = "base pointer (vertex array)" },
|
||||
["R_OtBase"] = { alias = "R_T6", size = 4, role = "base pointer (ordering table)" },
|
||||
}
|
||||
|
||||
M.MACRO_EXPANSION = {
|
||||
["phase_init"] = "init",
|
||||
["phase_bind"] = "bind",
|
||||
["phase_setup"] = "setup",
|
||||
["phase_work"] = "work",
|
||||
["phase_commit"] = "commit",
|
||||
["phase_terminate"] = "terminate",
|
||||
["REGION_PRIM_ARENA"] = "prim_arena",
|
||||
["REGION_FACE_ARENA"] = "face_arena",
|
||||
["REGION_VERTEX_ARENA"] = "vertex_arena",
|
||||
["REGION_OT_ARENA"] = "ot_arena",
|
||||
["REGION_HEAP_3D"] = "heap_3d_models",
|
||||
["REGION_CDROM_STREAM"] = "cdrom_stream",
|
||||
["REGION_VRAM"] = "vram_heap",
|
||||
["CADENCE_FRAME"] = "frame",
|
||||
["CADENCE_ONCE"] = "once",
|
||||
["CADENCE_ONDEMAND"] = "ondemand",
|
||||
}
|
||||
|
||||
M.KNOWN_PHASES = {
|
||||
["init"] = true, ["bind"] = true, ["setup"] = true,
|
||||
["work"] = true, ["commit"] = true, ["terminate"] = true,
|
||||
}
|
||||
M.KNOWN_REGIONS = {
|
||||
["prim_arena"] = true, ["face_arena"] = true,
|
||||
["vertex_arena"] = true, ["ot_arena"] = true,
|
||||
["heap_3d_models"] = true, ["cdrom_stream"] = true,
|
||||
["vram_heap"] = true,
|
||||
}
|
||||
M.KNOWN_CADENCES = {
|
||||
["frame"] = true, ["once"] = true, ["ondemand"] = true,
|
||||
}
|
||||
|
||||
M.TAPE_ATOM_MACROS = {
|
||||
["atom_annot"] = { kind = "work", binds = false },
|
||||
["atom_bind"] = { kind = "bind", binds = true },
|
||||
["atom_setup"] = { kind = "setup", binds = false },
|
||||
["atom_commit"] = { kind = "commit", binds = false },
|
||||
["atom_init"] = { kind = "init", binds = false },
|
||||
["atom_terminate"] = { kind = "terminate", binds = false },
|
||||
}
|
||||
|
||||
M.ATOM_PRAGMA_KINDS = {
|
||||
["resource"] = { kind = "string" },
|
||||
["region"] = { kind = "ident", allowed = M.KNOWN_REGIONS },
|
||||
["group"] = { kind = "ident" },
|
||||
["cadence"] = { kind = "ident", allowed = M.KNOWN_CADENCES },
|
||||
["async"] = { kind = "ident", allowed = { ["true"] = true, ["false"] = true } },
|
||||
}
|
||||
|
||||
-- Expose the lpeg_ok flag so callers can detect the LPeg-back path.
|
||||
-- True when LPeg was successfully required and the patterns above were
|
||||
-- compiled at module load time. False when running in fallback mode.
|
||||
M.lpeg_ok = lpeg_ok
|
||||
|
||||
return M
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,866 @@
|
||||
--- duffle_isa.lua — encoder / GTE / hardware tables.
|
||||
|
||||
--- @class InstructionImm
|
||||
--- @field arg integer
|
||||
--- @field signed boolean|nil
|
||||
--- @field width integer
|
||||
|
||||
--- @class InstructionValue
|
||||
--- @field dest integer
|
||||
--- @field op string
|
||||
--- @field sources integer[]|nil
|
||||
--- @field immediate integer|nil
|
||||
--- @field source integer|nil
|
||||
|
||||
--- @class InstructionRow
|
||||
--- @field cycles integer
|
||||
--- @field kind string
|
||||
--- @field reads integer[]|nil
|
||||
--- @field writes integer[]|nil
|
||||
--- @field imm InstructionImm[]|nil
|
||||
--- @field value InstructionValue|nil
|
||||
--- @field delay_slot boolean|nil
|
||||
--- @field suppress_arg1 table<string, string>|nil -- bag: GPR ident -> reason
|
||||
|
||||
--- @class TapeAtomMacroRow
|
||||
--- @field kind string
|
||||
--- @field binds boolean
|
||||
|
||||
--- @class GteCommandPort
|
||||
--- @field register string
|
||||
--- @field role string
|
||||
|
||||
--- @class GteCommandLatch
|
||||
--- @field register string
|
||||
--- @field required integer
|
||||
|
||||
--- @class GteCommandRow
|
||||
--- @field aliases string[]
|
||||
--- @field cycles integer
|
||||
--- @field inputs string[]
|
||||
--- @field outputs GteCommandPort[]
|
||||
--- @field latch GteCommandLatch[]
|
||||
|
||||
--- @class GteCrAliasGroup
|
||||
--- @field [1] integer -- C2 control-register slot
|
||||
--- @field [2] string[] -- aliases that share that slot
|
||||
|
||||
--- @class GtePackedSlotRelation
|
||||
--- @field slot integer
|
||||
--- @field first string
|
||||
--- @field second string
|
||||
|
||||
--- @class HardwareRelationPort
|
||||
--- @field domain string
|
||||
--- @field arg integer
|
||||
|
||||
--- @class HardwareRelationVisibility
|
||||
--- @field kind string
|
||||
--- @field required integer
|
||||
|
||||
--- @class HardwareRelationEvidence
|
||||
--- @field confidence string
|
||||
--- @field source string
|
||||
|
||||
--- @class HardwareRelationRow
|
||||
--- @field id string
|
||||
--- @field semantic string
|
||||
--- @field consumer string
|
||||
--- @field token string
|
||||
--- @field direction string
|
||||
--- @field reads HardwareRelationPort
|
||||
--- @field writes HardwareRelationPort
|
||||
--- @field visibility HardwareRelationVisibility|nil
|
||||
--- @field evidence HardwareRelationEvidence
|
||||
--- @field violation_kind string
|
||||
--- @field destination_match string|nil
|
||||
--- @field fanout_to string[]|nil
|
||||
--- @field required integer|nil
|
||||
--- @field clear_on_consumer boolean|nil
|
||||
--- @field stage boolean|nil
|
||||
--- @field cu2_transition boolean|nil
|
||||
--- @field status_register integer|nil
|
||||
|
||||
--- @class Cu2TransitionPolicy
|
||||
--- @field status_register integer
|
||||
--- @field enable_bit integer
|
||||
--- @field required integer
|
||||
--- @field visibility_kind string
|
||||
--- @field evidence HardwareRelationEvidence
|
||||
|
||||
--- @class GprRole
|
||||
--- @field name string
|
||||
--- @field pool boolean
|
||||
--- @field optional boolean
|
||||
--- @field carrier boolean
|
||||
|
||||
--- @class DuffleIsa
|
||||
--- @field GPR_ROLE table<string, GprRole>
|
||||
--- @field TAPE_ATOM_MACROS table<string, TapeAtomMacroRow>
|
||||
--- @field DELAY_MARKERS table<string, boolean>
|
||||
--- @field INSTRUCTION table<string, InstructionRow>
|
||||
--- @field GTE_COMMAND table<string, GteCommandRow>
|
||||
--- @field ALIAS_TO_CANONICAL table<string, string>
|
||||
--- @field instr fun(ident: string): InstructionRow|nil
|
||||
--- @field gte_canon fun(ident: string): string
|
||||
--- @field gte fun(ident: string): GteCommandRow|nil
|
||||
--- @field GTE_CR_ALIAS_GROUPS GteCrAliasGroup[]
|
||||
--- @field GTE_PACKED_SLOT_RELATIONS GtePackedSlotRelation[]
|
||||
--- @field OPERAND_READ_POSITIONS table<string, integer[]>
|
||||
--- @field GP0_CMD_SIZE table<integer, integer>
|
||||
--- @field GP0_CMD_BY_SHAPE table<string, integer>
|
||||
--- @field UNKNOWN_INSTRUCTION_CYCLES integer
|
||||
--- @field HARDWARE_RELATIONS HardwareRelationRow[]
|
||||
--- @field CU2_TRANSITION_POLICY Cu2TransitionPolicy
|
||||
|
||||
local M = {} ---@type DuffleIsa
|
||||
|
||||
-- Section 7: domain tables
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- One GprRole row per name. Construction order is the auto_reg pool order,
|
||||
-- then R_AT, then the three carriers. Index by name into M.GPR_ROLE.
|
||||
--- @type table<string, GprRole>
|
||||
M.GPR_ROLE = {
|
||||
{ name = "R_V0", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_V1", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_T0", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_T1", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_T2", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_T3", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_T4", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_T5", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_T6", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_T7", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_A0", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_A1", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_A2", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_A3", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_S0", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_S1", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_S2", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_S3", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_S4", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_S5", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_S6", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_S7", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_T8", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_T9", pool = true, optional = true, carrier = false },
|
||||
{ name = "R_AT", pool = false, optional = true, carrier = false },
|
||||
{ name = "R_TapePtr", pool = false, optional = true, carrier = true },
|
||||
{ name = "R_AtomJmp", pool = false, optional = true, carrier = true },
|
||||
{ name = "R_ScratchBase", pool = false, optional = true, carrier = true },
|
||||
}
|
||||
for _, row in ipairs(M.GPR_ROLE) do ---@type integer, GprRole
|
||||
M.GPR_ROLE[row.name] = row
|
||||
end
|
||||
|
||||
-- atom_info sub-calls: atom_bind, atom_reads, atom_writes, atom_view, atom_reg_types, atom_ctx, atom_phase.
|
||||
--- @type table<string, TapeAtomMacroRow>
|
||||
M.TAPE_ATOM_MACROS = {
|
||||
["atom_info"] = { kind = "info", binds = false },
|
||||
}
|
||||
|
||||
-- Empty C macros that prefix the next encoder. Zero words.
|
||||
-- BdSlot_ nop is one nop word. The marker is not the BD instruction.
|
||||
--- @type table<string, boolean> -- bag: marker prefix -> true
|
||||
M.DELAY_MARKERS = {
|
||||
["GteDelay_"] = true,
|
||||
["LdSlot_"] = true,
|
||||
["BdSlot_"] = true,
|
||||
["DmaSlot_"] = true,
|
||||
}
|
||||
|
||||
-- One row per encoder. Read through duffle.instr.
|
||||
--- @type table<string, InstructionRow>
|
||||
M.INSTRUCTION = {
|
||||
["BdSlot_"] = { cycles = 0, kind = "marker", },
|
||||
["LdSlot_"] = { cycles = 0, kind = "marker", },
|
||||
["add_s"] = { cycles = 1, kind = "alu", },
|
||||
["add_si"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, },}, },
|
||||
["add_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||
["add_u_self"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, value = { dest = 1, op = "add_u", sources = { 1, 2 }, }, },
|
||||
["add_ui"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, value = { dest = 1, immediate = 3, op = "add_ui", source = 2, }, },
|
||||
["add_ui_self"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = { 1 }, imm = { { arg = 2, signed = true, width = 16, }, }, value = { dest = 1, immediate = 2, op = "add_ui", source = 1, }, },
|
||||
["and"] = { cycles = 1, kind = "alu", },
|
||||
["and_i"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, width = 16, }, }, value = { dest = 1, immediate = 3, op = "and_i", source = 2, }, },
|
||||
["and_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||
["atom_bind"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||
["atom_info"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||
["atom_label"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||
["atom_offset"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||
["atom_reads"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||
["atom_writes"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||
["branch_equal"] = { cycles = 2, kind = "branch", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||
["branch_ge_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
|
||||
["branch_gt_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
|
||||
["branch_le_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
|
||||
["branch_lt_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
|
||||
["branch_ne"] = { cycles = 2, kind = "branch", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||
["call_addr"] = { cycles = 2, kind = "call", reads = {}, writes = { 1 }, },
|
||||
["call_reg"] = { cycles = 2, kind = "call", reads = { 1 }, writes = { 2 }, },
|
||||
["div_s"] = { cycles = 35, kind = "alu", reads = { 1, 2 }, writes = {}, },
|
||||
["div_u"] = { cycles = 35, kind = "alu", reads = { 1, 2 }, writes = {}, },
|
||||
["gte_load_v0"] = { cycles = 2, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
|
||||
["gte_load_v0v1v2"] = { cycles = 6, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
|
||||
["gte_load_v1"] = { cycles = 2, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
|
||||
["gte_load_v2"] = { cycles = 2, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
|
||||
["gte_lw"] = { cycles = 1, kind = "load", reads = { 2 }, writes = {}, },
|
||||
["gte_lwc2"] = { cycles = 1, kind = "load", },
|
||||
["gte_mv_from_ctrl_r"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = { 1 }, },
|
||||
["gte_mv_from_data_r"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = { 1 }, },
|
||||
["gte_mv_to_ctrl_r"] = { cycles = 1, kind = "cop2_xfer", reads = { 1 }, writes = {}, },
|
||||
["gte_mv_to_data_r"] = { cycles = 1, kind = "cop2_xfer", reads = { 1 }, writes = {}, },
|
||||
["gte_stotz"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = {}, },
|
||||
["gte_stsxy3"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = {}, },
|
||||
["gte_sw"] = { cycles = 1, kind = "store", reads = { 2 }, writes = {}, },
|
||||
["gte_swc2"] = { cycles = 1, kind = "store", },
|
||||
["jump"] = { cycles = 2, kind = "jump", reads = {}, writes = {}, },
|
||||
["jump_link"] = { cycles = 2, kind = "call", reads = { 1 }, writes = { 2 }, },
|
||||
["jump_reg"] = { cycles = 2, kind = "jump", reads = { 1 }, writes = {}, suppress_arg1 = { R_AtomJmp = "fixed mac_yield handshake", }, },
|
||||
["jump_rel"] = { cycles = 2, kind = "branch", delay_slot = true, },
|
||||
["li_s"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, value = { dest = 1, immediate = 3, op = "add_ui", source = 2, }, },
|
||||
["load_byte"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||
["load_byte_u"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||
["load_half"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||
["load_half_u"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||
["load_imm"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, },
|
||||
["load_ui"] = { cycles = 1, kind = "alu", reads = {}, writes = { 1 }, },
|
||||
["load_upper_i"] = { cycles = 1, kind = "alu", reads = {}, writes = { 1 }, imm = { { arg = 2, width = 16, }, }, value = { dest = 1, immediate = 2, op = "load_upper_i", }, },
|
||||
["load_word"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||
["mac_yield"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||
["mask_upper"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, },
|
||||
["mov_from_high"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, },
|
||||
["mov_from_low"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, },
|
||||
["mov_to_high"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = {}, },
|
||||
["mov_to_low"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = {}, },
|
||||
["mult_s"] = { cycles = 12, kind = "alu", reads = { 1, 2 }, writes = {}, },
|
||||
["mult_u"] = { cycles = 12, kind = "alu", reads = { 1, 2 }, writes = {}, },
|
||||
["nop"] = { cycles = 1, kind = "nop", reads = {}, writes = {}, },
|
||||
["nop2"] = { cycles = 2, kind = "nop", reads = {}, writes = {}, },
|
||||
["nor_u"] = { cycles = 1, kind = "alu", },
|
||||
["or_i"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, width = 16, }, }, value = { dest = 1, immediate = 3, op = "or_i", source = 2, }, },
|
||||
["or_i_self"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = { 1 }, imm = { { arg = 2, width = 16, }, }, value = { dest = 1, immediate = 2, op = "or_i", source = 1, }, },
|
||||
["or_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||
["or_u_self"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, value = { dest = 1, op = "or", sources = { 1, 2 }, }, },
|
||||
["set_lt_s"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||
["set_lt_si"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, },
|
||||
["set_lt_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||
["set_lt_ui"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, },
|
||||
["shift_aright"] = { cycles = 1, kind = "alu", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
|
||||
["shift_aright_var"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
|
||||
["shift_lleft"] = { cycles = 1, kind = "alu", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
|
||||
["shift_lleft_self"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = { 1 }, imm = { { arg = 2, width = 5, }, }, value = { dest = 1, immediate = 2, op = "shift_lleft", source = 1, }, },
|
||||
["shift_lleft_var"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||
["shift_lright"] = { cycles = 1, kind = "alu", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
|
||||
["slt_s"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||
["slt_si"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||
["slt_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||
["slt_ui"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16,}, }, },
|
||||
["store_byte"] = { cycles = 1, kind = "store", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||
["store_half"] = { cycles = 1, kind = "store", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||
["store_word"] = { cycles = 1, kind = "store", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||
["sub_s"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||
["sub_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||
["sys_mov_from_cop0"] = { cycles = 1, kind = "cop0_xfer", reads = {}, writes = { 1 }, },
|
||||
["sys_mov_to_cop0"] = { cycles = 1, kind = "cop0_xfer", reads = { 1 }, writes = {}, },
|
||||
["xor_i"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, width = 16, }, }, value = { dest = 1, immediate = 3, op = "xor_i", source = 2, }, },
|
||||
["xor_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||
}
|
||||
|
||||
-- One row per GTE command. Alias cycle numbers live here, not on INSTRUCTION.
|
||||
--- @type table<string, GteCommandRow>
|
||||
M.GTE_COMMAND = {
|
||||
["gte_cmdw_avsz3"] = {
|
||||
aliases = { "gte_avg_sort_z3", "gte_avsz3", "gte_cmdw_avg_sort_z3" },
|
||||
cycles = 5,
|
||||
inputs = { "C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3", "gte_cr_ZSF3" },
|
||||
outputs = {
|
||||
{ register = "C2_OTZ", role = "otz", },
|
||||
},
|
||||
latch = {
|
||||
{ register = "C2_OTZ", required = 4, },
|
||||
},
|
||||
},
|
||||
["gte_cmdw_avsz4"] = {
|
||||
aliases = { "gte_avg_sort_z4", "gte_avsz4", "gte_cmdw_avg_sort_z4" },
|
||||
cycles = 6,
|
||||
inputs = { "C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3", "gte_cr_ZSF4" },
|
||||
outputs = {
|
||||
{ register = "C2_OTZ", role = "otz", },
|
||||
},
|
||||
latch = {
|
||||
{ register = "C2_OTZ", required = 4, },
|
||||
},
|
||||
},
|
||||
["gte_cmdw_gpf"] = {
|
||||
aliases = {},
|
||||
cycles = 5,
|
||||
inputs = { "C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3" },
|
||||
outputs = {
|
||||
{ register = "C2_MAC1", role = "mac_result", },
|
||||
{ register = "C2_MAC2", role = "mac_result", },
|
||||
{ register = "C2_MAC3", role = "mac_result", },
|
||||
{ register = "C2_IR1", role = "latest_color", },
|
||||
{ register = "C2_IR2", role = "latest_color", },
|
||||
{ register = "C2_IR3", role = "latest_color", },
|
||||
},
|
||||
latch = {
|
||||
{ register = "C2_MAC1", required = 4, },
|
||||
{ register = "C2_MAC2", required = 4, },
|
||||
{ register = "C2_MAC3", required = 4, },
|
||||
{ register = "C2_IR1", required = 4, },
|
||||
{ register = "C2_IR2", required = 4, },
|
||||
{ register = "C2_IR3", required = 4, },
|
||||
},
|
||||
},
|
||||
["gte_cmdw_mvmva"] = {
|
||||
aliases = {},
|
||||
cycles = 8,
|
||||
inputs = {
|
||||
"C2_VXY0", "C2_VZ0",
|
||||
"C2_VXY1", "C2_VZ1",
|
||||
"C2_VXY2", "C2_VZ2",
|
||||
"C2_IR1", "C2_IR2", "C2_IR3",
|
||||
"gte_cr_RT11", "gte_cr_RT12", "gte_cr_RT13",
|
||||
"gte_cr_RT21", "gte_cr_RT22", "gte_cr_RT23",
|
||||
"gte_cr_RT31", "gte_cr_RT32", "gte_cr_RT33",
|
||||
"gte_cr_TRX", "gte_cr_TRY", "gte_cr_TRZ"
|
||||
},
|
||||
outputs = {
|
||||
{ register = "C2_IR1", role = "latest_color", },
|
||||
{ register = "C2_IR2", role = "latest_color", },
|
||||
{ register = "C2_IR3", role = "latest_color", },
|
||||
},
|
||||
latch = {
|
||||
{ register = "C2_IR1", required = 4, },
|
||||
{ register = "C2_IR2", required = 4, },
|
||||
{ register = "C2_IR3", required = 4, },
|
||||
},
|
||||
},
|
||||
["gte_cmdw_nclip"] = {
|
||||
aliases = { "gte_nclip" },
|
||||
cycles = 8,
|
||||
inputs = { "C2_SXY0", "C2_SXY1", "C2_SXY2" },
|
||||
outputs = {
|
||||
{ register = "C2_SZ3", role = "mac_result", },
|
||||
},
|
||||
latch = {
|
||||
{ register = "C2_SZ3", required = 4, },
|
||||
},
|
||||
},
|
||||
["gte_cmdw_op"] = {
|
||||
aliases = { "gte_cmdw_outer_product", "gte_cmdw_wedge" },
|
||||
cycles = 6,
|
||||
inputs = {},
|
||||
outputs = {
|
||||
{ register = "C2_IR1", role = "latest_color", },
|
||||
{ register = "C2_IR2", role = "latest_color", },
|
||||
{ register = "C2_IR3", role = "latest_color", },
|
||||
},
|
||||
latch = {
|
||||
{ register = "C2_IR1", required = 4, },
|
||||
{ register = "C2_IR2", required = 4, },
|
||||
{ register = "C2_IR3", required = 4, },
|
||||
},
|
||||
},
|
||||
["gte_cmdw_rtps"] = {
|
||||
aliases = { "gte_cmdw_rotate_translate_perspective_single", "gte_rtps" },
|
||||
cycles = 15,
|
||||
inputs = {
|
||||
"C2_VXY0", "C2_VZ0",
|
||||
"C2_VXY1", "C2_VZ1",
|
||||
"C2_VXY2", "C2_VZ2",
|
||||
"C2_RGB", "C2_OTZ",
|
||||
"C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3",
|
||||
"C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3",
|
||||
"gte_cr_RT11", "gte_cr_RT12", "gte_cr_RT13",
|
||||
"gte_cr_RT21", "gte_cr_RT22", "gte_cr_RT23",
|
||||
"gte_cr_RT31", "gte_cr_RT32", "gte_cr_RT33",
|
||||
"gte_cr_TRX", "gte_cr_TRY", "gte_cr_TRZ",
|
||||
"gte_cr_OFX", "gte_cr_OFY",
|
||||
"gte_cr_H",
|
||||
"gte_cr_DQA", "gte_cr_DQB"
|
||||
},
|
||||
outputs = {
|
||||
{ register = "C2_SXY2", role = "latest_screen_xy", },
|
||||
{ register = "C2_SZ2", role = "latest_screen_z", },
|
||||
{ register = "C2_OTZ", role = "otz", },
|
||||
{ register = "C2_IR0", role = "latest_color", },
|
||||
},
|
||||
latch = {
|
||||
{ register = "C2_SXY2", required = 4, },
|
||||
{ register = "C2_SZ2", required = 4, },
|
||||
{ register = "C2_OTZ", required = 4, },
|
||||
{ register = "C2_IR0", required = 4, },
|
||||
},
|
||||
},
|
||||
["gte_cmdw_rtpt"] = {
|
||||
aliases = { "gte_cmdw_rotate_translate_perspective_triple", "gte_rtpt" },
|
||||
cycles = 23,
|
||||
inputs = {
|
||||
"C2_VXY0", "C2_VZ0",
|
||||
"C2_VXY1", "C2_VZ1",
|
||||
"C2_VXY2", "C2_VZ2",
|
||||
"C2_RGB", "C2_OTZ",
|
||||
"C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3",
|
||||
"C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3",
|
||||
"gte_cr_RT11", "gte_cr_RT12", "gte_cr_RT13",
|
||||
"gte_cr_RT21", "gte_cr_RT22", "gte_cr_RT23",
|
||||
"gte_cr_RT31", "gte_cr_RT32", "gte_cr_RT33",
|
||||
"gte_cr_TRX", "gte_cr_TRY", "gte_cr_TRZ",
|
||||
"gte_cr_OFX", "gte_cr_OFY",
|
||||
"gte_cr_H",
|
||||
"gte_cr_DQA", "gte_cr_DQB"
|
||||
},
|
||||
outputs = {
|
||||
{ register = "C2_SXY0", role = "screen_xy[0]", },
|
||||
{ register = "C2_SXY1", role = "screen_xy[1]", },
|
||||
{ register = "C2_SXY2", role = "latest_screen_xy", },
|
||||
{ register = "C2_SZ3", role = "latest_screen_z", },
|
||||
{ register = "C2_OTZ", role = "otz", },
|
||||
},
|
||||
latch = {
|
||||
{ register = "C2_SXY0", required = 4, },
|
||||
{ register = "C2_SXY1", required = 4, },
|
||||
{ register = "C2_SXY2", required = 4, },
|
||||
{ register = "C2_SZ3", required = 4, },
|
||||
{ register = "C2_OTZ", required = 4, },
|
||||
},
|
||||
},
|
||||
["gte_cmdw_sqr"] = {
|
||||
aliases = {},
|
||||
cycles = 5,
|
||||
inputs = { "C2_IR1", "C2_IR2", "C2_IR3" },
|
||||
outputs = {
|
||||
{ register = "C2_MAC1", role = "mac_result", },
|
||||
{ register = "C2_MAC2", role = "mac_result", },
|
||||
{ register = "C2_MAC3", role = "mac_result", },
|
||||
{ register = "C2_IR1", role = "latest_color", },
|
||||
{ register = "C2_IR2", role = "latest_color", },
|
||||
{ register = "C2_IR3", role = "latest_color", },
|
||||
},
|
||||
latch = {
|
||||
{ register = "C2_MAC1", required = 4, },
|
||||
{ register = "C2_MAC2", required = 4, },
|
||||
{ register = "C2_MAC3", required = 4, },
|
||||
{ register = "C2_IR1", required = 4, },
|
||||
{ register = "C2_IR2", required = 4, },
|
||||
{ register = "C2_IR3", required = 4, },
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
--- @param ident string
|
||||
--- @return InstructionRow|nil
|
||||
function M.instr (ident) return M.INSTRUCTION [ident] end
|
||||
--- @param ident string
|
||||
--- @return string
|
||||
function M.gte_canon(ident) return M.ALIAS_TO_CANONICAL [ident] or ident end
|
||||
--- @param ident string
|
||||
--- @return GteCommandRow|nil
|
||||
function M.gte (ident) return M.GTE_COMMAND[M.gte_canon(ident)] end
|
||||
|
||||
--- @return nil
|
||||
local function build_alias_map()
|
||||
--- @type table<string, string> -- bag: alias or canon -> canon
|
||||
M.ALIAS_TO_CANONICAL = {}
|
||||
for canon, row in pairs(M.GTE_COMMAND) do ---@type string, GteCommandRow
|
||||
M.ALIAS_TO_CANONICAL[canon] = canon
|
||||
for _, alias in ipairs(row.aliases or {}) do ---@type integer, string
|
||||
M.ALIAS_TO_CANONICAL[alias] = canon
|
||||
end
|
||||
end
|
||||
end
|
||||
build_alias_map()
|
||||
|
||||
|
||||
--- GTE control-register alias groups.
|
||||
--- Aliases within a group write to the same C2 control-register slot (the HW double-maps some C2 slots across multiple PSX SDK / libgte conventions).
|
||||
--- Aliases across groups write to distinct C2 slots.
|
||||
---
|
||||
--- Cross-alias writes inside one atom body, or across the wave-context boundary, silently clobber each other.
|
||||
--- The `check_gte_cr_alias_writes` check warns about each pair per source. See `docs/gte_reference.md` §"Control-register alias table"
|
||||
--- for the HW rationale and the libgte outer-product convention.
|
||||
--- @type GteCrAliasGroup[]
|
||||
M.GTE_CR_ALIAS_GROUPS = {
|
||||
{ 24, { "gte_cr_RBK", "gte_cr_OFX" } }, -- background R vs screen offset X
|
||||
{ 25, { "gte_cr_GBK", "gte_cr_OFY" } }, -- background G vs screen offset Y
|
||||
{ 26, { "gte_cr_BBK", "gte_cr_H" } }, -- background B vs projection plane distance H
|
||||
}
|
||||
|
||||
-- Packed RT slots named by the gte.h packed-slot comment. First must be written before second.
|
||||
--- @type GtePackedSlotRelation[]
|
||||
M.GTE_PACKED_SLOT_RELATIONS = {
|
||||
{ slot = 2, first = "gte_cr_RT13", second = "gte_cr_RT22" },
|
||||
}
|
||||
|
||||
-- Operand-class table for the COP2->GPR load-delay check.
|
||||
-- Maps each emitting-token ident to the set of GPR operand positions it reads.
|
||||
-- Covers the current encoder vocabulary (`code/duffle/mips.h` + `code/duffle/gte.h`); add rows here as new encoders land.
|
||||
--
|
||||
-- Semantics:
|
||||
-- * A "GPR operand position" is the textual slot in the macro's argument list, 1-based; e.g. `load_word(rt, base, off)` has positional operands 1 (rt), 2 (base), 3 (off).
|
||||
-- The table reads operands 1 + 2 + 3 to find what GPRs the macro touches.
|
||||
-- * The check tracks one entry per destination GPR per MFC2 / CFC2 event.
|
||||
-- A subsequent event counts as a "use" iff any of its read operand positions reference that destination GPR's ident (e.g. `R_T0`).
|
||||
-- * Branch delay slots are out of scope (MIPS control-flow; tracked separately).
|
||||
--- @type table<string, integer[]> -- bag: encoder ident -> GPR operand positions
|
||||
M.OPERAND_READ_POSITIONS = {
|
||||
-- CPU ALU with one or two GPR operands. Reads every GPR operand.
|
||||
["add_ui"] = {1, 2},
|
||||
["li_s"] = {1, 2}, -- rt (write), imm16 (immediate)
|
||||
["add_ui_self"] = {1},
|
||||
["add_si"] = {1, 2},
|
||||
["add_u"] = {1, 2, 3},
|
||||
["add_u_self"] = {1, 2},
|
||||
["sub_s"] = {1, 2, 3},
|
||||
["sub_u"] = {1, 2, 3},
|
||||
["and_i"] = {1, 2},
|
||||
["and"] = {1, 2, 3},
|
||||
["or_i"] = {1, 2},
|
||||
["or_i_self"] = {1},
|
||||
["or"] = {1, 2, 3},
|
||||
["or_self"] = {1, 2},
|
||||
["xor_i"] = {1, 2},
|
||||
["xor"] = {1, 2, 3},
|
||||
["slt_s"] = {1, 2, 3},
|
||||
["slt_u"] = {1, 2, 3},
|
||||
["slt_si"] = {1, 2},
|
||||
["slt_ui"] = {1, 2},
|
||||
["mult_s"] = {1, 2},
|
||||
["mult_u"] = {1, 2},
|
||||
["div_s"] = {1, 2},
|
||||
["div_u"] = {1, 2},
|
||||
-- Shifts: shift_lleft(rd, rt, shamt); the rt operand is the value, rd is dest.
|
||||
["shift_lleft"] = {1, 2},
|
||||
["shift_lright"] = {1, 2},
|
||||
["shift_aright"] = {1, 2},
|
||||
["shift_lleft_self"] = {1},
|
||||
-- Loads: load_word(rt, base, off); the rt operand is the destination (it's written, not read) and base + off are non-GPR operands.
|
||||
-- The check treats the rt operand as a write, so the read-positions table for `load_*` is empty.
|
||||
["load_word"] = {},
|
||||
["load_half_u"] = {},
|
||||
["load_byte_u"] = {},
|
||||
["load_half"] = {},
|
||||
["load_byte"] = {},
|
||||
["load_upper_i"] = {},
|
||||
["load_ui"] = {},
|
||||
-- Stores write to memory; base + rt operands are non-read for load-delay purposes.
|
||||
["store_word"] = {},
|
||||
["store_half"] = {},
|
||||
["store_byte"] = {},
|
||||
-- Branches read rs (+ rt for beq/bne). The branch delay slot is out of scope.
|
||||
["branch_equal"] = {1, 2},
|
||||
["branch_ne"] = {1, 2},
|
||||
["branch_le_zero"] = {1},
|
||||
["branch_lt_zero"] = {1},
|
||||
["branch_ge_zero"] = {1},
|
||||
["branch_gt_zero"] = {1},
|
||||
-- Jumps / link: jr / jalr read rs only (the target). RD is the destination link.
|
||||
["jump_reg"] = {1},
|
||||
["jump_link"] = {1},
|
||||
["call_reg"] = {1},
|
||||
["call_addr"] = {},
|
||||
["jump"] = {},
|
||||
-- mask_upper is a 2-word macro: shift_lleft then shift_lright. The first reads rt.
|
||||
["mask_upper"] = {1, 2},
|
||||
-- move from/to HI/LO.
|
||||
["mov_from_high"] = {},
|
||||
["mov_from_low"] = {},
|
||||
["mov_to_high"] = {1},
|
||||
["mov_to_low"] = {1},
|
||||
-- GTE transfers / loads / stores / commands: the relevant table values live in the check itself.
|
||||
-- `gte_mv_to_*` writes its rt operand; `gte_mv_from_*` writes its rt operand; `gte_*` commands are atomic-from-the-CPU-POV
|
||||
-- once they issue (the CPU holds until the command completes, so load-delay violations don't surface here).
|
||||
["gte_mv_from_data_r"] = {},
|
||||
["gte_mv_from_ctrl_r"] = {},
|
||||
["gte_mv_to_data_r"] = {},
|
||||
["gte_mv_to_ctrl_r"] = {},
|
||||
["gte_lw"] = {},
|
||||
["gte_sw"] = {},
|
||||
["shift_lleft_var"] = {1, 2, 3}, -- rd, rt, rs (variable shift amount)
|
||||
["shift_aright_var"] = {1, 2, 3},
|
||||
}
|
||||
|
||||
-- GP0 packet sizes (total words including the 1-word tag) per GP0 cmd byte.
|
||||
-- Per PSX-SPX `docs/psx-spx/docs/graphicsprocessingunitgpu.md` §"GPU Render Polygon Commands":
|
||||
-- Each polygon command's word count = 1 (tag/cmd) + per-vertex (vertex + optional color + optional UV).
|
||||
-- F3: cmd + 3 vertices = 4 words; +1 tag = 5
|
||||
-- F4: cmd + 4 vertices = 5 words; +1 tag = 6
|
||||
-- G3: cmd + 3×(color + vertex) = 6 words; +1 tag = 7
|
||||
-- G4: cmd + 4×(color + vertex) = 8 words; +1 tag = 9
|
||||
-- FT3: cmd + tpage + clut + 3×(vertex + UV) = 7 words; +1 tag = 8
|
||||
-- FT4: cmd + tpage + clut + 4×(vertex + UV) = 9 words; +1 tag = 10
|
||||
-- GT3: cmd + tpage + clut + 3×(color + vertex + UV) = 9 words; +1 tag = 10
|
||||
-- GT4: cmd + tpage + clut + 4×(color + vertex + UV) = 12 words; +1 tag = 13
|
||||
--
|
||||
-- Cross-checked against code/duffle/gp.h struct sizes + the set_poly_* macros
|
||||
-- (which encode "len" = "words after tag"):
|
||||
-- set_poly_f3(p) -> set_len(p, 4) -> 5 total GP0 0x20
|
||||
-- set_poly_ft3(p) -> set_len(p, 7) -> 8 total GP0 0x24
|
||||
-- set_poly_f4(p) -> set_len(p, 5) -> 6 total GP0 0x28
|
||||
-- set_poly_ft4(p) -> set_len(p, 9) -> 10 total GP0 0x2C
|
||||
-- set_poly_g3(p) -> set_len(p, 6) -> 7 total GP0 0x30
|
||||
-- set_poly_gt3(p) -> set_len(p, 9) -> 10 total GP0 0x34
|
||||
-- set_poly_g4(p) -> set_len(p, 8) -> 9 total GP0 0x38
|
||||
-- set_poly_gt4(p) -> set_len(p, 12) -> 13 total GP0 0x3C
|
||||
--- @type table<integer, integer> -- bag: GP0 cmd byte -> word count
|
||||
M.GP0_CMD_SIZE = {
|
||||
[0x20] = 5, -- Poly_F3
|
||||
[0x24] = 8, -- Poly_FT3
|
||||
[0x28] = 6, -- Poly_F4
|
||||
[0x2C] = 10, -- Poly_FT4
|
||||
[0x30] = 7, -- Poly_G3
|
||||
[0x34] = 10, -- Poly_GT3
|
||||
[0x38] = 9, -- Poly_G4
|
||||
[0x3C] = 13, -- Poly_GT4
|
||||
}
|
||||
|
||||
-- Shape suffix (after `ac_format_` / `mac_format_` prefix) -> GP0 cmd byte.
|
||||
-- Lets the static-analysis check derive the cmd byte from a macro name like `mac_format_g4_color` -> `g4` -> 0x38 -> 9 expected words.
|
||||
--- @type table<string, integer> -- bag: shape suffix -> GP0 cmd byte
|
||||
M.GP0_CMD_BY_SHAPE = {
|
||||
["f3"] = 0x20, ["ft3"] = 0x24,
|
||||
["f4"] = 0x28, ["ft4"] = 0x2C,
|
||||
["g3"] = 0x30, ["gt3"] = 0x34,
|
||||
["g4"] = 0x38, ["gt4"] = 0x3C,
|
||||
}
|
||||
|
||||
--- @type integer
|
||||
M.UNKNOWN_INSTRUCTION_CYCLES = 1
|
||||
|
||||
-- Hardware-relation policy table.
|
||||
--
|
||||
-- The forward walker in `passes/static_analysis.lua::analyze_hardware_relations` reads every emitted word_event, matches its `encoder` against `row.token`, and:
|
||||
-- * stages the event as a producer in `atom.paths.forward_state`; or
|
||||
-- * matches it as a consumer against pending producers and records a hazard on `atom.paths.hazards` when the gap is below `visibility.required`.
|
||||
--
|
||||
-- Each row is the contract for one CPU-to-coprocessor transfer semantic (the coprocessor-to-CPU path mirrors the same shape).
|
||||
-- The `reads` / `writes` sub-tables carry the argument positions the analyzer inspects:
|
||||
-- * `writes.arg` is the destination operand (the producer's effect); the analyzer stages this register as a pending producer.
|
||||
-- * `reads` (when present) lists the operand positions the same token reads back from hardware; for MTC2 / CTC2 the producer reads the GPR source it is loading from.
|
||||
-- The `fanout_to` field (MTC2-IRGB row only) tells the consumer-match logic which downstream COP2 registers are transitively updated by the write.
|
||||
--
|
||||
-- Visibility semantics:
|
||||
-- * `kind = "post_producer_words"` means the consumer observes the producer's effect after `required` independent emitted words that are
|
||||
-- strictly between the producer and the consumer. The producer's own emitted slot is implicit (it counts as the slot of issue, not toward `required`)
|
||||
-- per the PSX-SPX rule: "Store delays are counted in numbers of clock cycles (not in numbers of opcodes).
|
||||
-- For 3 cycle delay, one must usually insert 3 cached opcodes (or one uncached opcode)."
|
||||
-- * `required` is the minimum count of intervening emitted words between producer and consumer.
|
||||
-- `required = 0` permits the consumer on the very next slot; `required < 0` would place the consumer on the same slot as the producer
|
||||
-- and is reserved for future "self-retires" relations.
|
||||
--
|
||||
-- Evidence:
|
||||
-- * `evidence.confidence` is one of `"exact"`, `"conservative"`, `"unknown"`. The severity comes from `violation_kind`;
|
||||
-- A hardware measurement that the vendor caveats may still classify as `"conservative"` even when the underlying timing is numerically known.
|
||||
-- * `evidence.source` is the upstream reference (file + line range) the row is sourced from. New rows must carry this citation.
|
||||
--
|
||||
-- Consumers:
|
||||
-- * passes/static_analysis.lua::analyze_hardware_relations (forward walker).
|
||||
-- * passes/static_analysis.lua::transfer_hazards CHECK_RULES reader (renders hazards onto `findings`).
|
||||
-- This table is consumed by the hardware-relation analyzer and hazard renderer.
|
||||
--- @type HardwareRelationRow[]
|
||||
M.HARDWARE_RELATIONS = {
|
||||
-- CPU → COP2 data register (MTC2). The ordinary default is 2 cached words between producer and consumer (cpuspecifications.md:407-419).
|
||||
{
|
||||
id = "mtc2_gpr_visibility",
|
||||
semantic = "MTC2",
|
||||
consumer = "cop2_input",
|
||||
token = "gte_mv_to_data_r",
|
||||
direction = "gpr_to_cop2_data",
|
||||
reads = { domain = "gpr", arg = 1 },
|
||||
writes = { domain = "cop2.data", arg = 2 },
|
||||
visibility = { kind = "post_producer_words", required = 2 },
|
||||
evidence = {
|
||||
confidence = "exact",
|
||||
source = "cpuspecifications.md:407-419",
|
||||
},
|
||||
violation_kind = "error",
|
||||
},
|
||||
-- CPU → COP2 data register when the destination is C2_IRGB (data 28).
|
||||
-- C2_IRGB drives the IR1/IR2/IR3 color-conversion fan-out, which extends the propagation delay to 3 cached words.
|
||||
-- `destination_match = "C2_IRGB"` is the row's filter; the analyzer consults this when the producer's destination operand equals "C2_IRGB".
|
||||
-- C2_ORGB (data 29) is read-only and is never classified as a writable fan-out destination.
|
||||
{
|
||||
id = "mtc2_irgb_visibility",
|
||||
semantic = "MTC2",
|
||||
consumer = "cop2_input",
|
||||
token = "gte_mv_to_data_r",
|
||||
direction = "gpr_to_cop2_data",
|
||||
reads = { domain = "gpr", arg = 1 },
|
||||
writes = { domain = "cop2.data", arg = 2 },
|
||||
destination_match = "C2_IRGB",
|
||||
fanout_to = { "C2_IR1", "C2_IR2", "C2_IR3" },
|
||||
visibility = { kind = "post_producer_words", required = 3 },
|
||||
evidence = {
|
||||
confidence = "exact",
|
||||
source = "cpuspecifications.md:407-419",
|
||||
},
|
||||
violation_kind = "error",
|
||||
},
|
||||
-- CPU → COP2 control register (CTC2). Ordinary minimum 2;
|
||||
-- no IRGB-style fan-out exists for control registers (per spec §3.6: only C2_IRGB has the 3-cycle fan-out on the data side).
|
||||
{
|
||||
id = "ctc2_gpr_visibility",
|
||||
semantic = "CTC2",
|
||||
consumer = "cop2_input",
|
||||
token = "gte_mv_to_ctrl_r",
|
||||
direction = "gpr_to_cop2_control",
|
||||
reads = { domain = "gpr", arg = 1 },
|
||||
writes = { domain = "cop2.ctrl", arg = 2 },
|
||||
visibility = { kind = "post_producer_words", required = 2 },
|
||||
evidence = {
|
||||
confidence = "exact",
|
||||
source = "cpuspecifications.md:407-419",
|
||||
},
|
||||
violation_kind = "error",
|
||||
},
|
||||
-- COP2 data → GPR (MFC2). One cached slot between the transfer and the first GPR consumer;
|
||||
-- the GPR is not updated until the instruction AFTER the MFC2 completes (geometrytransformationenginegte.md:29-32).
|
||||
{
|
||||
id = "mfc2_gpr_visibility",
|
||||
semantic = "MFC2",
|
||||
consumer = "gpr_read",
|
||||
token = "gte_mv_from_data_r",
|
||||
direction = "cop2_data_to_gpr",
|
||||
reads = { domain = "cop2.data", arg = 2 },
|
||||
writes = { domain = "gpr", arg = 1 },
|
||||
visibility = { kind = "post_producer_words", required = 1 },
|
||||
evidence = {
|
||||
confidence = "exact",
|
||||
source = "geometrytransformationenginegte.md:29-32",
|
||||
},
|
||||
violation_kind = "error",
|
||||
},
|
||||
-- COP2 control → GPR (CFC2). Same delay as MFC2 (cpuspecifications.md treats the two load-from-COP2 paths symmetrically).
|
||||
{
|
||||
id = "cfc2_gpr_visibility",
|
||||
semantic = "CFC2",
|
||||
consumer = "gpr_read",
|
||||
token = "gte_mv_from_ctrl_r",
|
||||
direction = "cop2_control_to_gpr",
|
||||
reads = { domain = "cop2.ctrl", arg = 2 },
|
||||
writes = { domain = "gpr", arg = 1 },
|
||||
visibility = { kind = "post_producer_words", required = 1 },
|
||||
evidence = {
|
||||
confidence = "exact",
|
||||
source = "cpuspecifications.md:382-419",
|
||||
},
|
||||
violation_kind = "error",
|
||||
},
|
||||
-- COP0 control → GPR (MFC0).
|
||||
-- One cached slot; the analyzer treats `sys_mov_from_cop0(rt, 12)` (the SR/CU2 transfer) as the same shape as the COP2 load-delay path.
|
||||
-- The semantic-level SR/CU2 transition models the load delay;
|
||||
-- SR.CU2 bounded-value propagation is modeled separately).
|
||||
{
|
||||
id = "mfc0_gpr_visibility",
|
||||
semantic = "MFC0",
|
||||
consumer = "gpr_read",
|
||||
token = "sys_mov_from_cop0",
|
||||
direction = "cop0_control_to_gpr",
|
||||
reads = { domain = "cop0.ctrl", arg = 2 },
|
||||
writes = { domain = "gpr", arg = 1 },
|
||||
visibility = { kind = "post_producer_words", required = 1 },
|
||||
evidence = {
|
||||
confidence = "exact",
|
||||
source = "cpuspecifications.md:171-178",
|
||||
},
|
||||
violation_kind = "error",
|
||||
},
|
||||
-- Memory -> COP2 data register (LWC2).
|
||||
-- The memory-side timing is not measured by the vendored GTE latch experiment, so this relation has no numeric retirement threshold.
|
||||
-- The LWC2 destination has TWO retirement regimes (per PSX-SPX):
|
||||
-- * GTE-command consumer (`gte_cmdw_*`): the GTE pipeline LATCHES the LWC2 result, so a `gte_cmdw_*`
|
||||
-- in the very next slot uses the latched value. Gap = 0 is allowed. (Per `docs/psx-spx/docs/gtepipelinetimings.md:271-274`.)
|
||||
-- * Any other consumer: standard MIPS load delay applies. Gap = 1 required. (Per `docs/psx-spx/docs/cpuspecifications.md:407-419`.)
|
||||
-- Two separate relations so the walker can dispatch by consumer type and emit different severities
|
||||
-- (the GTE-command path is `info` because the latch is intentional; the non-GTE-consumer path is `error` because the missing nop is a real bug).
|
||||
{
|
||||
id = "lwc2_to_gte_command",
|
||||
semantic = "LWC2_to_GTE",
|
||||
consumer = "cop2_input",
|
||||
token = "gte_lw",
|
||||
direction = "memory_to_cop2_data",
|
||||
reads = { domain = "memory", arg = 2 },
|
||||
writes = { domain = "cop2.data", arg = 1 },
|
||||
required = 0, -- GTE-command consumer: gap = 0 OK (latched).
|
||||
evidence = {
|
||||
confidence = "measured",
|
||||
source = "gtepipelinetimings.md:271-274",
|
||||
},
|
||||
violation_kind = "info",
|
||||
clear_on_consumer = true,
|
||||
},
|
||||
{
|
||||
id = "lwc2_to_other_consumer",
|
||||
semantic = "LWC2_to_other",
|
||||
consumer = "cop2_input",
|
||||
token = "gte_lw",
|
||||
direction = "memory_to_cop2_data",
|
||||
reads = { domain = "memory", arg = 2 },
|
||||
writes = { domain = "cop2.data", arg = 1 },
|
||||
required = 1, -- Non-GTE-consumer: standard MIPS load delay.
|
||||
evidence = {
|
||||
confidence = "inferred",
|
||||
source = "cpuspecifications.md:407-419",
|
||||
},
|
||||
violation_kind = "error",
|
||||
clear_on_consumer = true,
|
||||
},
|
||||
-- COP2 data register -> memory (SWC2). A read of C2 state, not a CPU-to-COP2 write.
|
||||
-- The policy row stays in for direction/provenance; staging it as a later command-input producer is suppressed.
|
||||
{
|
||||
id = "swc2_memory_write",
|
||||
semantic = "SWC2",
|
||||
consumer = "gpr_read",
|
||||
token = "gte_sw",
|
||||
direction = "cop2_data_to_memory",
|
||||
reads = { domain = "cop2.data", arg = 1 },
|
||||
writes = { domain = "memory", arg = 2 },
|
||||
visibility = { kind = "none", required = 0 },
|
||||
evidence = {
|
||||
confidence = "exact",
|
||||
source = "cpuspecifications.md:79",
|
||||
},
|
||||
violation_kind = "info",
|
||||
stage = false,
|
||||
},
|
||||
-- MTC0 Status/SR.CU2. The ordinary COP0 store has no general store-delay relation;
|
||||
-- this row feeds the dedicated CU2 transition logic in the same forward walk and is therefore not staged in `pending`.
|
||||
{
|
||||
id = "mtc0_cu2_visibility",
|
||||
semantic = "MTC0",
|
||||
consumer = "gpr_read",
|
||||
token = "sys_mov_to_cop0",
|
||||
direction = "gpr_to_cop0_status",
|
||||
reads = { domain = "gpr", arg = 1 },
|
||||
writes = { domain = "cop0.status", arg = 2 },
|
||||
status_register = 12,
|
||||
visibility = { kind = "post_producer_words", required = 2 },
|
||||
evidence = {
|
||||
confidence = "conservative",
|
||||
source = "cpuspecifications.md:543,625-628",
|
||||
},
|
||||
violation_kind = "warning",
|
||||
stage = false,
|
||||
cu2_transition = true,
|
||||
},
|
||||
}
|
||||
|
||||
-- Bounded Status/SR.CU2 transition policy.
|
||||
-- The value lattice and the transition consumer both read this immutable row; no second value pass is permitted.
|
||||
-- The source says the enable/disable transition takes "2 clock cycles or so", so the boundary is conservative rather than exact.
|
||||
--- @type Cu2TransitionPolicy
|
||||
M.CU2_TRANSITION_POLICY = {
|
||||
status_register = 12,
|
||||
enable_bit = 0x40000000,
|
||||
required = 2,
|
||||
visibility_kind = "post_producer_words",
|
||||
evidence = {
|
||||
confidence = "conservative",
|
||||
source = "cpuspecifications.md:543,625-628",
|
||||
},
|
||||
}
|
||||
|
||||
return M
|
||||
@@ -0,0 +1,89 @@
|
||||
--- duffle_paths.lua — Single-line bootstrap helper for the tape-atom Lua scripts.
|
||||
---
|
||||
--- Each entry script (ps1_meta.lua + the 7 passes/*.lua files) starts with one of:
|
||||
--- ```lua
|
||||
--- -- Entry script (ps1_meta.lua — `arg[0]` is set):
|
||||
--- local duffle = dofile((arg[0]:match("(.*[/\\])") or "./") .. "duffle_paths.lua")
|
||||
---
|
||||
--- -- Pass module (debug.getinfo path resolution; works both standalone and when require'd):
|
||||
--- local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||
--- local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
--- ```
|
||||
---
|
||||
--- That small bootstrap: (a) locates this helper via `arg[0]` / `debug.getinfo`,
|
||||
--- (b) loads it (which sets `package.path` + `package.cpath`),
|
||||
--- (c) at the bottom calls `require("duffle")` (now resolvable since `package.path` was just set) and returns the duffle M.
|
||||
--- Net effect: the caller gets the duffle module in one statement; no separate `dofile(...)` + `require("duffle")` dance.
|
||||
---
|
||||
|
||||
--- @class DufflePaths
|
||||
--- @field setup fun(): nil
|
||||
|
||||
local M = {} ---@type DufflePaths
|
||||
|
||||
-- Cache key for the repo root. Stored in `package.loaded` (process-global) so all 8 entry scripts + passes scripts share one resolution.
|
||||
local CACHE_KEY = "__duffle_repo_root__" ---@type string
|
||||
|
||||
--- Resolve the repo root from this script's own path. Zero shell spawn.
|
||||
--- `duffle_paths.lua` always lives at `<repo>/scripts/duffle_paths.lua`, so the repo root is the parent of the directory containing this script.
|
||||
--- We derive it directly from `debug.getinfo(1, "S").source` (returns `@<path>` for the currently-running chunk).
|
||||
---
|
||||
--- If `debug.getinfo` can't parse this script's path (shouldn't happen — dofile always populates source), return nil and let `M.setup()` fail loud.
|
||||
--- @return string|nil
|
||||
local function find_repo_root()
|
||||
if package.loaded[CACHE_KEY] then return package.loaded[CACHE_KEY] end
|
||||
|
||||
local source = debug.getinfo(1, "S").source ---@type string
|
||||
-- Strip the leading `@` (Lua's dofile marker) and the trailing `/duffle_paths.lua` filename.
|
||||
-- What remains is the directory containing this script, i.e. `<repo>/scripts/`.
|
||||
local scripts_dir = source and source:match("^@?(.*)[/\\]duffle_paths%.lua$") ---@type string|nil
|
||||
if not scripts_dir then return nil end
|
||||
|
||||
-- The repo root is the parent of `scripts/`. Strip the trailing `scripts/` (with or without trailing slash).
|
||||
local root = scripts_dir:gsub("scripts[\\/]?$", "") ---@type string
|
||||
root = root:gsub("\\", "/")
|
||||
if root == "" then root = "./" end
|
||||
if not root:match("/$") then root = root .. "/" end
|
||||
package.loaded[CACHE_KEY] = root
|
||||
return root
|
||||
end
|
||||
|
||||
--- Set `package.path` (for `require("duffle")` + `require("passes.X")`) and `package.cpath` (for `lpeg.dll`).
|
||||
---
|
||||
--- This script does NOT touch the OS environment: no `os.setenv`, no `os.putenv`, no `$PATH` mods.
|
||||
--- It just sets `package.path` and `package.cpath` (the standard Lua way to register module search dirs).
|
||||
--- lpeg is built by `update_deps.ps1` to `toolchain/lpeg/`, which we wire into `package.cpath` here (so `require("lpeg")` from `duffle.lua` resolves without any global state).
|
||||
--- @return nil
|
||||
function M.setup()
|
||||
local repo_root = find_repo_root() ---@type string|nil
|
||||
if not repo_root then
|
||||
-- Unreachable in practice: find_repo_root() derives the repo root from this script's own source path via debug.getinfo(1, "S").source (no subprocess, no git CLI, <1ms).
|
||||
-- A nil return means the source path did not match the expected <repo>/scripts/duffle_paths.lua layout — a packaging bug, not a "missing git repo" condition.
|
||||
-- os.exit(2) is retained so a real failure surfaces loud rather than silently producing an unconfigured module table.
|
||||
os.exit(2)
|
||||
end
|
||||
|
||||
local scripts_dir = repo_root .. "scripts/" ---@type string
|
||||
local passes_dir = repo_root .. "scripts/passes/" ---@type string
|
||||
package.path = scripts_dir .. "?.lua;"
|
||||
.. scripts_dir .. "?/init.lua;"
|
||||
.. passes_dir .. "?.lua;"
|
||||
.. passes_dir .. "?/init.lua;"
|
||||
.. package.path
|
||||
|
||||
-- lpeg: built by `update_deps.ps1` to `toolchain/lpeg/lpeg.dll`.
|
||||
-- lfs: compiled from pcsx-redux's vendored luafilesystem source to `toolchain/lfs/lfs.dll`.
|
||||
-- Wire both directories into cpath so `require("lpeg")` and `require("lfs")` resolve.
|
||||
local lpeg_dir = repo_root .. "toolchain/lpeg/" ---@type string
|
||||
local lfs_dir = repo_root .. "toolchain/lfs/" ---@type string
|
||||
package.cpath = lpeg_dir .. "?.dll;"
|
||||
.. lfs_dir .. "?.dll;"
|
||||
.. package.cpath
|
||||
end
|
||||
|
||||
-- Run the setup as a side effect.
|
||||
M.setup()
|
||||
|
||||
-- Now that package.path includes scripts/, `require("duffle")` resolves.
|
||||
-- Return the duffle module so callers can do `local duffle = dofile(...duffle_paths.lua)` in one line.
|
||||
return require("duffle")
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,511 @@
|
||||
-- elf32.lua — Pure-Lua ELF32 format helpers with no lfs / no lpeg dependency.
|
||||
-- The reload helper's `parse_manifest` (scripts/pcsx_debug_helper/reload.lua)
|
||||
-- and the metaprogram's `read_elf_sections` + `read_nm` (scripts/elf_dwarf.lua)
|
||||
-- both parsed ELF32 headers from wire bytes.
|
||||
--
|
||||
-- This module contains the format constants and the byte-level walker.
|
||||
--- The metaprogram side keeps `read_u32_le` / `read_u16_le` as local forwarders; the helper side calls `E.*` directly.
|
||||
--
|
||||
-- **Adapter contract (explicit pass style):**
|
||||
-- The helper VM's `Support.File` exposes byte-read methods that require `self` (fileffi.lua:225-227),
|
||||
-- so callers wrap once in a 1-line adapter that strips `self`.
|
||||
-- The parsers here operate on the unwrapped form.
|
||||
-- Reads are flat function calls — `E.read_u8(adapter, off)`, `E.read_u32(adapter, off)`, `E.size(adapter)`.
|
||||
-- read_u8(adapter, off) -> integer | nil
|
||||
-- read_u16(adapter, off) -> integer | nil
|
||||
-- read_u32(adapter, off) -> integer | nil
|
||||
-- size(adapter) -> integer
|
||||
--
|
||||
-- **Convention:** every offset in the constants tables is a zero-based wire offset.
|
||||
-- The `+ 1` conversion happens only at the `string.byte` boundary inside the readers.
|
||||
--
|
||||
-- spec: System V ABI gABI v1.2 §"ELF Header" (Table 1) + §"Section Header Table"
|
||||
-- spec: System V ABI gABI v1.2 §"Symbol Table" (Elf32_Sym layout)
|
||||
|
||||
--- @class Elf32Adapter
|
||||
--- @field read_u8_at fun(off: integer): integer|nil
|
||||
--- @field read_u16_at fun(off: integer): integer|nil
|
||||
--- @field read_u32_at fun(off: integer): integer|nil
|
||||
--- @field read_size fun(): integer
|
||||
|
||||
--- @class Elf32Header
|
||||
--- @field e_entry integer
|
||||
--- @field e_shoff integer
|
||||
--- @field e_shentsize integer
|
||||
--- @field e_shnum integer
|
||||
--- @field e_shstrndx integer
|
||||
--- @field error string|nil
|
||||
|
||||
--- @class Elf32Section
|
||||
--- @field sh_name integer
|
||||
--- @field sh_type integer
|
||||
--- @field sh_flags integer
|
||||
--- @field sh_addr integer
|
||||
--- @field sh_offset integer
|
||||
--- @field sh_size integer
|
||||
--- @field sh_link integer
|
||||
--- @field name string
|
||||
|
||||
--- @class Elf32Sym
|
||||
--- @field value integer
|
||||
--- @field size integer
|
||||
--- @field info integer
|
||||
--- @field shndx integer
|
||||
|
||||
--- @class Elf32HeaderLayout
|
||||
--- @field magic_offset integer
|
||||
--- @field magic string
|
||||
--- @field class_offset integer
|
||||
--- @field endian_offset integer
|
||||
--- @field header_bytes integer
|
||||
--- @field e_entry_offset integer
|
||||
--- @field e_shoff_offset integer
|
||||
--- @field e_shentsize_offset integer
|
||||
--- @field e_shnum_offset integer
|
||||
--- @field e_shstrndx_offset integer
|
||||
|
||||
--- @class Elf32SectionLayout
|
||||
--- @field sh_name_offset integer
|
||||
--- @field sh_type_offset integer
|
||||
--- @field sh_flags_offset integer
|
||||
--- @field sh_addr_offset integer
|
||||
--- @field sh_offset_offset integer
|
||||
--- @field sh_size_offset integer
|
||||
--- @field sh_link_offset integer
|
||||
--- @field sh_entsize_bytes integer
|
||||
|
||||
--- @class Elf32SymLayout
|
||||
--- @field st_name integer
|
||||
--- @field st_value integer
|
||||
--- @field st_size integer
|
||||
--- @field st_info integer
|
||||
--- @field sym_entry_bytes integer
|
||||
|
||||
--- @class Elf32Mod
|
||||
--- @field ELFCLASS32 integer
|
||||
--- @field ELFDATA2LSB integer
|
||||
--- @field EM_MIPS integer
|
||||
--- @field SHT_SYMTAB integer
|
||||
--- @field SHT_STRTAB integer
|
||||
--- @field SHT_NOBITS integer
|
||||
--- @field SHF_WRITE integer
|
||||
--- @field SHF_ALLOC integer
|
||||
--- @field SHF_EXECINSTR integer
|
||||
--- @field ELF32_HEADER Elf32HeaderLayout
|
||||
--- @field ELF32_SECTION Elf32SectionLayout
|
||||
--- @field ELF32_SYM Elf32SymLayout
|
||||
--- @field dw_dwarf32_terminator integer
|
||||
--- @field read_u32 fun(adapter: Elf32Adapter, off: integer): integer|nil
|
||||
--- @field read_u16 fun(adapter: Elf32Adapter, off: integer): integer|nil
|
||||
--- @field read_u8 fun(adapter: Elf32Adapter, off: integer): integer|nil
|
||||
--- @field size fun(adapter: Elf32Adapter): integer
|
||||
--- @field read_u32_le fun(buf: string, off: integer): integer
|
||||
--- @field read_u16_le fun(buf: string, off: integer): integer
|
||||
--- @field validate_adapter fun(adapter: any): boolean, string|nil
|
||||
--- @field get_str fun(strtab: string, off: integer): string|nil
|
||||
--- @field parse_elf32_headers fun(adapter: Elf32Adapter): Elf32Header|nil, string|nil
|
||||
--- @field walk_sections fun(adapter: Elf32Adapter, hdr: Elf32Header): Elf32Section[]|nil, string|nil
|
||||
--- @field read_section_bytes fun(adapter: Elf32Adapter, section: Elf32Section): string|nil
|
||||
--- @field read_named_section fun(adapter: Elf32Adapter, sections: Elf32Section[], name: string): string|nil, string|nil
|
||||
--- @field collect_symbols fun(adapter: Elf32Adapter, sections: Elf32Section[]): table<string, Elf32Sym>|nil, string|nil
|
||||
|
||||
local M = {} ---@type Elf32Mod
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Little-endian readers (bit-weighted accumulator, math.floor only)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Read a 4-byte little-endian unsigned integer from `adapter` at zero-based wire offset `off`.
|
||||
---
|
||||
--- Bit weights are written as `0x100`, `0x10000`, `0x1000000` (i.e. 2^8, 2^16, 2^24) so the LE byte positions are visually explicit:
|
||||
--- byte 0 contributes its value directly;
|
||||
--- byte 1 is shifted left by 8; byte 2 by 16; byte 3 by 24.
|
||||
---
|
||||
--- math.floor (not LuaJIT's `>>`) keeps the body portable across LuaJIT 2.0/2.1 and plain Lua 5.x. `string.byte` receives `+ 1` at the boundary.
|
||||
---
|
||||
--- **Call form:** explicit-pass. The reader receives `adapter` as the first positional argument and the offset as the second; no `self` is passed.
|
||||
--- Test fixtures declare `function(offset) ... end` and the parsers call them via dot syntax `adapter.read_u8_at(off)`.
|
||||
--- The colon form `adapter:read_u8_at(off)` would prepend the adapter table as `offset` and break the contract.
|
||||
--- @param adapter Elf32Adapter
|
||||
--- @param off integer -- zero-based wire offset
|
||||
--- @return integer|nil
|
||||
function M.read_u32(adapter, off)
|
||||
return adapter.read_u8_at(off)
|
||||
+ adapter.read_u8_at(off + 0x01) * 0x00000100
|
||||
+ adapter.read_u8_at(off + 0x02) * 0x00010000
|
||||
+ adapter.read_u8_at(off + 0x03) * 0x01000000
|
||||
end
|
||||
|
||||
--- Read a 2-byte little-endian unsigned integer from `adapter` at zero-based wire offset `off`.
|
||||
--- @param adapter Elf32Adapter
|
||||
--- @param off integer -- zero-based wire offset
|
||||
--- @return integer|nil
|
||||
function M.read_u16(adapter, off)
|
||||
return adapter.read_u8_at(off)
|
||||
+ adapter.read_u8_at(off + 0x01) * 0x00000100
|
||||
end
|
||||
|
||||
--- Read a 1-byte unsigned integer from `adapter` at zero-based wire offset `off`.
|
||||
--- @param adapter Elf32Adapter
|
||||
--- @param off integer -- zero-based wire offset
|
||||
--- @return integer|nil
|
||||
function M.read_u8(adapter, off)
|
||||
return adapter.read_u8_at(off)
|
||||
end
|
||||
|
||||
--- Total adapter byte length.
|
||||
--- @param adapter Elf32Adapter
|
||||
--- @return integer
|
||||
function M.size(adapter)
|
||||
return adapter.read_size()
|
||||
end
|
||||
|
||||
--- Forwarders kept for backward compat with scripts/elf_dwarf.lua.
|
||||
--- The metaprogram side keeps `read_u32_le` / `read_u16_le`;
|
||||
--- both layers now use the same byte-level helpers under the hood.
|
||||
--- @param buf string
|
||||
--- @param off integer
|
||||
--- @return integer
|
||||
function M.read_u32_le(buf, off)
|
||||
local byte_off = off + 1 ---@type integer
|
||||
return buf:byte(byte_off)
|
||||
+ buf:byte(byte_off + 0x01) * 0x00000100
|
||||
+ buf:byte(byte_off + 0x02) * 0x00010000
|
||||
+ buf:byte(byte_off + 0x03) * 0x01000000
|
||||
end
|
||||
|
||||
--- Read a 2-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
|
||||
--- @param buf string
|
||||
--- @param off integer -- zero-based wire offset
|
||||
--- @return integer
|
||||
function M.read_u16_le(buf, off)
|
||||
local byte_off = off + 1 ---@type integer
|
||||
return buf:byte(byte_off) + buf:byte(byte_off + 0x01) * 0x00000100
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Format constants
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- ELF format constants (System V ABI gABI v1.2).
|
||||
M.ELFCLASS32 = 1 -- spec: gABI v1.2 §"ELF Header" — EI_CLASS byte
|
||||
M.ELFDATA2LSB = 1 -- spec: gABI v1.2 §"ELF Header" — EI_DATA byte
|
||||
M.EM_MIPS = 8 -- spec: gABI v1.2 §"Machine Information" — MIPS architecture
|
||||
|
||||
-- Section type constants (System V ABI gABI v1.2 §"Section Header Table").
|
||||
M.SHT_SYMTAB = 2 -- spec: gABI v1.2 §"Section Types" — symbol table
|
||||
M.SHT_STRTAB = 3 -- spec: gABI v1.2 §"Section Types" — string table
|
||||
M.SHT_NOBITS = 8 -- spec: gABI v1.2 §"Section Types" — no space in file
|
||||
|
||||
-- Section flag constants (System V ABI gABI v1.2 §"Section Header Table").
|
||||
M.SHF_WRITE = 0x1 -- spec: gABI v1.2 §"Section Attributes" — writable
|
||||
M.SHF_ALLOC = 0x2 -- spec: gABI v1.2 §"Section Attributes" — occupies memory
|
||||
M.SHF_EXECINSTR = 0x4 -- spec: gABI v1.2 §"Section Attributes" — executable
|
||||
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- ELF32 header layout (System V ABI gABI v1.2 §"ELF Header" Table 1)
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- All offsets are zero-based wire offsets. The header is 52 bytes total (header_bytes = 0x34 = 52).
|
||||
--- @type Elf32HeaderLayout
|
||||
M.ELF32_HEADER = {
|
||||
magic_offset = 0x00, -- 4 bytes; expected "\127ELF"
|
||||
magic = "\127ELF",
|
||||
class_offset = 0x04, -- 1 byte; 1 = ELF32, 2 = ELF64
|
||||
endian_offset = 0x05, -- 1 byte; 1 = little-endian, 2 = big-endian
|
||||
header_bytes = 0x34, -- ELF32 header is 52 bytes total
|
||||
e_entry_offset = 0x18, -- 4-byte LE; entry-point virtual address
|
||||
e_shoff_offset = 0x20, -- 4-byte LE; section-header table file offset
|
||||
e_shentsize_offset = 0x2E, -- 2-byte LE; section-header entry size in bytes
|
||||
e_shnum_offset = 0x30, -- 2-byte LE; number of section headers
|
||||
e_shstrndx_offset = 0x32, -- 2-byte LE; index of section-name string table
|
||||
}
|
||||
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- ELF32 section-header layout (System V ABI gABI v1.2 §"Section Header Table")
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- Each entry is 40 bytes (sh_entsize_bytes = 0x28 = 40);
|
||||
-- zero-based, field offsets relative to the start of the entry.
|
||||
--- @type Elf32SectionLayout
|
||||
M.ELF32_SECTION = {
|
||||
sh_name_offset = 0x00, -- 4-byte LE; offset into .shstrtab
|
||||
sh_type_offset = 0x04, -- 4-byte LE; section type (SHT_*)
|
||||
sh_flags_offset = 0x08, -- 4-byte LE; section flags (SHF_*)
|
||||
sh_addr_offset = 0x0C, -- 4-byte LE; virtual address at execution
|
||||
sh_offset_offset = 0x10, -- 4-byte LE; section's file offset
|
||||
sh_size_offset = 0x14, -- 4-byte LE; section's size in bytes
|
||||
sh_link_offset = 0x18, -- 4-byte LE; link to a related section
|
||||
sh_entsize_bytes = 0x28, -- spec: gABI v1.2 §"Section Header Table" — 40 bytes per entry
|
||||
}
|
||||
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- ELF32 symbol-table entry layout (System V ABI gABI v1.2 §"Symbol Table")
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- Each entry is 16 bytes (sym_entry_bytes = 0x10 = 16);
|
||||
-- zero-based, field offsets relative to the start of the entry.
|
||||
--- @type Elf32SymLayout
|
||||
M.ELF32_SYM = {
|
||||
st_name = 0x00, -- 4-byte LE; offset into the linked string table
|
||||
st_value = 0x04, -- 4-byte LE; symbol value (address / absolute)
|
||||
st_size = 0x08, -- 4-byte LE; symbol size in bytes
|
||||
st_info = 0x0C, -- 1 byte; binding (high nibble) + type (low nibble)
|
||||
sym_entry_bytes = 0x10, -- spec: gABI v1.2 §"Symbol Table" — 16 bytes per entry
|
||||
}
|
||||
|
||||
-- DWARF32 initial-length terminator (DWARF4 §7.4) — kept here so the metaprogram's elf_dwarf.lua can drop its own copy of the same constant.
|
||||
M.dw_dwarf32_terminator = 0xFFFFFFFF
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Adapter validation
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Validate that `adapter` exposes the byte-read surface.
|
||||
--- Returns true on success, false + a stable error code on failure.
|
||||
--- The helper side calls this before parse_manifest to reject callers before any byte is read.
|
||||
--- @param adapter any
|
||||
--- @return boolean, string|nil
|
||||
function M.validate_adapter(adapter)
|
||||
if type(adapter) ~= "table" then return false, "bad_file_adapter" end
|
||||
if type(adapter.read_u8_at) ~= "function" then return false, "bad_file_adapter" end
|
||||
if type(adapter.read_u16_at) ~= "function" then return false, "bad_file_adapter" end
|
||||
if type(adapter.read_u32_at) ~= "function" then return false, "bad_file_adapter" end
|
||||
if type(adapter.read_size) ~= "function" then return false, "bad_file_adapter" end
|
||||
return true, nil
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- String-table reader
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Extract a NUL-terminated C string from `strtab` at zero-based offset `off`.
|
||||
--- Returns nil if `off` is out of range or the string is not NUL-terminated.
|
||||
--- @param strtab string
|
||||
--- @param off integer
|
||||
--- @return string|nil
|
||||
function M.get_str(strtab, off)
|
||||
if off < 0 or off >= #strtab then return nil end
|
||||
local end_pos = strtab:find("\0", off + 1, true) ---@type integer|nil
|
||||
if not end_pos then return nil end
|
||||
return strtab:sub(off + 1, end_pos - 1)
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Header / section / symbol walkers
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Read the ELF32 header through `adapter` and validate the magic, class, and data encoding.
|
||||
--- Returns a table on success:
|
||||
--- { e_entry, e_shoff, e_shentsize, e_shnum, e_shstrndx, error = nil }
|
||||
--- On failure returns nil + a stable error code:
|
||||
--- bad_magic, unsupported_elf_class, unsupported_elf_data, truncated_header
|
||||
--- The header's machine field is NOT validated here — callers (e.g. the helper's prime path) decide whether to require EM_MIPS before symbol reads.
|
||||
--- @param adapter Elf32Adapter
|
||||
--- @return Elf32Header|nil, string|nil
|
||||
function M.parse_elf32_headers(adapter)
|
||||
local ok, err = M.validate_adapter(adapter) ---@type boolean, string|nil
|
||||
if not ok then return nil, err end
|
||||
|
||||
-- 4-byte magic: 0x7F 'E' 'L' 'F'.
|
||||
-- The byte readers take the adapter explicitly.
|
||||
-- The production `Support.File` adapter is wrapped by the caller to drop its implicit `self` so the parser shape is flat pass-style.
|
||||
local b1 = M.read_u8(adapter, 0) ---@type integer|nil
|
||||
local b2 = M.read_u8(adapter, 1) ---@type integer|nil
|
||||
local b3 = M.read_u8(adapter, 2) ---@type integer|nil
|
||||
local b4 = M.read_u8(adapter, 3) ---@type integer|nil
|
||||
if not (b1 and b2 and b3 and b4)
|
||||
or not (b1 == 0x7f and b2 == 0x45 and b3 == 0x4c and b4 == 0x46) then
|
||||
return nil, "bad_magic"
|
||||
end
|
||||
|
||||
local class = M.read_u8(adapter, M.ELF32_HEADER.class_offset) ---@type integer|nil
|
||||
if class ~= M.ELFCLASS32 then
|
||||
return nil, "unsupported_elf_class"
|
||||
end
|
||||
|
||||
local data = M.read_u8(adapter, M.ELF32_HEADER.endian_offset) ---@type integer|nil
|
||||
if data ~= M.ELFDATA2LSB then
|
||||
return nil, "unsupported_elf_data"
|
||||
end
|
||||
|
||||
local e_entry = M.read_u32(adapter, M.ELF32_HEADER.e_entry_offset) ---@type integer|nil
|
||||
local e_shoff = M.read_u32(adapter, M.ELF32_HEADER.e_shoff_offset) ---@type integer|nil
|
||||
local e_shentsize = M.read_u16(adapter, M.ELF32_HEADER.e_shentsize_offset) ---@type integer|nil
|
||||
local e_shnum = M.read_u16(adapter, M.ELF32_HEADER.e_shnum_offset) ---@type integer|nil
|
||||
local e_shstrndx = M.read_u16(adapter, M.ELF32_HEADER.e_shstrndx_offset) ---@type integer|nil
|
||||
if not (e_entry and e_shoff and e_shentsize and e_shnum and e_shstrndx) then
|
||||
return nil, "truncated_header"
|
||||
end
|
||||
|
||||
return {
|
||||
e_entry = e_entry,
|
||||
e_shoff = e_shoff,
|
||||
e_shentsize = e_shentsize,
|
||||
e_shnum = e_shnum,
|
||||
e_shstrndx = e_shstrndx,
|
||||
error = nil,
|
||||
}
|
||||
end
|
||||
|
||||
--- Read one section-header entry from `adapter` at `sh_off`.
|
||||
--- Returns a table with the wire fields plus a (yet-unresolved) `name` field.
|
||||
--- @param adapter Elf32Adapter
|
||||
--- @param sh_off integer
|
||||
--- @return Elf32Section|nil, string|nil
|
||||
local function read_section_entry(adapter, sh_off)
|
||||
local entry = { ---@type Elf32Section
|
||||
sh_name = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_name_offset),
|
||||
sh_type = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_type_offset),
|
||||
sh_flags = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_flags_offset),
|
||||
sh_addr = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_addr_offset),
|
||||
sh_offset = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_offset_offset),
|
||||
sh_size = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_size_offset),
|
||||
sh_link = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_link_offset),
|
||||
name = "",
|
||||
}
|
||||
if not (entry.sh_name and entry.sh_type and entry.sh_flags and entry.sh_addr
|
||||
and entry.sh_offset and entry.sh_size and entry.sh_link) then
|
||||
return nil, "truncated_section_headers"
|
||||
end
|
||||
return entry, nil
|
||||
end
|
||||
|
||||
--- Walk every section header in `hdr` and return a 1-based array of entries
|
||||
--- (the section at logical index 0 is at array position 1, etc.).
|
||||
--- Each entry has the wire fields plus a resolved `name` derived from `.shstrtab`.
|
||||
--- Returns nil + a stable error code on failure: truncated_section_headers, missing_shstrtab, truncated_strtab
|
||||
--- @param adapter Elf32Adapter
|
||||
--- @param hdr Elf32Header
|
||||
--- @return Elf32Section[]|nil, string|nil
|
||||
function M.walk_sections(adapter, hdr)
|
||||
if not hdr or hdr.error then return nil, hdr and hdr.error or "truncated_section_headers" end
|
||||
|
||||
local file_size = M.size(adapter) ---@type integer
|
||||
if hdr.e_shoff + hdr.e_shnum * hdr.e_shentsize > file_size then
|
||||
return nil, "truncated_section_headers"
|
||||
end
|
||||
|
||||
-- Read every section header first; we need .shstrtab to resolve names.
|
||||
local sections = {} ---@type Elf32Section[]
|
||||
for i = 0, hdr.e_shnum - 1 do ---@type integer
|
||||
local sh_off = hdr.e_shoff + i * hdr.e_shentsize ---@type integer
|
||||
local entry, err = read_section_entry(adapter, sh_off) ---@type Elf32Section|nil, string|nil
|
||||
if not entry then return nil, err end
|
||||
sections[i + 1] = entry
|
||||
end
|
||||
|
||||
if hdr.e_shstrndx >= hdr.e_shnum then
|
||||
return nil, "missing_shstrtab"
|
||||
end
|
||||
|
||||
local shstrtab = sections[hdr.e_shstrndx + 1] ---@type Elf32Section|nil
|
||||
if not shstrtab or shstrtab.sh_type ~= M.SHT_STRTAB then
|
||||
return nil, "missing_shstrtab"
|
||||
end
|
||||
if shstrtab.sh_offset + shstrtab.sh_size > file_size then
|
||||
return nil, "truncated_section_headers"
|
||||
end
|
||||
local shstrtab_bytes = M.read_section_bytes(adapter, shstrtab) ---@type string|nil
|
||||
if not shstrtab_bytes then return nil, "truncated_section_headers" end
|
||||
|
||||
for _, s in ipairs(sections) do ---@type integer, Elf32Section
|
||||
s.name = M.get_str(shstrtab_bytes, s.sh_name) or ""
|
||||
end
|
||||
|
||||
return sections, nil
|
||||
end
|
||||
|
||||
--- Read the bytes of one section. Returns a string, or nil if the adapter returns nil for any byte (out-of-bounds).
|
||||
--- The caller is responsible fors sizing the buffer (the section's sh_offset + sh_size must fit in adapter.size).
|
||||
--- @param adapter Elf32Adapter
|
||||
--- @param section Elf32Section
|
||||
--- @return string|nil
|
||||
function M.read_section_bytes(adapter, section)
|
||||
local size = section.sh_size ---@type integer
|
||||
if size == 0 then return "" end
|
||||
local out = {} ---@type string[]
|
||||
for i = 0, size - 1 do ---@type integer
|
||||
local b = M.read_u8(adapter, section.sh_offset + i) ---@type integer|nil
|
||||
if b == nil then return nil end
|
||||
out[#out + 1] = string.char(b)
|
||||
end
|
||||
return table.concat(out)
|
||||
end
|
||||
|
||||
--- Convenience: walk sections, then look up the named section, then read its bytes.
|
||||
--- Returns nil + a stable error code if the section is absent or out-of-bounds.
|
||||
--- @param adapter Elf32Adapter
|
||||
--- @param sections Elf32Section[]
|
||||
--- @param name string
|
||||
--- @return string|nil, string|nil
|
||||
function M.read_named_section(adapter, sections, name)
|
||||
if not sections then return nil, "missing_section" end
|
||||
for _, s in ipairs(sections) do ---@type integer, Elf32Section
|
||||
if s.name == name then
|
||||
local bytes = M.read_section_bytes(adapter, s) ---@type string|nil
|
||||
if not bytes then return nil, "truncated_section_data" end
|
||||
return bytes, nil
|
||||
end
|
||||
end
|
||||
return nil, "missing_section"
|
||||
end
|
||||
|
||||
--- Walk every SHT_SYMTAB section in `sections` and accumulate symbols by name.
|
||||
--- Each stored entry is `{ value = st_value, size = st_size, info = st_info, shndx = st_shndx }`.
|
||||
--- Both STB_LOCAL and STB_GLOBAL symbols are included; the live ELF stores `smem` as a local symbol.
|
||||
--- Returns nil + a stable error code on failure: missing_symtab_strtab, truncated_section_headers
|
||||
--- @param adapter Elf32Adapter
|
||||
--- @param sections Elf32Section[]
|
||||
--- @return table<string, Elf32Sym>|nil, string|nil
|
||||
function M.collect_symbols(adapter, sections)
|
||||
if not sections then return nil, "missing_sections" end
|
||||
local symbols = {} ---@type table<string, Elf32Sym> -- bag: symbol name -> Elf32Sym
|
||||
local file_size = M.size(adapter) ---@type integer
|
||||
for _, s in ipairs(sections) do ---@type integer, Elf32Section
|
||||
if s.sh_type == M.SHT_SYMTAB then
|
||||
local strtab = sections[s.sh_link + 1] ---@type Elf32Section|nil
|
||||
if not strtab or strtab.sh_type ~= M.SHT_STRTAB then
|
||||
return nil, "missing_symtab_strtab"
|
||||
end
|
||||
if strtab.sh_offset + strtab.sh_size > file_size then
|
||||
return nil, "truncated_section_headers"
|
||||
end
|
||||
local strtab_bytes = M.read_section_bytes(adapter, strtab) ---@type string|nil
|
||||
if not strtab_bytes then return nil, "truncated_section_headers" end
|
||||
if s.sh_offset + s.sh_size > file_size then
|
||||
return nil, "truncated_section_headers"
|
||||
end
|
||||
local symtab_bytes = M.read_section_bytes(adapter, s) ---@type string|nil
|
||||
if not symtab_bytes then return nil, "truncated_section_headers" end
|
||||
local n = #symtab_bytes / M.ELF32_SYM.sym_entry_bytes ---@type number
|
||||
for j = 0, n - 1 do ---@type integer
|
||||
local e = s.sh_offset + j * M.ELF32_SYM.sym_entry_bytes ---@type integer
|
||||
local st_name = M.read_u32(adapter, e + M.ELF32_SYM.st_name) ---@type integer|nil
|
||||
if st_name then
|
||||
local st_value = M.read_u32(adapter, e + M.ELF32_SYM.st_value) ---@type integer|nil
|
||||
local st_size = M.read_u32(adapter, e + M.ELF32_SYM.st_size) ---@type integer|nil
|
||||
local st_info = M.read_u8(adapter, e + M.ELF32_SYM.st_info) ---@type integer|nil
|
||||
-- st_shndx is at offset 14 (2 bytes) — derived from the layout
|
||||
-- the metaprogram reads too. Inline the read to keep the
|
||||
-- adapter as the only I/O surface.
|
||||
local b1 = M.read_u8(adapter, e + 14) ---@type integer|nil
|
||||
local b2 = M.read_u8(adapter, e + 15) ---@type integer|nil
|
||||
if not (b1 and b2) then
|
||||
return nil, "truncated_section_headers"
|
||||
end
|
||||
local st_shndx = b1 + b2 * 0x100 ---@type integer
|
||||
local name = M.get_str(strtab_bytes, st_name) or "" ---@type string
|
||||
if name ~= "" then
|
||||
symbols[name] = {
|
||||
value = st_value,
|
||||
size = st_size,
|
||||
info = st_info,
|
||||
shndx = st_shndx,
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
return symbols, nil
|
||||
end
|
||||
|
||||
return M
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,105 @@
|
||||
# scripts/gdb/gdb_tape_atoms.gdb
|
||||
#
|
||||
# Wrapper for the tape-atom step-debug helpers.
|
||||
# The 9 user commands are defined here as STUBS (degraded-state messages).
|
||||
# The real implementations + the per-atom data tables are emitted by `passes/atoms_source_map.lua`
|
||||
# (post-link invocation: `ps1_meta.lua --atoms-source-map --gdb-runtime --elf <elf>`) into `build/gdb_tape_atoms_runtime.gdb`.
|
||||
# Sourcing that file RE-DEFINES the commands with real implementations.
|
||||
#
|
||||
# If `build/gdb_tape_atoms_runtime.gdb` is missing or stale, the stubs remain (E1: no source map).
|
||||
# The user just needs to re-run `build_psyq.ps1` to regenerate.
|
||||
|
||||
# ?? Stub commands (defined here so they're always present, even if the runtime file is missing). The runtime file overrides these if sourced. ??
|
||||
|
||||
define tape_atoms
|
||||
echo "[gdb_tape_atoms] STUB: runtime file build/gdb_tape_atoms_runtime.gdb not found."
|
||||
echo "[gdb_tape_atoms] STUB: run .\\build_psyq.ps1 to regenerate, then re-source this file."
|
||||
end
|
||||
document tape_atoms
|
||||
List every tape atom symbol in the loaded ELF with its .rodata address and word count.
|
||||
STUB state: runtime file not sourced. Run build_psyq.ps1 to regenerate.
|
||||
end
|
||||
|
||||
define break_atom
|
||||
echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
end
|
||||
document break_atom
|
||||
Set a breakpoint at the start of tape atom <name>. STUB state.
|
||||
end
|
||||
|
||||
define step_atom
|
||||
echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
end
|
||||
document step_atom
|
||||
Resume execution until the next atom boundary. STUB state.
|
||||
end
|
||||
|
||||
define next_atom
|
||||
echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
end
|
||||
document next_atom
|
||||
Alias for step_atom. STUB state.
|
||||
end
|
||||
|
||||
define where_in_atom
|
||||
echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
end
|
||||
document where_in_atom
|
||||
Report current atom name, .rodata addr, word offset, and source line (if known). STUB state.
|
||||
end
|
||||
|
||||
define stepi_inside_atom
|
||||
echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
end
|
||||
document stepi_inside_atom
|
||||
One MIPS-instruction step, then where_in_atom. STUB state.
|
||||
end
|
||||
|
||||
define show_c2
|
||||
printf "C2[ 0] 0x%08x\n", $c2_data[0]
|
||||
printf "C2[ 7] 0x%08x [otz]\n", $c2_data[7]
|
||||
printf "C2[12] 0x%08x [sxy0]\n", $c2_data[12]
|
||||
printf "C2[13] 0x%08x [sxy1]\n", $c2_data[13]
|
||||
printf "C2[14] 0x%08x [sxy2]\n", $c2_data[14]
|
||||
printf "C2[24] 0x%08x [mac0]\n", $c2_data[24]
|
||||
printf "...\n"
|
||||
echo "(STUB state: only 7 representative regs shown. Run build_psyq.ps1 for full dump.)"
|
||||
end
|
||||
document show_c2
|
||||
Pretty-print all 32 C2 data registers as hex + named alias. STUB state (7 reg subset).
|
||||
end
|
||||
|
||||
define show_c2ctl
|
||||
printf "C2CTL[ 0] 0x%08x\n", $c2_control[0]
|
||||
printf "...\n"
|
||||
echo "(STUB state: only 1 reg shown. Run build_psyq.ps1 for full dump.)"
|
||||
end
|
||||
document show_c2ctl
|
||||
Pretty-print all 32 C2 control registers. STUB state (1 reg subset).
|
||||
end
|
||||
|
||||
define wave_ctx
|
||||
printf "$t4 = R_FaceCursor 0x%08x\n", $t4
|
||||
printf "$t5 = R_VertBase 0x%08x\n", $t5
|
||||
printf "$t6 = R_OtBase 0x%08x\n", $t6
|
||||
printf "$t7 = R_PrimCursor 0x%08x\n", $t7
|
||||
end
|
||||
document wave_ctx
|
||||
Pretty-print the 4 wave-context GPRs ($t4..$t7). (wave_ctx works in stub state too.)
|
||||
end
|
||||
|
||||
|
||||
# ?? Source the runtime file (re-defines commands with real impls + data). ??
|
||||
|
||||
# Try to source from project-root-relative path first (the typical case).
|
||||
# If the user is in a different CWD, the source will fail and stubs remain.
|
||||
# The runtime file path is computed relative to the ELF's source map convention (build/gdb_tape_atoms_runtime.gdb).
|
||||
echo [gdb_tape_atoms] Wrapper loaded. Sourcing runtime file...
|
||||
# Suppress the "Redefine command" prompts that would otherwise appear when the runtime file overrides the 9 stub commands defined above.
|
||||
# The runtime's `define` blocks are intended to overwrite ? there's no ambiguity to confirm.
|
||||
set confirm off
|
||||
|
||||
# Source the runtime file (re-defines commands with real impls + data).
|
||||
source build/gdb_tape_atoms_runtime.gdb
|
||||
set confirm on
|
||||
echo [gdb_tape_atoms] Runtime sourced successfully (9 commands now have real implementations).
|
||||
@@ -0,0 +1,94 @@
|
||||
# scripts/launch_pcsx_debug.ps1
|
||||
#
|
||||
# One-shot launcher for debug sessions:
|
||||
# Starts pcsx-redux with the .ps-exe loaded, the gdb stub enabled,
|
||||
# AND the pcsx_debug_helper Lua plugin loaded so external CLI tools (gdb's `shell` command, etc.)
|
||||
# can read GTE state via http://localhost:8080/api/v1/lua/gte (the gdb stub doesn't expose COP2 at all).
|
||||
#
|
||||
# usage:
|
||||
# .\scripts\launch_pcsx_debug.ps1
|
||||
# .\scripts\launch_pcsx_debug.ps1 -ExePath build\hello_gte.ps-exe
|
||||
# .\scripts\launch_pcsx_debug.ps1 -HelperZip scripts\pcsx_debug_helper.zip
|
||||
#
|
||||
# After launch:
|
||||
# - gdb: target remote localhost:3333
|
||||
# - web: curl http://localhost:8080/api/v1/lua/gte
|
||||
#
|
||||
# Companion: scripts/debug_psyq.ps1 (bare launch — no .ps-exe, no helper).
|
||||
|
||||
[CmdletBinding()]
|
||||
param(
|
||||
[string]$PcsxPath = (Join-Path $PSScriptRoot '..\toolchain\pcsx-redux\vsprojects\x64\Release\pcsx-redux.exe'),
|
||||
[string]$ExePath = (Join-Path $PSScriptRoot '..\build\hello_gte.ps-exe'),
|
||||
[string]$HelperZip = (Join-Path $PSScriptRoot 'pcsx_debug_helper.zip'),
|
||||
[int] $GdbPort = 3333,
|
||||
[int] $WebPort = 8080
|
||||
)
|
||||
|
||||
$ErrorActionPreference = 'Stop'
|
||||
|
||||
# ── Pre-checks ──
|
||||
foreach ($p in @($PcsxPath, $ExePath, $HelperZip)) {
|
||||
if (-not (Test-Path $p)) {
|
||||
Write-Error "Missing: $p"
|
||||
exit 1
|
||||
}
|
||||
}
|
||||
|
||||
# Kill any existing pcsx-redux so the archive file isn't locked.
|
||||
Get-Process pcsx-redux -ErrorAction SilentlyContinue | Stop-Process -Force
|
||||
Start-Sleep -Seconds 2
|
||||
|
||||
# ── Launch ──
|
||||
$absExe = [System.IO.Path]::GetFullPath($ExePath)
|
||||
$absZip = [System.IO.Path]::GetFullPath($HelperZip)
|
||||
|
||||
$args = @(
|
||||
'-gdb', '-run'
|
||||
'-loadexe', "`"$absExe`""
|
||||
'-archive', "`"$absZip`""
|
||||
)
|
||||
|
||||
Write-Host "Launching pcsx-redux..." -ForegroundColor Cyan
|
||||
Write-Host " ps-exe : $absExe"
|
||||
Write-Host " helper zip: $absZip"
|
||||
Write-Host " gdb : localhost:$GdbPort"
|
||||
Write-Host " web : localhost:$WebPort/api/v1/lua/gte"
|
||||
Write-Host ""
|
||||
|
||||
Start-Process -FilePath $PcsxPath -ArgumentList $args | Out-Null
|
||||
|
||||
# ── Wait for both endpoints to come up ──
|
||||
$deadline = (Get-Date).AddSeconds(15)
|
||||
while ((Get-Date) -lt $deadline) {
|
||||
$gdbUp = $false
|
||||
$webUp = $false
|
||||
try {
|
||||
$tcp = New-Object System.Net.Sockets.TcpClient
|
||||
$tcp.BeginConnect('localhost', $GdbPort, $null, $null) | Out-Null
|
||||
Start-Sleep -Milliseconds 100
|
||||
$gdbUp = $tcp.Connected
|
||||
$tcp.Close()
|
||||
} catch { $gdbUp = $false }
|
||||
try {
|
||||
$r = Invoke-WebRequest -Uri "http://localhost:$WebPort/" -UseBasicParsing -TimeoutSec 1 -ErrorAction SilentlyContinue
|
||||
$webUp = $r.StatusCode -ne 0
|
||||
} catch { $webUp = $false }
|
||||
if ($gdbUp -and $webUp) { break }
|
||||
Start-Sleep -Milliseconds 500
|
||||
}
|
||||
|
||||
# ── Smoke-test the gte handler ──
|
||||
try {
|
||||
$r = Invoke-WebRequest -Uri "http://localhost:$WebPort/api/v1/lua/gte" -UseBasicParsing -TimeoutSec 5
|
||||
$firstLine = ([System.Text.Encoding]::UTF8.GetString($r.Content) -split "`n")[0]
|
||||
Write-Host "GTE handler OK: $firstLine" -ForegroundColor Green
|
||||
}
|
||||
catch {
|
||||
Write-Warning "GTE handler NOT responding: $_"
|
||||
Write-Host "Check the pcsx-redux Lua Console for debug cli messages." -ForegroundColor Yellow
|
||||
}
|
||||
|
||||
Write-Host ""
|
||||
Write-Host "pcsx-redux running. PIDs:" -ForegroundColor Cyan
|
||||
Get-Process pcsx-redux | Select-Object Id, ProcessName | Format-Table
|
||||
+492
-872
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,626 @@
|
||||
--- passes/atoms_source_map.lua — Per-.word source-line map emitter for tape atoms.
|
||||
---
|
||||
--- Writer: this pass, given `atom.paths` (the per-atom mutable surface owned by `emission_model`). Readers:
|
||||
--- `passes/dwarf_injection.lua` (synthesizes DW_TAG_inlined_subroutine + per-word line program rows) and
|
||||
--- the gdb-runtime wrapper at `scripts/gdb/gdb_tape_atoms.gdb` (loads the source map via `source <path>`).
|
||||
---
|
||||
--- Inputs from `atom.paths`: the ordered `items` stream, dense `word_events`, `invocations` views. Outputs:
|
||||
--- one `WORD N LINE L TEXT T` line per emitted `.word`, plus the per-word provenance form that DWARF synthesis consumes.
|
||||
---
|
||||
--- Two output forms:
|
||||
--- 1. Markdown form: Handled by `passes/report.lua` (writes `<module>.atoms.md`).
|
||||
--- The render functions `render_source_map` + `render_provenance` are exported for `report.lua` to call directly.
|
||||
--- Compile artifacts (`*.macs.h`, `*.offsets.h`) stay in `<source_dir>/gen/`.
|
||||
--- 2. `gdb_tape_atoms_runtime.gdb`: Post-link opt-in (`ctx.flags.gdb_runtime`),
|
||||
--- so the gdb wrapper script + the generated runtime script share the same canonical location.
|
||||
--- Triggered by `--post-link` or `--gdb-runtime`.
|
||||
---
|
||||
--- Output forma (sourcemap.txt form):
|
||||
--- ```
|
||||
--- # FORMAT_VERSION 1
|
||||
--- # auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT
|
||||
--- ATOM <name> "<abs-source-path>" <total_words>
|
||||
--- WORD 0 LINE 49 TEXT load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||
--- WORD 1 LINE 49 TEXT load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||
--- ... (one WORD line per .word emitted by the atom body) ...
|
||||
--- ENDATOM
|
||||
--- ATOM <next-name> "<abs-source-path>" <total_words>
|
||||
--- ...
|
||||
--- ENDATOM
|
||||
--- ```
|
||||
--- Marker records are zero-width in `atom.paths.items`, so they emit no WORD rows in the dense word view.
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Module-scope requires + package.path setup
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source`
|
||||
-- (works both standalone + when require'd). `duffle_paths.lua` sets package.path then returns `require("duffle")`
|
||||
-- at the bottom, so the dofile value IS the duffle module.
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" ---@type string
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") ---@type DuffleExport
|
||||
local elf_dwarf = require("elf_dwarf") ---@type ElfDwarfMod
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Constants
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Format version emitted as the first line. Bump + add a migration test if the format changes;
|
||||
-- the gdb runtime loader rejects mismatches (E2).
|
||||
local FORMAT_VERSION = 1 ---@type integer
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Type declarations
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- @class AtomSourceMapCtx
|
||||
--- @field shared PassShared
|
||||
--- @field out_root string
|
||||
--- @field flags PassFlags
|
||||
--- @field project_root string|nil
|
||||
|
||||
--- @class WordMapEntry
|
||||
--- @field pos integer
|
||||
--- @field line integer
|
||||
--- @field text string
|
||||
--- @field body_line integer
|
||||
--- @field gpr_keys string[]|nil
|
||||
--- @field invocation InvocationRecord|nil
|
||||
|
||||
--- @class NmAddr
|
||||
--- @field [1] integer -- st_value
|
||||
--- @field [2] integer -- st_size
|
||||
|
||||
--- @class GdbAtomRecord
|
||||
--- @field idx integer|nil
|
||||
--- @field name string
|
||||
--- @field src_path string
|
||||
--- @field file_base string
|
||||
--- @field addr integer
|
||||
--- @field size_bytes integer
|
||||
--- @field words integer
|
||||
--- @field entries WordMapEntry[]
|
||||
|
||||
--- @class ElfDwarfMod
|
||||
--- @field read_nm fun(elf_path: Path): table<string, NmAddr>
|
||||
|
||||
--- @class AtomSourceMapPass
|
||||
--- @field render_source_map fun(src: SourceFile): string
|
||||
--- @field render_provenance fun(src: SourceFile, wc: WordCounts): string
|
||||
--- @field render_atom_source_map fun(atom: AtomEntry): string
|
||||
--- @field render_atom_provenance fun(atom: AtomEntry, wc: WordCounts, rel_path: string): string
|
||||
--- @field run fun(ctx: PassCtx): PassResult
|
||||
|
||||
--- @class AtomEntry
|
||||
--- @field paths AtomPaths|nil
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Atom-path renderers
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Join word boundaries (from `items`) to per-word call text + source lines (from `word_events`).
|
||||
--- @param atom AtomEntry
|
||||
--- @return WordMapEntry[]
|
||||
--- @return integer
|
||||
local function canonical_word_entries(atom)
|
||||
local paths = atom.paths or {} ---@type AtomPaths
|
||||
local events = paths.word_events or {} ---@type WordEvent[]
|
||||
local word_items = {} ---@type EmissionItem[]
|
||||
for _, item in ipairs(paths.items or {}) do ---@type integer, EmissionItem
|
||||
if item.kind == "word" then word_items[#word_items + 1] = item end
|
||||
end
|
||||
|
||||
local entries = {} ---@type WordMapEntry[]
|
||||
for index, event in ipairs(events) do ---@type integer, WordEvent
|
||||
local item = word_items[index] or {} ---@type EmissionItem
|
||||
entries[#entries + 1] = {
|
||||
pos = event.i or (index - 1),
|
||||
line = event.call_line or item.line or 0,
|
||||
text = event.call_text or item.call_text or "",
|
||||
body_line = event.body_line or item.body_line or item.line or 0,
|
||||
gpr_keys = event.gpr_keys,
|
||||
invocation = (event.outermost_invocation_id
|
||||
and paths.invocations
|
||||
and paths.invocations[event.outermost_invocation_id]) or nil,
|
||||
}
|
||||
end
|
||||
return entries, #events
|
||||
end
|
||||
|
||||
--- Render one atom's provenance stanza. Format 1 line shapes:
|
||||
--- `WORD N CALL <src-path>:<src-line> MACRO <name> "<def-path>:<def-line>" BODY <line>` (component invocation)
|
||||
--- `WORD N CALL <src-path>:<src-line> RAW` (raw `.word` outside any mac_* component)
|
||||
--- Component identity comes from the outermost invocation record; the count-table lookup confirms the component was declared in `corpus.word_counts`
|
||||
--- (populated by word_count_eval + components passes).
|
||||
--- @param src SourceFile
|
||||
--- @param atom AtomEntry
|
||||
--- @param wc WordCounts
|
||||
--- @return string[]
|
||||
--- @return integer
|
||||
local function emit_provenance_stanza(src, atom, wc)
|
||||
local lines = {} ---@type string[]
|
||||
local rel_path = src.path:gsub("\\\\", "/") ---@type string
|
||||
local entries, total = canonical_word_entries(atom) ---@type WordMapEntry[], integer
|
||||
lines[#lines + 1] = string.format('ATOM %s "%s" 0', atom.raw_name or atom.name, rel_path)
|
||||
|
||||
for _, entry in ipairs(entries) do ---@type integer, WordMapEntry
|
||||
local inv = entry.invocation ---@type InvocationRecord|nil
|
||||
local macro_count = inv and wc["mac_" .. inv.component_name] ---@type integer|nil
|
||||
if inv and macro_count ~= nil then
|
||||
lines[#lines + 1] = string.format('WORD %d CALL %s:%d MACRO %s "%s:%d" BODY %d'
|
||||
, entry.pos, rel_path, entry.line, inv.component_name
|
||||
, inv.def_path or "", inv.def_line or 0, entry.body_line)
|
||||
else
|
||||
lines[#lines + 1] = string.format("WORD %d CALL %s:%d RAW", entry.pos, rel_path, entry.line)
|
||||
end
|
||||
end
|
||||
|
||||
lines[1] = lines[1]:gsub(" 0$", " " .. tostring(total))
|
||||
lines[#lines + 1] = "ENDATOM"
|
||||
return lines, total
|
||||
end
|
||||
|
||||
--- Render the full provenance file content for one source.
|
||||
--- @param src SourceFile
|
||||
--- @param wc WordCounts
|
||||
--- @return string
|
||||
local function render_provenance(src, wc)
|
||||
local lines = {} ---@type string[]
|
||||
lines[#lines + 1] = "# FORMAT_VERSION 1"
|
||||
lines[#lines + 1] = "# auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT"
|
||||
lines[#lines + 1] = "# Per-.word provenance: maps each emitted .word to its call site (atom body"
|
||||
lines[#lines + 1] = "# file:line) and, when the word was emitted by a `mac_X(...)` component invocation,"
|
||||
lines[#lines + 1] = "# the component's definition file:line + the per-word BODY line. Used by"
|
||||
lines[#lines + 1] = "# dwarf_injection to synthesize DW_TAG_inlined_subroutine instances + per-word"
|
||||
lines[#lines + 1] = "# line program rows for native source-level step into component bodies."
|
||||
|
||||
--- @param atom AtomEntry
|
||||
--- @return nil
|
||||
local function append(atom)
|
||||
local stanza = emit_provenance_stanza(src, atom, wc) ---@type string[]
|
||||
for _, line in ipairs(stanza) do lines[#lines + 1] = line end ---@type integer, string
|
||||
end
|
||||
for _, atom in ipairs(src.scan.atoms or {}) do ---@type integer, AtomEntry
|
||||
if atom.paths then append(atom) end
|
||||
end
|
||||
for _, atom in ipairs(src.scan.raw_atoms or {}) do ---@type integer, AtomEntry
|
||||
if atom.paths then append(atom) end
|
||||
end
|
||||
|
||||
return table.concat(lines, "\n") .. "\n"
|
||||
end
|
||||
|
||||
--- Render one atom's stanza for the sourcemap.txt form (ATOM header line, N WORD lines, ENDATOM marker).
|
||||
--- Returns (lines, total_words).
|
||||
--- @param src SourceFile
|
||||
--- @param atom AtomEntry
|
||||
--- @return string[]
|
||||
--- @return integer
|
||||
local function emit_atom_stanza(src, atom)
|
||||
local lines = {} ---@type string[]
|
||||
local rel_path = src.path:gsub("\\\\", "/") ---@type string
|
||||
local entries, total = canonical_word_entries(atom) ---@type WordMapEntry[], integer
|
||||
|
||||
lines[#lines + 1] = string.format('ATOM %s "%s" 0', atom.raw_name or atom.name, rel_path)
|
||||
for _, entry in ipairs(entries) do ---@type integer, WordMapEntry
|
||||
lines[#lines + 1] = string.format("WORD %d LINE %d TEXT %s",
|
||||
entry.pos, entry.line, entry.text)
|
||||
end
|
||||
lines[1] = lines[1]:gsub(" 0$", " " .. tostring(total))
|
||||
lines[#lines + 1] = "ENDATOM"
|
||||
return lines, total
|
||||
end
|
||||
|
||||
--- Render the full source map file content for one source (one .atoms.sourcemap.txt per source). Mirrors offsets.lua's
|
||||
--- `project_atoms` shape: scan.atoms + scan.raw_atoms, no kind filter.
|
||||
--- @param src SourceFile
|
||||
--- @return string
|
||||
local function render_source_map(src)
|
||||
local lines = {} ---@type string[]
|
||||
lines[#lines + 1] = "# FORMAT_VERSION " .. FORMAT_VERSION
|
||||
lines[#lines + 1] = "# auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT"
|
||||
|
||||
--- @param atom AtomEntry
|
||||
--- @return nil
|
||||
local function append(atom)
|
||||
local stanza = emit_atom_stanza(src, atom) ---@type string[]
|
||||
for _, line in ipairs(stanza) do lines[#lines + 1] = line end ---@type integer, string
|
||||
end
|
||||
for _, atom in ipairs(src.scan.atoms or {}) do ---@type integer, AtomEntry
|
||||
if atom.paths then append(atom) end
|
||||
end
|
||||
for _, atom in ipairs(src.scan.raw_atoms or {}) do ---@type integer, AtomEntry
|
||||
if atom.paths then append(atom) end
|
||||
end
|
||||
|
||||
return table.concat(lines, "\n") .. "\n"
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- gdb-runtime emission (post-link, addresses via nm)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Escape a string for embedding in a gdb `set $var = "..."` literal.
|
||||
--- gdb uses C-style escaping; we escape `\` and `"` (newlines were flattened earlier).
|
||||
--- @param s string
|
||||
--- @return string
|
||||
local function gdb_escape(s)
|
||||
return (s:gsub("\\", "\\\\"):gsub('"', '\\"'))
|
||||
end
|
||||
|
||||
--- Build the list of atoms with addresses + word entries. Shared helper for the gdb-runtime file emission.
|
||||
--- @param ctx PassCtx
|
||||
--- @return GdbAtomRecord[]
|
||||
local function build_atom_table(ctx)
|
||||
local addrs = elf_dwarf.read_nm(ctx.flags.elf_path) ---@type table<string, NmAddr>
|
||||
local corpus = ctx.shared and ctx.shared.corpus ---@type Corpus|nil
|
||||
local matched = {} ---@type GdbAtomRecord[]
|
||||
|
||||
for _, src in ipairs(corpus.source_order or {}) do ---@type integer, SourceFile
|
||||
local file_base = src.path:match("([^/\\\\]+)$") or src.path ---@type string
|
||||
--- @param atom AtomEntry
|
||||
--- @return nil
|
||||
local function append(atom)
|
||||
if not atom.paths then return end
|
||||
local name = atom.raw_name or atom.name ---@type string
|
||||
local info = addrs[name] ---@type NmAddr|nil
|
||||
if not info then return end
|
||||
local entries, total = canonical_word_entries(atom) ---@type WordMapEntry[], integer
|
||||
matched[#matched + 1] = {
|
||||
name = name,
|
||||
src_path = src.path,
|
||||
file_base = file_base,
|
||||
addr = info[1],
|
||||
size_bytes = info[2],
|
||||
words = total,
|
||||
entries = entries,
|
||||
}
|
||||
end
|
||||
for _, atom in ipairs((src.scan or {}).atoms or {}) do append(atom) end ---@type integer, AtomEntry
|
||||
for _, atom in ipairs((src.scan or {}).raw_atoms or {}) do append(atom) end ---@type integer, AtomEntry
|
||||
end
|
||||
|
||||
-- Deterministic order: sort by address (matches `nm` output ordering).
|
||||
--- @param a GdbAtomRecord
|
||||
--- @param b GdbAtomRecord
|
||||
--- @return boolean
|
||||
table.sort(matched, function(a, b) return a.addr < b.addr end)
|
||||
for i, a in ipairs(matched) do a.idx = i - 1 end ---@type integer, GdbAtomRecord
|
||||
return matched
|
||||
end
|
||||
|
||||
--- Append the 9 gdb command definitions to `lines`. Pure gdb scripting — addresses come from `nm`,
|
||||
--- the convenience vars set in `emit_gdb_runtime` provide printf args, and
|
||||
--- each command is a static sequence of `printf` / `tbreak` / `if ... end` blocks.
|
||||
--- The Lua pass emits N atoms' worth of lines; runtime iteration is gdb's job.
|
||||
---
|
||||
--- Why hardcoded per-atom: gdb's `$` substitution doesn't concat inside var names — `$__atom_name_$__i` in a `while`
|
||||
--- loop resolves to one literal identifier, not `name_i`. Compile-time emission is the only path.
|
||||
--- @param lines string[]
|
||||
--- @param matched GdbAtomRecord[]
|
||||
--- @return nil
|
||||
local function append_gdb_commands(lines, matched)
|
||||
-- ── tape_atoms ──
|
||||
-- Hardcoded one printf per atom. No loop.
|
||||
lines[#lines + 1] = "define tape_atoms"
|
||||
for _, a in ipairs(matched) do ---@type integer, GdbAtomRecord
|
||||
-- gdb 12.1 quirk: literals in printf args require an attached target.
|
||||
-- Use the per-atom convenience vars set above as printf args.
|
||||
lines[#lines + 1] = string.format(' printf " %%-32s @ 0x%%08x %%4d words\\n", $__atom_name_%d, $__atom_addr_%d, $__atom_words_%d',
|
||||
a.idx, a.idx, a.idx)
|
||||
end
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = "document tape_atoms"
|
||||
lines[#lines + 1] = " List every tape atom symbol in the loaded ELF with .rodata addr + word count."
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
-- ── break_atom (generic) + per-atom break_atom_X ──
|
||||
lines[#lines + 1] = "define break_atom"
|
||||
lines[#lines + 1] = ' echo "Usage: break_atom_<exact_name> (pick from the list below)"'
|
||||
for _, a in ipairs(matched) do ---@type integer, GdbAtomRecord
|
||||
lines[#lines + 1] = string.format(' printf " break_atom_%%-32s\\n", $__atom_name_%d', a.idx)
|
||||
end
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = "document break_atom"
|
||||
lines[#lines + 1] = " Generic help: lists the per-atom break_atom_<name> commands."
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
for _, a in ipairs(matched) do ---@type integer, GdbAtomRecord
|
||||
lines[#lines + 1] = string.format("define break_atom_%s", a.name)
|
||||
lines[#lines + 1] = string.format(" break *$__atom_addr_%d", a.idx)
|
||||
lines[#lines + 1] = string.format(' printf " Breakpoint set at %s (0x%%08x)\\n", $__atom_addr_%d', a.name, a.idx)
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = string.format("document break_atom_%s", a.name)
|
||||
lines[#lines + 1] = string.format(" Set a breakpoint at %s.", a.name)
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = ""
|
||||
end
|
||||
|
||||
-- ── step_atom / next_atom ──
|
||||
-- Hardcoded one tbreak per atom. No loop.
|
||||
lines[#lines + 1] = "define step_atom"
|
||||
for _, a in ipairs(matched) do ---@type integer, GdbAtomRecord
|
||||
lines[#lines + 1] = string.format(" tbreak *$__atom_addr_%d", a.idx)
|
||||
end
|
||||
lines[#lines + 1] = " continue"
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = "document step_atom"
|
||||
lines[#lines + 1] = " Set one-shot BPs at every atom + continue. Stops at the next atom boundary."
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
lines[#lines + 1] = "define next_atom"
|
||||
lines[#lines + 1] = " step_atom"
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = "document next_atom"
|
||||
lines[#lines + 1] = " Alias for step_atom."
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
-- ── where_in_atom ──
|
||||
-- Hardcoded one outer-if per atom; inside, one inner-if per WORD entry.
|
||||
lines[#lines + 1] = "define where_in_atom"
|
||||
lines[#lines + 1] = " set $__pc = (unsigned int)$pc"
|
||||
lines[#lines + 1] = " set $__matched = 0"
|
||||
for _, a in ipairs(matched) do ---@type integer, GdbAtomRecord
|
||||
-- Precompute end_addr (gdb 12.1's expression evaluator chokes on `addr + words*4`).
|
||||
lines[#lines + 1] = string.format(" set $__end_%d = $__atom_addr_%d + $__atom_words_%d * 4", a.idx, a.idx, a.idx)
|
||||
lines[#lines + 1] = string.format(" if $__pc >= $__atom_addr_%d && $__pc < $__end_%d", a.idx, a.idx)
|
||||
lines[#lines + 1] = string.format(' printf "atom: %%s\\n", $__atom_name_%d', a.idx)
|
||||
lines[#lines + 1] = ' printf "addr: 0x%08x\\n", $__pc'
|
||||
lines[#lines + 1] = string.format(" set $__word = ($__pc - $__atom_addr_%d) / 4", a.idx)
|
||||
lines[#lines + 1] = string.format(' printf "word: %%d/%%d\\n", $__word, $__atom_words_%d', a.idx)
|
||||
-- One inner-if per WORD entry. Each word's line + text hardcoded.
|
||||
for _, we in ipairs(a.entries) do ---@type integer, WordMapEntry
|
||||
lines[#lines + 1] = string.format(" if $__word == %d", we.pos)
|
||||
-- Escape TEXT for printf format string.
|
||||
local escaped_text = we.text:gsub("%%", "%%%%"):gsub('"', '\\"') ---@type string
|
||||
lines[#lines + 1] = string.format(' printf "source: %%s:%%d %%s\\n", $__atom_file_%d, %d, "%s"', a.idx, we.line, escaped_text)
|
||||
lines[#lines + 1] = " end"
|
||||
end
|
||||
-- Fallback for words beyond the source map (shouldn't happen if nm matches).
|
||||
local max_word = 0 ---@type integer
|
||||
if #a.entries > 0 then max_word = a.entries[#a.entries].pos end
|
||||
lines[#lines + 1] = string.format(' if $__word > %d', max_word)
|
||||
lines[#lines + 1] = ' printf "source: (no source-map entry for word %%d; map may be stale)\\n", $__word'
|
||||
lines[#lines + 1] = " end"
|
||||
lines[#lines + 1] = " set $__matched = 1"
|
||||
lines[#lines + 1] = " end"
|
||||
end
|
||||
lines[#lines + 1] = " if !$__matched"
|
||||
lines[#lines + 1] = ' echo PC is not inside any known atom (in .text or unmapped region).'
|
||||
lines[#lines + 1] = " end"
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = "document where_in_atom"
|
||||
lines[#lines + 1] = " Report current atom name, .rodata addr, word offset, and source line."
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
-- ── stepi_inside_atom ──
|
||||
-- Hardcoded one if-containment-check per atom (no loop).
|
||||
-- Precompute end_addr in Lua so we don't ask gdb to evaluate `addr + words*4` inside the if condition
|
||||
-- (gdb 12.1's expression evaluator chokes on the `*` and emits a misleading 'function malloc' error in some gdb builds).
|
||||
lines[#lines + 1] = "define stepi_inside_atom"
|
||||
lines[#lines + 1] = " set $__in_atom = 0"
|
||||
lines[#lines + 1] = " set $__did_step = 0"
|
||||
lines[#lines + 1] = " set $__pc = (unsigned int)$pc"
|
||||
for _, a in ipairs(matched) do ---@type integer, GdbAtomRecord
|
||||
-- Precompute end_addr in the convenience var (single expression gdb handles).
|
||||
lines[#lines + 1] = string.format(" set $__end_%d = $__atom_addr_%d + $__atom_words_%d * 4", a.idx, a.idx, a.idx)
|
||||
lines[#lines + 1] = string.format(" if $__pc >= $__atom_addr_%d && $__pc < $__end_%d", a.idx, a.idx)
|
||||
lines[#lines + 1] = " set $__in_atom = 1"
|
||||
lines[#lines + 1] = " stepi"
|
||||
lines[#lines + 1] = " set $__did_step = 1"
|
||||
lines[#lines + 1] = " end"
|
||||
end
|
||||
lines[#lines + 1] = " if !$__did_step"
|
||||
lines[#lines + 1] = ' echo [gdb_tape_atoms] stepi_inside_atom: PC is not inside any atom; refusing to step.'
|
||||
lines[#lines + 1] = " end"
|
||||
lines[#lines + 1] = " where_in_atom"
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = "document stepi_inside_atom"
|
||||
lines[#lines + 1] = " One MIPS-instruction step, then where_in_atom. The step-and-see-source-line workflow."
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
-- ── wave_ctx ──
|
||||
lines[#lines + 1] = "define wave_ctx"
|
||||
lines[#lines + 1] = ' printf "$t4 = R_FaceCursor 0x%08x\\n", $t4'
|
||||
lines[#lines + 1] = ' printf "$t5 = R_VertBase 0x%08x\\n", $t5'
|
||||
lines[#lines + 1] = ' printf "$t6 = R_OtBase 0x%08x\\n", $t6'
|
||||
lines[#lines + 1] = ' printf "$t7 = R_PrimCursor 0x%08x\\n", $t7'
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = "document wave_ctx"
|
||||
lines[#lines + 1] = " Pretty-print the 4 wave-context GPRs ($t4=R_FaceCursor, $t5=R_VertBase, $t6=R_OtBase, $t7=R_PrimCursor). Requires target attached."
|
||||
lines[#lines + 1] = "end"
|
||||
end
|
||||
|
||||
--- Emit the gdb-runtime file (post-link). Pure gdb scripting — addresses come from `mipsel-none-elf-nm -S`, get embedded
|
||||
--- in `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`, and load via `set $var = ...` + `define ... end` blocks at gdb source-time.
|
||||
--- @param ctx PassCtx
|
||||
--- @return nil
|
||||
local function emit_gdb_runtime(ctx)
|
||||
if not (ctx.flags and ctx.flags.gdb_runtime) then return end
|
||||
local elf_path = ctx.flags.elf_path ---@type string|nil
|
||||
if not elf_path or elf_path == "" then
|
||||
io.stderr:write("[atoms_source_map] --gdb-runtime requires --elf <elf>\n")
|
||||
return
|
||||
end
|
||||
if lfs.attributes(elf_path, "mode") ~= "file" then
|
||||
io.stderr:write(string.format(
|
||||
"[atoms_source_map] --gdb-runtime: ELF not found at %s\n", elf_path))
|
||||
return
|
||||
end
|
||||
|
||||
local matched = build_atom_table(ctx) ---@type GdbAtomRecord[]
|
||||
if #matched == 0 then
|
||||
io.stderr:write("[atoms_source_map] --gdb-runtime: no atoms matched against nm symbols (stale scan?).\n")
|
||||
return
|
||||
end
|
||||
|
||||
local lines = {} ---@type string[]
|
||||
lines[#lines + 1] = "# Auto-generated by ps1_meta.lua (passes/atoms_source_map.lua)"
|
||||
lines[#lines + 1] = "# DO NOT EDIT — re-run ps1_meta.lua --atoms-source-map --gdb-runtime to regenerate"
|
||||
lines[#lines + 1] = "# Sourced by scripts/gdb/gdb_tape_atoms.gdb (the wrapper)."
|
||||
lines[#lines + 1] = "# Pure gdb scripting — no Python, no Tcl, no Guile required."
|
||||
lines[#lines + 1] = "# Commands are FULLY HARDCODED per-atom because gdb doesn't do nested"
|
||||
lines[#lines + 1] = "# `$` substitution in var names (`$foo_$i` is one literal identifier)."
|
||||
lines[#lines + 1] = "# Per-atom convenience vars ($__atom_name_<i> etc.) are set so gdb's"
|
||||
lines[#lines + 1] = "# `printf` has valid expression args (gdb 12.1 quirks: literals in"
|
||||
lines[#lines + 1] = "# printf args require an attached target; convenience-var args do not)."
|
||||
lines[#lines + 1] = string.format("# %d atoms from ELF: %s", #matched, elf_path)
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
-- Format version + count + ELF path (the latter is referenced by the load-line).
|
||||
lines[#lines + 1] = "set $__atom_format_version = " .. FORMAT_VERSION
|
||||
lines[#lines + 1] = string.format("set $__atom_count = %d", #matched)
|
||||
lines[#lines + 1] = string.format('set $__elf_path = "%s"', gdb_escape(elf_path))
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
-- Per-atom convenience vars (used as printf args; literals aren't accepted
|
||||
-- without an attached target on gdb 12.1).
|
||||
for _, a in ipairs(matched) do ---@type integer, GdbAtomRecord
|
||||
lines[#lines + 1] = string.format('set $__atom_name_%d = "%s"', a.idx, gdb_escape(a.name))
|
||||
lines[#lines + 1] = string.format("set $__atom_addr_%d = 0x%x", a.idx, a.addr)
|
||||
lines[#lines + 1] = string.format("set $__atom_words_%d = %d", a.idx, a.words)
|
||||
lines[#lines + 1] = string.format('set $__atom_file_%d = "%s"', a.idx, gdb_escape(a.file_base))
|
||||
end
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
-- The 9 commands (each `define ... end` overrides the wrapper's stub).
|
||||
lines[#lines + 1] = "# ── 9 user commands (overrides wrapper stubs) ──"
|
||||
append_gdb_commands(lines, matched)
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
-- Confirmation line for the source operator.
|
||||
lines[#lines + 1] = 'printf "[gdb_tape_atoms] runtime loaded %d atoms from %s\\n", $__atom_count, $__elf_path'
|
||||
|
||||
local out_path ---@type string
|
||||
-- Move out of `<out_root>/gdb_tape_atoms_runtime.gdb` to `<out_root>/../gdb_tape_atoms_runtime.gdb` when the conventional `<out_root>` is `<build>/gen`
|
||||
-- (any equivalent spelling — relative, absolute backslash, absolute forward-slash, trailing-separator variants).
|
||||
-- This puts the gdb runtime alongside the ELF at `build/` rather than under the report subdir.
|
||||
--- @param p string
|
||||
--- @return boolean
|
||||
local function ends_with_gen_dir(p)
|
||||
if type(p) ~= "string" then return false end
|
||||
return p:match("[/\\]gen[/\\]?$") ~= nil or p == "build/gen" or p == "build\\gen"
|
||||
end
|
||||
if ends_with_gen_dir(ctx.out_root) then
|
||||
-- Strip the trailing `/gen` segment, then write the runtime script under `build/`.
|
||||
-- e.g. "C:/projects/Pikuma/ps1/build/gen" -> "C:/projects/Pikuma/ps1/build".
|
||||
local parent = ctx.out_root:gsub("[/\\]gen[/\\]?$", "") ---@type string
|
||||
out_path = parent .. "/gdb_tape_atoms_runtime.gdb"
|
||||
else
|
||||
out_path = ctx.out_root .. "/gdb_tape_atoms_runtime.gdb"
|
||||
end
|
||||
duffle.ensure_dir(duffle.dirname(out_path))
|
||||
duffle.write_file_lf(out_path, table.concat(lines, "\n") .. "\n")
|
||||
-- io.stderr:write(string.format("[atoms_source_map] wrote %s (%d atoms)\n", out_path, #matched))
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- M — module exports
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
local M = {} ---@type AtomSourceMapPass
|
||||
|
||||
-- Expose the pure render functions so `report.lua` and the focused tests can call them directly without triggering the file-emit path.
|
||||
M.render_source_map = render_source_map
|
||||
M.render_provenance = render_provenance
|
||||
|
||||
--- Render ONE atom's sourcemap stanza.
|
||||
--- @param atom AtomEntry
|
||||
--- @return string
|
||||
function M.render_atom_source_map(atom)
|
||||
assert(type(atom) == "table", "render_atom_source_map: atom must be a table")
|
||||
assert(type(atom.paths) == "table", "render_atom_source_map: atom.paths must be a table")
|
||||
local entries, total = canonical_word_entries(atom) ---@type WordMapEntry[], integer
|
||||
local lines = {} ---@type string[]
|
||||
lines[#lines + 1] = string.format("ATOM %s %d", (atom.raw_name or atom.name), total)
|
||||
for _, entry in ipairs(entries) do ---@type integer, WordMapEntry
|
||||
local word_line = string.format("WORD %d LINE %d TEXT %s", ---@type string
|
||||
entry.pos, entry.line, entry.text)
|
||||
local keys = {} ---@type string[]
|
||||
for pos = 1, 16 do ---@type integer
|
||||
local k = entry.gpr_keys and entry.gpr_keys[pos] ---@type string|nil
|
||||
if type(k) == "string" and k:sub(1, 7) == "reguse:" then
|
||||
keys[#keys + 1] = k
|
||||
end
|
||||
end
|
||||
if #keys > 0 then
|
||||
word_line = word_line .. " KEYS " .. table.concat(keys, ",")
|
||||
end
|
||||
lines[#lines + 1] = word_line
|
||||
end
|
||||
lines[#lines + 1] = "ENDATOM"
|
||||
return table.concat(lines, "\n") .. "\n"
|
||||
end
|
||||
|
||||
--- Render ONE atom's provenance stanza — no per-file format header, no enumeration of other atoms.
|
||||
---
|
||||
--- `rel_path` is the source path (forward-slashes) embedded in every `CALL` line.
|
||||
--- The .md caller (report.lua) is expected to derive this once per `## <source>` heading and pass it down for each atom in that source.
|
||||
--- @param atom AtomEntry
|
||||
--- @param wc WordCounts
|
||||
--- @param rel_path string
|
||||
--- @return string
|
||||
function M.render_atom_provenance(atom, wc, rel_path)
|
||||
assert(type(atom) == "table", "render_atom_provenance: atom must be a table")
|
||||
assert(type(atom.paths) == "table", "render_atom_provenance: atom.paths must be a table")
|
||||
assert(type(rel_path) == "string", "render_atom_provenance: rel_path must be a string")
|
||||
local entries, total = canonical_word_entries(atom) ---@type WordMapEntry[], integer
|
||||
local lines = {} ---@type string[]
|
||||
lines[#lines + 1] = string.format("ATOM %s %d", (atom.raw_name or atom.name), total)
|
||||
for _, entry in ipairs(entries) do ---@type integer, WordMapEntry
|
||||
local inv = entry.invocation ---@type InvocationRecord|nil
|
||||
local macro_count = inv and wc and wc["mac_" .. inv.component_name] ---@type integer|nil
|
||||
if inv and macro_count ~= nil then
|
||||
lines[#lines + 1] = string.format('WORD %d CALL %s:%d MACRO %s "%s:%d" BODY %d'
|
||||
, entry.pos, rel_path, entry.line, inv.component_name, inv.def_path or "", inv.def_line or 0, entry.body_line)
|
||||
else
|
||||
lines[#lines + 1] = string.format(
|
||||
"WORD %d CALL %s:%d RAW", entry.pos, rel_path, entry.line)
|
||||
end
|
||||
end
|
||||
return table.concat(lines, "\n") .. "\n"
|
||||
end
|
||||
|
||||
--- Pass entry. For each source that declares at least one tape atom,
|
||||
--- emit two files in `<out_root>/`: `<basename>.atoms.sourcemap.txt` (per-word call-site map) and `<basename>.atoms.provenance.txt`
|
||||
--- (per-word definition + body line, resolved via the outermost `mac_X(...)` invocation).
|
||||
--- When `ctx.flags.gdb_runtime` is true and `ctx.flags.elf_path` exists, also emit the post-link gdb script `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`.
|
||||
--- @param ctx PassCtx
|
||||
--- @return PassResult
|
||||
function M.run(ctx)
|
||||
local outputs = {} ---@type PassOutputEntry[]
|
||||
local errors = {} ---@type Finding[]
|
||||
local warnings = {} ---@type Finding[]
|
||||
|
||||
local corpus = ctx.shared and ctx.shared.corpus ---@type Corpus|nil
|
||||
if type(corpus) ~= "table" or type(corpus.source_order) ~= "table" then
|
||||
error("atoms_source_map.run requires ctx.shared.corpus.source_order (canonical corpus).", 0)
|
||||
end
|
||||
|
||||
-- Word counts come from `corpus.word_counts` (populated by word_count_eval + components passes).
|
||||
local wc = corpus.word_counts or {} ---@type WordCounts
|
||||
if not next(wc) then
|
||||
warnings[#warnings + 1] = {
|
||||
line = 0,
|
||||
msg = "atoms_source_map: corpus.word_counts is empty; the word-counts + components passes may not have populated it. Check the PASSES dep edges.",
|
||||
}
|
||||
end
|
||||
|
||||
-- atoms.sourcemap.txt + atoms.provenance.txt content moved to report.lua via `<module>.atoms.md` markdown file.
|
||||
-- This pass emits only the post-link gdb_runtime artifact (see emit_gdb_runtime below).
|
||||
|
||||
-- Optionally emit the gdb-runtime form (post-link, one file per build).
|
||||
if ctx.flags and ctx.flags.gdb_runtime then
|
||||
emit_gdb_runtime(ctx)
|
||||
end
|
||||
|
||||
return { outputs = outputs, errors = errors, warnings = warnings }
|
||||
end
|
||||
|
||||
return M
|
||||
@@ -0,0 +1,366 @@
|
||||
--- passes/auto_reg.lua — Per-phase automatic GPR allocator + gen/auto_reg.h emitter.
|
||||
---
|
||||
--- Reads the per-source + corpus-level `atom_auto_regs` + `phase_auto_regs` registries populated by `passes/scan_source.lua`.
|
||||
--- Runs a deterministic first-fit allocator in the `R_T0..R_T7 + R_V0..R_V1` pool (10 physical GPRs).
|
||||
--- Emits one `#define R_<Sym>_Code R_Tn_Code` per marker into per-directory `gen/auto_reg.h`.
|
||||
---
|
||||
--- User-pinned GPRs : The corpus's `register_alias_registry` is consulted to exclude GPRs the user has pinned via
|
||||
--- `atom_reg` + `_Code` defs (e.g. carriers like `R_ResolveScratch = R_T4 atom_reg`).
|
||||
--- These GPRs are unavailable to EVERY atom's source pool.
|
||||
--- Carriers are preserved across atoms by context discipline and must never be reallocated.
|
||||
--- Per-atom body parsing also catches alias references (R_<Alias>) and hardcoded R_Tn references,
|
||||
--- so the user can write either `R_T4` or `R_ResolveScratch` in an atom body and the pass will exclude R_T4 from that atom's pool.
|
||||
---
|
||||
--- Conflict detection: If the user hardcodes `R_Tn` in an atom body that shares a phase with an auto-reg that picked `R_Tn`,
|
||||
--- emit `phase_register_clash` as an info finding (no build stop).
|
||||
--- Should be unreachable after the user-pinning + body-parsing fix above; kept as a defensive safety net.
|
||||
---
|
||||
--- Pool exhaustion: If a phase declares more `R_<Sym>` mappings than the 10-register pool can hold,
|
||||
--- emit `phase_register_pool_exhausted` as a build-stopping error.
|
||||
|
||||
--- @alias GprIdent string
|
||||
|
||||
--- @class GprAllocMap
|
||||
--- @field [string] GprIdent -- bag: auto-reg symbol -> physical GPR
|
||||
|
||||
--- @class AutoRegOutput
|
||||
--- @field auto_reg_h string
|
||||
|
||||
--- @class AutoRegResult
|
||||
--- @field outputs AutoRegOutput[]
|
||||
--- @field errors Finding[]
|
||||
--- @field warnings Finding[]
|
||||
|
||||
--- @class AutoRegPass
|
||||
--- @field run fun(ctx: PassCtx): AutoRegResult
|
||||
--- @field POOL GprIdent[]
|
||||
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" ---@type string
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") ---@type DuffleExport
|
||||
local isa = require("duffle_isa") ---@type DuffleIsa
|
||||
|
||||
--- ════════════════════════════════════════════════════════════════════════════
|
||||
--- THE GPR ALLOCATION POOL — what's allocatable, and (more importantly) WHY
|
||||
--- ════════════════════════════════════════════════════════════════════════════
|
||||
---
|
||||
--- The auto-reg pass picks physical GPRs for `atom_auto_reg(...)` / `phase_auto_reg(...)` markers.
|
||||
--- The 24-register pool covers R2-R25 (the user/atom allocatable surface):
|
||||
--- R_T0..R_T7, R_V0..R_V1, R_A0..A3, R_S0..S7, R_T8..T9.
|
||||
--- Excluded (and never added to the pool):
|
||||
--- R_0 (code 0) — Hardwired zero. Cannot be written.
|
||||
--- R_AT (code 1) — Assembler temporary. Reserved by the MIPS O32 ABI.
|
||||
--- R_A0..A3 — Explicitly omitted above even though their integer codes map to POOL entries;
|
||||
--- the pool-construction loop below only references the POOL string literals, never the integer codes, so they are NOT auto-allocated by default.
|
||||
--- (A0-A3 become available when the user adds them to POOL or hardcodes an R_A0 reference in the atom body.)
|
||||
--- R_K0/K1 (codes 26-27) — Kernel / interrupt handler reserves. Never touched by user code.
|
||||
--- R_GP/SP/FP/RA (codes 28-31) — R_SP/R_FP/R_RA are tape-runtime carriers between tape_enter and tape_exit; R_GP stays the host global pointer.
|
||||
---
|
||||
local POOL = {} ---@type GprIdent[]
|
||||
for _, row in ipairs(isa.GPR_ROLE) do ---@type integer, GprRole
|
||||
if row.pool then
|
||||
POOL[#POOL + 1] = row.name
|
||||
end
|
||||
end
|
||||
|
||||
-- Map from integer MIPS GPR code (the `code` field on AliasEntry) to the physical GPR ident in POOL.
|
||||
-- The standard MIPS O32 ABI register numbering matches mips.h's R_*_Code #defines (mips.h).
|
||||
-- Only the POOL entries matter for auto_reg — non-pool aliases
|
||||
-- (R_AT=1, R_A0..A3=4..7, R_T8=24, R_T9=25, R_K0/K1=26..27, R_GP/SP/FP/RA=28..31)
|
||||
-- are deliberately omitted — see the comment block above for the WHY of each exclusion.
|
||||
local INT_CODE_TO_POOL_GPR = { ---@type table<integer, GprIdent> -- bag: MIPS GPR code -> POOL ident
|
||||
[2] = "R_V0", [3] = "R_V1",
|
||||
[4] = "R_A0", [5] = "R_A1", [6] = "R_A2", [7] = "R_A3",
|
||||
[8] = "R_T0", [9] = "R_T1", [10] = "R_T2", [11] = "R_T3",
|
||||
[12] = "R_T4", [13] = "R_T5", [14] = "R_T6", [15] = "R_T7",
|
||||
[16] = "R_S0", [17] = "R_S1", [18] = "R_S2", [19] = "R_S3",
|
||||
[20] = "R_S4", [21] = "R_S5", [22] = "R_S6", [23] = "R_S7",
|
||||
[24] = "R_T8", [25] = "R_T9",
|
||||
}
|
||||
|
||||
-- Stable sort for deterministic allocation order.
|
||||
--- @param tbl table<string, string> -- bag: key set only; values unused
|
||||
--- @return string[]
|
||||
local function stable_sort_keys(tbl)
|
||||
local keys = {} ---@type string[]
|
||||
for k in pairs(tbl) do keys[#keys + 1] = k end ---@type string
|
||||
table.sort(keys)
|
||||
return keys
|
||||
end
|
||||
|
||||
-- Allocate one phase's auto-reg mappings.
|
||||
-- Returns (allocated_map, errors). On pool exhaustion, errors is populated and the function halts.
|
||||
--- @param phase_label string
|
||||
--- @param decls table<string, string> -- bag: auto-reg symbol -> decl payload
|
||||
--- @return GprAllocMap
|
||||
--- @return Finding[]
|
||||
local function allocate_phase(phase_label, decls)
|
||||
-- Deep-copy POOL into a fresh sequence table. The original `table.unpack and table.unpack(POOL) or { unpack(POOL) }`
|
||||
-- idiom wraps the unpacked values in a single inner table under LuaJIT 5.1 (`table.unpack` is nil; the `or` returns one value),
|
||||
-- which corrupts the pool into `{ {R_T0, R_T1, ...} }` — making `table.remove(pool, 1)` return the inner table on iteration.
|
||||
local pool = {} ---@type GprIdent[]
|
||||
for i = 1, #POOL do pool[i] = POOL[i] end ---@type integer
|
||||
local result = {} ---@type GprAllocMap
|
||||
local errors = {} ---@type Finding[]
|
||||
for _, sym in ipairs(stable_sort_keys(decls)) do ---@type integer, string
|
||||
local next_gpr = table.remove(pool, 1) ---@type GprIdent|nil
|
||||
if not next_gpr then
|
||||
errors[#errors + 1] = {
|
||||
line = 0,
|
||||
msg = string.format("phase_register_pool_exhausted: "
|
||||
.. "phase '%s' requested symbol '%s' but the pool has no remaining registers "
|
||||
.. "(max 24 per phase: R_T0..R_T7 + R_V0..R_V1 + R_A0..R_A3 + R_S0..R_S7 + R_T8..R_T9). Split the phase or use hardcoded GPRs."
|
||||
, phase_label, sym),
|
||||
}
|
||||
return result, errors
|
||||
end
|
||||
result[sym] = next_gpr
|
||||
end
|
||||
return result, errors
|
||||
end
|
||||
|
||||
-- Build two projections from corpus.register_alias_registry:
|
||||
-- user_pinned -- { [physical_gpr_ident] = true } -- GPRs unavailable to auto_reg globally (wave-context carriers, file-scope pinned aliases)
|
||||
-- alias_to_gpr -- { [alias_ident] = physical_gpr_ident } -- for body parsing
|
||||
-- Both projections are derived from the same set of entries: every AliasEntry in register_alias_registry has `has_atom_reg = true`
|
||||
-- (only those entries are added to the registry; see passes/scan_source.lua parse_enum_entry).
|
||||
-- Each entry's `code` is the integer MIPS GPR number (0..31); INT_CODE_TO_POOL_GPR translates it back to the physical GPR ident.
|
||||
-- Aliases whose `code` points to a non-POOL GPR (e.g. R_S0, R_T8, R_K1) are ignored —
|
||||
-- they don't affect the auto_reg pool, and they're already excluded from POOL above.
|
||||
--- @param corpus Corpus
|
||||
--- @return table<GprIdent, boolean>
|
||||
--- @return table<string, GprIdent>
|
||||
local function build_user_pins(corpus)
|
||||
local user_pinned = {} ---@type table<GprIdent, boolean> -- bag: pinned physical GPR -> true
|
||||
local alias_to_gpr = {} ---@type table<string, GprIdent> -- bag: alias ident -> physical GPR
|
||||
if not corpus.register_alias_registry then return user_pinned, alias_to_gpr end
|
||||
for alias_name, alias_entry in pairs(corpus.register_alias_registry) do ---@type string, AliasEntry
|
||||
if alias_entry.has_atom_reg and alias_entry.code then
|
||||
local gpr = INT_CODE_TO_POOL_GPR[alias_entry.code] ---@type GprIdent|nil
|
||||
if gpr then
|
||||
user_pinned[gpr] = true
|
||||
alias_to_gpr[alias_name] = gpr
|
||||
end
|
||||
end
|
||||
end
|
||||
return user_pinned, alias_to_gpr
|
||||
end
|
||||
|
||||
--- Find every physical GPR referenced in the atom body, via EITHER:
|
||||
--- (a) A hardcoded physical GPR ident (R_T\d+|R_V\d+|R_A\d+|R_S\d+) — the existing regex;
|
||||
--- (b) An alias ident (R_<Alias>) resolved via alias_to_gpr back to its physical GPR ident.
|
||||
--- Returns { [physical_gpr_ident] = count }.
|
||||
--- Clash-detection and source-pool-exclusion logic only needs the presence of each GPR (boolean test),
|
||||
--- but keeping count preserves the original find_hardcoded_rn shape so callers can switch without churn.
|
||||
--- The alias pattern is sorted lexicographically to keep the regex deterministic.
|
||||
--- @param body_text string
|
||||
--- @param alias_to_gpr table<string, GprIdent> -- bag: alias ident -> physical GPR
|
||||
--- @return table<GprIdent, integer>
|
||||
local function find_used_gprs(body_text, alias_to_gpr)
|
||||
local found = {} ---@type table<GprIdent, integer> -- bag: physical GPR -> hit count
|
||||
-- (a) Hardcoded physical GPRs (R_T0..R_T7, R_V0..R_V1, R_A0..R_A3, R_S0..R_S7).
|
||||
for gpr in body_text:gmatch("(R_T%d+|R_V%d+|R_A%d+|R_S%d+)") do ---@type GprIdent
|
||||
found[gpr] = (found[gpr] or 0) + 1
|
||||
end
|
||||
-- (b) Alias references (R_<Alias>) resolved to physical GPRs via the registry.
|
||||
-- Sorted by name so the regex is byte-stable across runs.
|
||||
if alias_to_gpr and next(alias_to_gpr) then
|
||||
local aliases = {} ---@type string[]
|
||||
for alias_name in pairs(alias_to_gpr) do ---@type string
|
||||
aliases[#aliases + 1] = alias_name
|
||||
end
|
||||
table.sort(aliases)
|
||||
local pattern = "(" .. table.concat(aliases, "|") .. ")" ---@type string
|
||||
for alias_name in body_text:gmatch(pattern) do ---@type string
|
||||
local gpr = alias_to_gpr[alias_name] ---@type GprIdent|nil
|
||||
if gpr and not found[gpr] then
|
||||
found[gpr] = 1
|
||||
end
|
||||
end
|
||||
end
|
||||
return found
|
||||
end
|
||||
|
||||
-- Emit one gen/auto_reg.h header per directory.
|
||||
--- @param out_dir string
|
||||
--- @param dir string
|
||||
--- @param sources SourceFile[]
|
||||
--- @param mappings GprAllocMap
|
||||
--- @return string|nil
|
||||
local function emit_auto_reg_h(out_dir, dir, sources, mappings)
|
||||
if not mappings or next(mappings) == nil then return end
|
||||
local out_path = out_dir .. "/" .. "auto_reg.h" ---@type string
|
||||
duffle.ensure_dir(out_dir)
|
||||
local lines = { ---@type string[]
|
||||
"#ifdef INTELLISENSE_DIRECTIVES",
|
||||
"#pragma once",
|
||||
"#endif",
|
||||
"// Auto-generated by ps1_meta.lua (passes/auto_reg.lua) — DO NOT EDIT",
|
||||
"// Directory: " .. dir:gsub("/", "\\"),
|
||||
}
|
||||
for _, src in ipairs(sources) do ---@type integer, SourceFile
|
||||
lines[#lines + 1] = "// source: " .. src.path
|
||||
end
|
||||
lines[#lines + 1] = "// Per-phase register allocations resolved by the lua pass."
|
||||
lines[#lines + 1] = "// R_<Sym>_Code = <chosen GPR's _Code constant> for every marker in this directory."
|
||||
lines[#lines + 1] = ""
|
||||
for _, sym in ipairs(stable_sort_keys(mappings)) do ---@type integer, string
|
||||
local gpr = mappings[sym] ---@type GprIdent
|
||||
local gpr_code = gpr .. "_Code" ---@type string
|
||||
lines[#lines + 1] = "#define " .. sym .. "_Code " .. gpr_code
|
||||
end
|
||||
lines[#lines + 1] = ""
|
||||
duffle.write_file_lf(out_path, table.concat(lines, "\n") .. "\n")
|
||||
print(" -> " .. out_path)
|
||||
return out_path
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Pass entry
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
local M = {} ---@type AutoRegPass
|
||||
|
||||
--- @param ctx PassCtx
|
||||
--- @return AutoRegResult
|
||||
function M.run(ctx)
|
||||
local outputs = {} ---@type AutoRegOutput[]
|
||||
local errors = {} ---@type Finding[]
|
||||
local warnings = {} ---@type Finding[]
|
||||
|
||||
local corpus = ctx.shared and ctx.shared.corpus ---@type Corpus|nil
|
||||
if type(corpus) ~= "table" then
|
||||
error("auto_reg.run requires ctx.shared.corpus", 0)
|
||||
end
|
||||
|
||||
-- 0. Build the user-pinned GPR exclusion set + alias-to-GPR resolution map.
|
||||
-- Wave-context carriers (e.g. `R_ResolveScratch = R_T4 atom_reg` in hello_camera.atom.c)
|
||||
-- MUST NOT be allocated to any auto-reg marker — they're preserved across atoms by the wave-context discipline.
|
||||
-- The corpus's register_alias_registry is the source of truth for these opt-in pins.
|
||||
-- Body references to those aliases (via alias_to_gpr) are also excluded on a per-atom basis in step 2 below.
|
||||
local user_pinned, alias_to_gpr = build_user_pins(corpus) ---@type table<GprIdent, boolean>, table<string, GprIdent>
|
||||
|
||||
-- 1. Allocate phase pools first (phase declarations take precedence over per-atom declarations).
|
||||
local phase_allocations = {} ---@type table<string, GprAllocMap> -- bag: phase_label -> alloc map
|
||||
for phase_label, decls in pairs(corpus.phase_auto_regs or {}) do ---@type string, table<string, string>
|
||||
local mapping, errs = allocate_phase(phase_label, decls) ---@type GprAllocMap, Finding[]
|
||||
for sym, gpr in pairs(mapping) do ---@type string, GprIdent
|
||||
phase_allocations[phase_label] = phase_allocations[phase_label] or {}
|
||||
phase_allocations[phase_label][sym] = gpr
|
||||
end
|
||||
for _, e in ipairs(errs) do ---@type integer, Finding
|
||||
errors[#errors + 1] = e
|
||||
end
|
||||
end
|
||||
|
||||
-- 2. Allocate per-atom auto-regs. If the atom scope matches a phase, reuse the phase pool.
|
||||
-- Otherwise, allocate a private pool for the atom.
|
||||
-- The phase membership is in `corpus.atom_phases[phase_label].atoms` (an array of atom names declared via `atom_phase(<phase>)`
|
||||
-- in the atom's `atom_info` line). Build a reverse map `atom_name -> phase_label` so the lookup is O(1) per atom scope.
|
||||
local atom_name_to_phase = {} ---@type table<AtomName, string> -- bag: atom name -> phase label
|
||||
for phase_label, entry in pairs(corpus.atom_phases or {}) do ---@type string, AtomPhaseGroup
|
||||
for _, atom_name in ipairs(entry.atoms or {}) do ---@type integer, AtomName
|
||||
atom_name_to_phase[atom_name] = phase_label
|
||||
end
|
||||
end
|
||||
|
||||
local atom_allocations = {} ---@type table<AtomName, GprAllocMap> -- bag: atom scope -> alloc map
|
||||
for atom_scope, decls in pairs(corpus.atom_auto_regs or {}) do ---@type AtomName, table<string, string>
|
||||
local phase_label = atom_name_to_phase[atom_scope] ---@type string|nil
|
||||
-- Build the atom's source pool: start with the full POOL, subtract:
|
||||
-- (a) every GPR already committed (phase allocations + prior atom allocations)
|
||||
-- (b) every USER-PINNED GPR (wave-context carriers + file-scope pinned aliases)
|
||||
-- (c) every GPR referenced in the atom's body — either hardcoded R_X or alias R_Xxx
|
||||
-- (the latter resolved via alias_to_gpr; this catches cases where the user wrote R_ResolveScratch instead of R_T4 directly)
|
||||
-- Atoms whose scope matches a phase share the global pool with the phase allocations;
|
||||
-- the original `source_pool = phase_allocations[phase_label]` form used the phase
|
||||
-- allocation MAP as a pool, but that map has no array part, so `table.remove(source_pool, 1)`
|
||||
-- returned nil and every atom-with-phase marker errored with `phase_register_pool_exhausted`.
|
||||
local used = {} ---@type table<GprIdent, boolean> -- bag: committed or body-referenced GPR -> true
|
||||
for _, m in pairs(phase_allocations) do for _, gpr in pairs(m) do used[gpr] = true end end ---@type integer, GprAllocMap
|
||||
for _, m in pairs(atom_allocations) do for _, gpr in pairs(m) do used[gpr] = true end end ---@type integer, GprAllocMap
|
||||
-- (c) Body references — scan the atom body for hardcoded + alias-resolved GPRs.
|
||||
-- Folded into `used` so the source_pool exclusion is a single check.
|
||||
local atom = corpus.atoms_by_name and corpus.atoms_by_name[atom_scope] ---@type AtomEntry|nil
|
||||
if atom and atom.body then
|
||||
local body_used = find_used_gprs(atom.body, alias_to_gpr) ---@type table<GprIdent, integer>
|
||||
for gpr in pairs(body_used) do used[gpr] = true end ---@type GprIdent
|
||||
end
|
||||
local source_pool = {} ---@type GprIdent[]
|
||||
for _, gpr in ipairs(POOL) do ---@type integer, GprIdent
|
||||
-- Exclude (a) prior commitments, (b) USER-PINNED GPRs (wave-context carriers declared via atom_reg + _Code defs, preserved across atoms globally).
|
||||
if not used[gpr] and not user_pinned[gpr] then
|
||||
source_pool[#source_pool + 1] = gpr
|
||||
end
|
||||
end
|
||||
local result = {} ---@type GprAllocMap
|
||||
for _, sym in ipairs(stable_sort_keys(decls)) do ---@type integer, string
|
||||
local next_gpr = table.remove(source_pool, 1) ---@type GprIdent|nil
|
||||
if not next_gpr then
|
||||
errors[#errors + 1] = {
|
||||
line = 0,
|
||||
msg = string.format("phase_register_pool_exhausted: atom '%s' requested symbol '%s' "
|
||||
.. "but no free registers remain in its scope pool."
|
||||
, atom_scope, sym),
|
||||
}
|
||||
else
|
||||
result[sym] = next_gpr
|
||||
end
|
||||
end
|
||||
atom_allocations[atom_scope] = result
|
||||
end
|
||||
|
||||
-- 3. Conflict-with-hardcoded detection (defensive — should be unreachable now).
|
||||
-- The source_pool exclusion in step 2 (b) + (c) already accounts for both user-pinned GPRs
|
||||
-- and body-referenced GPRs (hardcoded R_Tn OR alias R_<Alias>).
|
||||
-- An auto-reg allocation that matched an existing body reference would be impossible by construction.
|
||||
-- This warning is kept as a defensive safety net for cases the body scanner might miss
|
||||
-- (e.g. macros that expand to register references the scanner cannot resolve).
|
||||
-- For each resolved (scope, sym) -> R_Tn mapping, scan the atom body source for used GPRs.
|
||||
for atom_scope, decls in pairs(atom_allocations) do ---@type AtomName, GprAllocMap
|
||||
local atom = corpus.atoms_by_name and corpus.atoms_by_name[atom_scope] ---@type AtomEntry|nil
|
||||
if atom and atom.body then
|
||||
local used_in_body = find_used_gprs(atom.body, alias_to_gpr) ---@type table<GprIdent, integer>
|
||||
for sym, allocated_gpr in pairs(decls) do ---@type string, GprIdent
|
||||
if used_in_body[allocated_gpr] and used_in_body[allocated_gpr] > 0 then
|
||||
warnings[#warnings + 1] = {
|
||||
line = atom.line or 0,
|
||||
msg = string.format("phase_register_clash: atom '%s' has hardcoded '%s' in its body AND an auto-reg marker '%s' "
|
||||
.. "that was allocated to '%s' (same phase). Resolve by removing the hardcoded reference or renaming the auto-reg."
|
||||
, atom_scope, allocated_gpr, sym, allocated_gpr),
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
-- 4. Emit per-directory gen/auto_reg.h.
|
||||
-- For each source directory that has atom_auto_regs or phase_auto_regs entries, emit one header.
|
||||
local sources_by_dir = corpus.sources_by_dir or {} ---@type table<string, SourceFile[]>
|
||||
for dir, sources in pairs(sources_by_dir) do ---@type string, SourceFile[]
|
||||
local per_dir_mappings = {} ---@type GprAllocMap
|
||||
for _, src in ipairs(sources) do ---@type integer, SourceFile
|
||||
-- Collect every (sym -> gpr) entry that originated from a source in this directory.
|
||||
-- `src.scan.atom_auto_regs` is keyed by ATOM SCOPE NAME; `pairs(t)` iterates KEYS so `scope_name` here is the scope ident (e.g. "cube_g4_face").
|
||||
-- The previous `for _, scan_atom_auto` form silently assigned the VALUE (a `{sym = sym}` table) to the variable,
|
||||
-- which made `atom_allocations[scan_atom_auto]` a table-indexed lookup that never resolved.
|
||||
for scope_name in pairs(src.scan and src.scan.atom_auto_regs or {}) do ---@type string
|
||||
for sym, gpr in pairs(atom_allocations[scope_name] or {}) do ---@type string, GprIdent
|
||||
per_dir_mappings[sym] = gpr
|
||||
end
|
||||
end
|
||||
for scope_name in pairs(src.scan and src.scan.phase_auto_regs or {}) do ---@type string
|
||||
for sym, gpr in pairs(phase_allocations[scope_name] or {}) do ---@type string, GprIdent
|
||||
per_dir_mappings[sym] = gpr
|
||||
end
|
||||
end
|
||||
end
|
||||
local out_dir = dir .. "/gen" ---@type string
|
||||
local out_path = emit_auto_reg_h(out_dir, dir, sources, per_dir_mappings) ---@type string|nil
|
||||
if out_path then outputs[#outputs + 1] = { auto_reg_h = out_path } end
|
||||
end
|
||||
return { outputs = outputs, errors = errors, warnings = warnings }
|
||||
end
|
||||
|
||||
M.POOL = POOL
|
||||
|
||||
return M
|
||||
+777
-554
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,366 @@
|
||||
--- passes/emission_model.lua: Per-atom emission projection.
|
||||
---
|
||||
--- The `emission-model` pass owns `atom.paths`, the canonical per-atom mutable surface for atoms and raw atoms with bodies in `ctx.shared.corpus.source_order`.
|
||||
--- For each atom, the pass invokes `duffle.project_emission(body_text, components, word_counts, components)`.
|
||||
--- It stores the ordered `items` stream plus the dense `word_events` / `markers` / `invocations` views on `atom.paths`.
|
||||
---
|
||||
--- Public boundary:
|
||||
--- * `M.run(ctx)` is the only entry point.
|
||||
--- * The pass returns `{outputs = {}, errors = ..., warnings = ...}`.
|
||||
--- Pass kind = `validation`. Findings record on the result; the orchestrator does not exit non-zero.
|
||||
---
|
||||
--- Source-order discipline:
|
||||
--- * `corpus.source_order` sets the source-record order.
|
||||
--- * Within each source, the pass visits `src.scan.atoms` and `src.scan.raw_atoms` in declaration order.
|
||||
---
|
||||
--- Per-atom projection fields on `atom.paths`:
|
||||
--- `tokens`, `line_in_body`, `items`, `word_events`, `markers`, `invocations`, `errors`, `warnings`.
|
||||
--- The construction walk appends `items` and derives each dense view from that ordered stream.
|
||||
---
|
||||
--- Component expansion and construction validation:
|
||||
--- * known `mac_X(...)` calls recursively expand component bodies;
|
||||
--- * invocation records retain monotonic IDs, parent IDs, immediate call text, and the immutable outermost root call text;
|
||||
--- * invocation construction stamps `debug_skip` from `corpus.components[name].debug_skip` at the construction site (no second pass, no source parse, no parallel lookup);
|
||||
--- * component cycles close balanced invocation boundaries and emit a `cycle` construction error at the recursive edge;
|
||||
--- * declared-vs-measured component word counts emit `count_mismatch` construction errors; opaque uncounted macros emit warnings.
|
||||
---
|
||||
--- `passes.scan_source` strips its private `_code_macros` / `_code_macro_bodies` tables before this pass runs.
|
||||
|
||||
--- @class BodyToken
|
||||
--- @field tok string
|
||||
--- @field rel integer
|
||||
|
||||
--- @class EmissionItem
|
||||
--- @field kind string
|
||||
--- @field encoder string|nil
|
||||
--- @field args string[]|nil
|
||||
--- @field i integer|nil
|
||||
--- @field word_count integer|nil
|
||||
--- @field line integer|nil
|
||||
--- @field call_text string|nil
|
||||
--- @field root_call_text string|nil
|
||||
--- @field invocation_ids integer[]|nil
|
||||
--- @field outermost_invocation_id integer|nil
|
||||
--- @field gpr_keys string[]|nil
|
||||
--- @field ident string|nil
|
||||
--- @field isa_kind string|nil
|
||||
--- @field nop_words integer|nil
|
||||
--- @field is_yield boolean|nil
|
||||
--- @field is_load boolean|nil
|
||||
--- @field is_branch boolean|nil
|
||||
--- @field is_unconditional_jump boolean|nil
|
||||
--- @field is_terminal_jump boolean|nil
|
||||
--- @field gp0_shape string|nil
|
||||
--- @field name string|nil
|
||||
--- @field target string|nil
|
||||
--- @field word_index integer|nil
|
||||
--- @field consuming_encoder string|nil
|
||||
--- @field consuming_arg_pos integer|nil
|
||||
--- @field invocation_id integer|nil
|
||||
|
||||
--- @class WordEvent
|
||||
--- @field i integer
|
||||
--- @field encoder string
|
||||
--- @field args string[]
|
||||
--- @field def_path string
|
||||
--- @field def_line integer
|
||||
--- @field call_text string|nil
|
||||
--- @field root_call_text string|nil
|
||||
--- @field invocation_ids integer[]
|
||||
--- @field outermost_invocation_id integer
|
||||
--- @field word_count integer
|
||||
--- @field gpr_keys string[]|nil
|
||||
--- @field ident string
|
||||
--- @field kind string
|
||||
--- @field nop_words integer
|
||||
--- @field is_yield boolean
|
||||
--- @field is_load boolean
|
||||
--- @field is_branch boolean
|
||||
--- @field is_unconditional_jump boolean
|
||||
--- @field is_terminal_jump boolean
|
||||
--- @field gp0_shape string|nil
|
||||
--- @field body_line integer|nil
|
||||
--- @field call_line integer|nil
|
||||
--- @field call_path string|nil
|
||||
|
||||
--- @class EmissionMarker
|
||||
--- @field kind string
|
||||
--- @field name string
|
||||
--- @field line integer
|
||||
--- @field word_index integer
|
||||
--- @field target string|nil
|
||||
--- @field consuming_encoder string|nil
|
||||
--- @field consuming_arg_pos integer|nil
|
||||
|
||||
-- Finding: see ps1_meta.lua
|
||||
|
||||
--- @class AtomPaths
|
||||
--- @field tokens BodyToken[]
|
||||
--- @field line_in_body table<integer, integer> -- bag: body byte offset -> 1-based line
|
||||
--- @field items EmissionItem[]
|
||||
--- @field word_events WordEvent[]
|
||||
--- @field markers EmissionMarker[]
|
||||
--- @field invocations InvocationRecord[]
|
||||
--- @field errors Finding[]
|
||||
--- @field warnings Finding[]
|
||||
|
||||
--- @class EmissionModelPass
|
||||
--- @field run fun(ctx: PassCtx): PassResult
|
||||
|
||||
local M = {} ---@type EmissionModelPass
|
||||
|
||||
-- ─────────────────────────────────────────────────────────────────────────
|
||||
-- Bootstrap: load `duffle_paths.lua` via debug.getinfo so the module works standalone (run as `luajit passes/emission_model.lua`) and when require'd from the orchestrator.
|
||||
-- ─────────────────────────────────────────────────────────────────────────
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" ---@type string
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") ---@type DuffleExport
|
||||
|
||||
-- ─────────────────────────────────────────────────────────────────────────
|
||||
-- Helpers
|
||||
-- ─────────────────────────────────────────────────────────────────────────
|
||||
|
||||
-- Convert the recursive walk's body-relative line numbers into physical source lines once.
|
||||
-- The walker builds `line_of` from `body_text` and stamps body-relative line numbers (1..N) into `item.line` and `invocation.call_line`.
|
||||
-- This function converts those values to physical source lines at the close site with the forwarded source `line_of` closure.
|
||||
-- `call_line` discipline:
|
||||
-- * ROOT invocations (`inv.parent_id == 0`) receive body-relative `call_line` values directly from `M.LineIndex(body_text)` in the walker.
|
||||
-- The source `line_of` closure supplies physical lines at the close site, so this function converts each root value exactly once.
|
||||
-- * INNER invocations (`inv.parent_id ~= 0`) receive physical `call_line` values directly from the COMPONENT's `line_of` in the walker.
|
||||
-- Recursive descent forwards that closure through `corpus.components[name].line_of`; those values arrive physical and remain unchanged.
|
||||
--
|
||||
-- After this function, every `inv.call_line` is physical. DWARF and provenance output read it directly.
|
||||
-- The word-event loop forwards the already-physical `outer_inv.call_line` into `we.call_line` for words inside an invocation.
|
||||
--- @param projection EmissionProjection
|
||||
--- @param atom_record AtomEntry
|
||||
--- @param src SourceFile
|
||||
--- @param corpus Corpus
|
||||
--- @return nil
|
||||
local function stamp_root_provenance(projection, atom_record, src, corpus)
|
||||
local root_line_of = src.scan and src.scan.line_of ---@type (fun(pos: integer): integer)|nil
|
||||
assert(type(root_line_of) == "function"
|
||||
, "emission_model: src.scan.line_of is required (canonical LineIndex closure over the source text) to stamp physical provenance")
|
||||
assert(type(atom_record.body_off) == "number"
|
||||
, "emission_model: atom_record.body_off (byte offset of the body's first byte in source) is required to derive `root_body_line`. The scanner must populate body_off for every atom record.")
|
||||
-- `root_body_line` is the physical source line of the ATOM HEADER byte containing the opening `{`; that byte is one byte BEFORE `atom_record.body_off`.
|
||||
-- The walker assigns line 2 to the body's first content line because line 1 is the trailing `\n` after `{`. Body-text line k therefore maps to `root_body_line + (k - 1)`.
|
||||
-- `body_off - 1` points at the opening `{`, whose line index identifies the header line. `body_off` points after `{` and would shift every word row forward by one line.
|
||||
local root_body_line = root_line_of(atom_record.body_off - 1) or atom_record.line or 0 ---@type integer
|
||||
local components = corpus.components or {} ---@type table<string, Component>
|
||||
local word_items = {} ---@type EmissionItem[]
|
||||
|
||||
for _, item in ipairs(projection.items) do ---@type integer, EmissionItem
|
||||
if item.kind == "word" then word_items[#word_items + 1] = item end
|
||||
end
|
||||
|
||||
-- Resolve one word's physical body line, where the byte containing that word appears in source.
|
||||
-- * Component expansions carry `invocation_ids`; the component's full-file `line_of` leaves `item.line` physical.
|
||||
-- * Raw tokens in the root atom body carry an empty `invocation_ids` list and a body-relative `item.line`; convert them here.
|
||||
--- @param event WordEvent
|
||||
--- @param item EmissionItem
|
||||
--- @return integer
|
||||
local function body_line_for(event, item)
|
||||
local ids = event.invocation_ids or {} ---@type integer[]
|
||||
-- The innermost open invocation identifies which line index the walker used.
|
||||
-- A component `line_of` makes `item.line` physical; the atom's `body_text` line index makes it body-relative.
|
||||
if ids and #ids > 0 then
|
||||
local inner_id = ids[#ids] ---@type integer
|
||||
local inner_inv = inner_id and projection.invocations[inner_id] ---@type InvocationRecord|nil
|
||||
if inner_inv then
|
||||
local component = components[inner_inv.component_name] ---@type Component|nil
|
||||
if component and component.line_of then
|
||||
-- Walker used `comp.line_of`, which is the source's physical LineIndex. item.line is already physical.
|
||||
return item.line or 0
|
||||
end
|
||||
end
|
||||
end
|
||||
-- RAW root-body word: item.line is body-text's 1-based line number (the first content line is line 2 because line 1 is the trailing `\n` after `{`).
|
||||
-- Convert body-text-relative → physical using `root_body_line + (item.line - 1)`.
|
||||
return (root_body_line or 0) + (item.line or 1) - 1
|
||||
end
|
||||
|
||||
-- Stamp the root source path onto invocation records whose `call_path` the walker left empty.
|
||||
-- The walker passes `body_entry.source` to `emit_invoke_begin`; `M.project_emission` creates the root `body_entry` with source `""`, leaving its `call_path` empty.
|
||||
-- This stamp gives every invocation a physical `call_path` matching `passes/atoms_source_map.lua`'s in-memory provenance projection.
|
||||
local root_path = src.path or "" ---@type string
|
||||
for _, inv in ipairs(projection.invocations) do ---@type integer, InvocationRecord
|
||||
if inv.call_path == nil or inv.call_path == "" then
|
||||
inv.call_path = root_path
|
||||
end
|
||||
end
|
||||
|
||||
-- Normalize `inv.call_line` to a physical source line.
|
||||
-- * ROOT invocations (`parent_id == 0`) carry body-relative `call_line` values from `M.LineIndex(body_text)`; convert them once with `root_body_line`.
|
||||
-- * INNER invocations (`parent_id ~= 0`) carry physical `call_line` values from the component's `line_of`; retain them unchanged.
|
||||
for _, inv in ipairs(projection.invocations) do ---@type integer, InvocationRecord
|
||||
if inv.parent_id == 0 then
|
||||
inv.call_line = (root_body_line or 0) + (inv.call_line or 1) - 1
|
||||
end
|
||||
end
|
||||
|
||||
-- Build `body_lines` for each invocation.
|
||||
-- `atoms_source_map` and `dwarf_injection` read `inv.body_lines[k]` directly from the invocation record created here.
|
||||
-- Component words already carry physical `item.line` values from the walker's COMPONENT line index, so `body_line_for` returns them unchanged.
|
||||
for _, inv in ipairs(projection.invocations) do ---@type integer, InvocationRecord
|
||||
local sw = inv.start_word ---@type integer
|
||||
local ew = inv.end_word ---@type integer
|
||||
local bls = {} ---@type integer[]
|
||||
for i = sw, ew do ---@type integer
|
||||
local it = projection.items and projection.items[i] ---@type EmissionItem|nil
|
||||
if it and it.kind == "word" then
|
||||
local fake_event = { invocation_ids = { inv.id } } ---@type WordEvent
|
||||
bls[#bls + 1] = body_line_for(fake_event, it) or 0
|
||||
end
|
||||
end
|
||||
inv.body_lines = bls
|
||||
end
|
||||
|
||||
-- Resolve each `word_event`'s physical `body_line` and `call_line`.
|
||||
-- For words inside an invocation, `we.call_line` identifies the OUTER atom source line containing the `mac_X(...)` token that triggered expansion.
|
||||
-- The root-invocation conversion above makes every `inv.call_line` physical; forward it directly and use each raw word's `body_line` as the fallback.
|
||||
for index, we in ipairs(projection.word_events) do ---@type integer, WordEvent
|
||||
local item = word_items[index] or {} ---@type EmissionItem
|
||||
local body_line = body_line_for(we, item) ---@type integer
|
||||
item.line = body_line
|
||||
we.body_line = body_line
|
||||
|
||||
local call_line = body_line ---@type integer
|
||||
local outer_id = we.outermost_invocation_id or 0 ---@type integer
|
||||
local outer_inv = projection.invocations[outer_id] ---@type InvocationRecord|nil
|
||||
if outer_inv then
|
||||
-- `outer_inv.call_line` is physical after the conversion loop above, so use it directly.
|
||||
call_line = outer_inv.call_line
|
||||
end
|
||||
we.call_line = call_line
|
||||
|
||||
if we.def_path == nil or we.def_path == "" then we.def_path = src.path or "" end
|
||||
if we.def_line == nil or we.def_line == 0 then we.def_line = atom_record.line or 0 end
|
||||
if we.call_path == nil or we.call_path == "" then we.call_path = src.path or "" end
|
||||
end
|
||||
end
|
||||
|
||||
-- Project one atom record into `atom.paths`.
|
||||
-- Mutates the atom record in-place and returns the projection (for pass-level error/warning accumulation).
|
||||
--- @param atom_record AtomEntry
|
||||
--- @param src SourceFile
|
||||
--- @param corpus Corpus
|
||||
--- @return EmissionProjection
|
||||
local function project_atom(atom_record, src, corpus)
|
||||
local body = atom_record.body or "" ---@type string
|
||||
local wc = corpus.word_counts or {} ---@type WordCounts
|
||||
local comps = corpus.components or {} ---@type table<string, Component>
|
||||
local schema = nil ---@type RegUseSchema|nil
|
||||
if atom_record.reg_use_schema_name then
|
||||
schema = corpus.reg_use_schemas and corpus.reg_use_schemas[atom_record.reg_use_schema_name]
|
||||
end
|
||||
-- That construction site stamps `invocation.debug_skip` while appending each record to `proj.invocations`.
|
||||
local proj = duffle.project_emission(body, comps, wc, comps, { ---@type EmissionProjection
|
||||
reg_use_schema = schema,
|
||||
reg_use_param = atom_record.reg_use_param_name,
|
||||
atom_name = atom_record.name,
|
||||
schema_name = atom_record.reg_use_schema_name,
|
||||
})
|
||||
if atom_record.reg_use_schema_name and not schema then
|
||||
proj.errors[#proj.errors + 1] = {
|
||||
kind = "error",
|
||||
check = "reguse_missing_schema",
|
||||
msg = string.format("RegUse schema %q is missing", atom_record.reg_use_schema_name),
|
||||
schema_name = atom_record.reg_use_schema_name,
|
||||
}
|
||||
end
|
||||
for _, err in ipairs(corpus.reg_use_errors or {}) do ---@type integer, RegUseError
|
||||
if err.schema_name == atom_record.reg_use_schema_name then
|
||||
proj.errors[#proj.errors + 1] = {
|
||||
kind = "error",
|
||||
check = err.kind,
|
||||
line = err.line or err.source_line or 0,
|
||||
msg = err.msg or "",
|
||||
source = err.source or err.source_file,
|
||||
schema_name = err.schema_name,
|
||||
}
|
||||
end
|
||||
end
|
||||
local paths = { ---@type AtomPaths
|
||||
tokens = atom_record.body_tokens or {},
|
||||
line_in_body = duffle.build_body_line_index(body),
|
||||
items = proj.items,
|
||||
word_events = proj.word_events,
|
||||
markers = proj.markers,
|
||||
invocations = proj.invocations,
|
||||
errors = proj.errors,
|
||||
warnings = proj.warnings,
|
||||
}
|
||||
stamp_root_provenance(proj, atom_record, src, corpus)
|
||||
atom_record.paths = paths
|
||||
return proj
|
||||
end
|
||||
|
||||
-- ─────────────────────────────────────────────────────────────────────────
|
||||
-- Run the emission-model pass.
|
||||
-- ─────────────────────────────────────────────────────────────────────────
|
||||
|
||||
--- @param ctx PassCtx -- { shared = { corpus = ... }, out_root, ... }
|
||||
--- @return PassResult
|
||||
function M.run(ctx)
|
||||
local outputs = {} ---@type PassOutputEntry[]
|
||||
local errors = {} ---@type Finding[]
|
||||
local warnings = {} ---@type Finding[]
|
||||
|
||||
local corpus = ctx and ctx.shared and ctx.shared.corpus ---@type Corpus|nil
|
||||
if type(corpus) ~= "table" then error("emission_model: ctx.shared.corpus is required (canonical projection)", 0) end
|
||||
if type(corpus.source_order) ~= "table" then error("emission_model: ctx.shared.corpus.source_order is required", 0) end
|
||||
|
||||
-- Project once, collect errors + warnings for one atom.
|
||||
-- Kind must be one of: atom | atom_proc | raw_atom | comp_bare | comp_proc.
|
||||
--- @param atom AtomEntry
|
||||
--- @param src SourceFile
|
||||
--- @return nil
|
||||
local function process_atom(atom, src)
|
||||
if not (atom and atom.body) then return end
|
||||
local kind = atom.kind ---@type string
|
||||
if kind ~= "atom" and kind ~= "atom_proc" and kind ~= "raw_atom" and kind ~= "comp_bare" and kind ~= "comp_proc" then
|
||||
return
|
||||
end
|
||||
local proj = project_atom(atom, src, corpus) ---@type EmissionProjection
|
||||
for _, e in ipairs(proj.errors) do ---@type integer, Finding
|
||||
-- Finding.kind is severity. Finding.check holds the diagnostic code
|
||||
-- (cycle / count_mismatch / unbalanced / reguse_*).
|
||||
errors[#errors + 1] = {
|
||||
kind = "error",
|
||||
check = e.check,
|
||||
line = e.line,
|
||||
msg = e.msg,
|
||||
source = e.source or src.path,
|
||||
schema_name = e.schema_name,
|
||||
}
|
||||
end
|
||||
for _, w in ipairs(proj.warnings) do ---@type integer, Finding
|
||||
warnings[#warnings + 1] = {
|
||||
kind = "warning",
|
||||
check = w.check,
|
||||
line = w.line,
|
||||
msg = w.msg,
|
||||
}
|
||||
end
|
||||
end
|
||||
|
||||
-- Walk `corpus.source_order`; within each source, visit atoms followed by raw_atoms.
|
||||
-- Recognized kinds (atom | atom_proc | raw_atom | comp_bare | comp_proc) each receive the atom.paths projection via duffle.project_emission.
|
||||
-- Components are macros inlined into atom bodies; focused tests and isolated component analyses consume atom.paths directly.
|
||||
for _, src in ipairs(corpus.source_order) do ---@type integer, SourceFile
|
||||
local scan = src.scan or {} ---@type SourceScan
|
||||
for _, atom in ipairs(scan.atoms or {}) do ---@type integer, AtomEntry
|
||||
process_atom(atom, src)
|
||||
end
|
||||
for _, atom in ipairs(scan.raw_atoms or {}) do ---@type integer, AtomEntry
|
||||
process_atom(atom, src)
|
||||
end
|
||||
end
|
||||
|
||||
return {
|
||||
outputs = outputs,
|
||||
errors = errors,
|
||||
warnings = warnings,
|
||||
}
|
||||
end
|
||||
|
||||
return M
|
||||
+269
-341
@@ -1,391 +1,319 @@
|
||||
-- passes/offsets.lua
|
||||
--
|
||||
-- Generate <module>/gen/<basename>.offsets.h with branch offset
|
||||
-- immediates for every atom_offset(F, T) reference in atom bodies.
|
||||
-- Ported from scripts/tape_atom.offset_gen.meta.lua:148-389.
|
||||
--
|
||||
-- The branch offset regression we just fixed in commit 98e27c2 must
|
||||
-- NOT return. The fix was in duffle.lua's split_top_level_commas +
|
||||
-- tape_atom_annotation_pass.lua's compute_component_word_count.
|
||||
-- word_count_eval.count_token_words preserves the fix.
|
||||
--
|
||||
-- THIS MODULE ALSO REQUIRES the recent fix to duffle.lua's
|
||||
-- split_top_level_commas (the second-half of the 98e27c2 fix):
|
||||
-- top-level comments must be appended to the previous token, not
|
||||
-- stripped, so the emit path preserves `// trailing comment` text
|
||||
-- for convert_line_comments_to_block to convert to `/* */`.
|
||||
--
|
||||
-- Coding standard: tabs (1/level), EmmyLua annotations, no regex.
|
||||
--- passes/offsets.lua — Branch-offset generator.
|
||||
---
|
||||
--- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`)
|
||||
--- for `MipsAtom_(name)` and leftover `MipsCode code_*` declarations, computes the word offset
|
||||
--- (ELF symbol is the C ident; raw `code_*` is leftover, not the atom rule)
|
||||
--- from each `atom_offset(F, T)` marker to its target `atom_label(T)` declaration, and emits
|
||||
--- `gen/offsets.h` with one `#define _atom_offset_F_T = N` per branch.
|
||||
---
|
||||
--- Per-directory aggregation: every source in the same directory contributes to the same `gen/offsets.h`.
|
||||
--- The directory itself is the namespace; the filename does not repeat the module name.
|
||||
---
|
||||
--- The offset is `target_word - branch_word - 1` (the standard MIPS branch-immediate encoding: branch_offset = relative_pc_in_words - 1).
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Module-scope requires + package.path setup
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
local script_path = arg and arg[0] or "?"
|
||||
local last_sep = 0
|
||||
for i = 1, #script_path do
|
||||
local c = script_path:sub(i, i)
|
||||
if c == "/" or c == "\\" then last_sep = i end
|
||||
end
|
||||
local script_dir = last_sep == 0 and "./" or script_path:sub(1, last_sep)
|
||||
package.path = script_dir .. "../?.lua;" .. script_dir .. "../?/init.lua;" .. script_dir .. "?.lua;" .. package.path
|
||||
|
||||
local duffle = require("duffle")
|
||||
local trim = duffle.trim
|
||||
local read_ident = duffle.read_ident
|
||||
local is_space = duffle.is_space
|
||||
local is_alpha = duffle.is_alpha
|
||||
local is_alnum = duffle.is_alnum
|
||||
local skip_ws_and_cmt = duffle.skip_ws_and_cmt
|
||||
local skip_str_or_cmt = duffle.skip_str_or_cmt
|
||||
local split_top_level_commas = duffle.split_top_level_commas
|
||||
local read_parens = duffle.read_parens
|
||||
local read_braces = duffle.read_braces
|
||||
local write_file = duffle.write_file
|
||||
local dirname = duffle.dirname
|
||||
local basename_no_ext = duffle.basename_no_ext
|
||||
local ensure_dir = duffle.ensure_dir
|
||||
|
||||
local word_count_eval = require("word_count_eval")
|
||||
local count_token_words = word_count_eval.count_token_words
|
||||
-- Bootstrap: same as entry scripts. See `ps1_meta.lua` for the rationale.
|
||||
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
||||
-- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator.
|
||||
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
|
||||
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" ---@type string
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") ---@type DuffleExport
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Local helpers (ported from offset_gen.meta.lua lines 67-93)
|
||||
-- Constants
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
local function starts_with(s, prefix)
|
||||
if #s < #prefix then return false end
|
||||
for i = 1, #prefix do
|
||||
if s:sub(i, i) ~= prefix:sub(i, i) then return false end
|
||||
end
|
||||
return true
|
||||
end
|
||||
-- Offset macro/enum naming prefixes (the emitted header uses these).
|
||||
local OFFSET_MACRO_PREFIX = "_atom_offset_" ---@type string
|
||||
local OFFSET_ENUM_PREFIX = "atom_offset_" ---@type string
|
||||
|
||||
local function to_upper(s) return s:upper() end
|
||||
|
||||
local function to_alnum_underscore(s)
|
||||
local out = ""
|
||||
for i = 1, #s do
|
||||
local c = s:sub(i, i)
|
||||
if is_alnum(c) then out = out .. c
|
||||
else out = out .. "_" end
|
||||
end
|
||||
return out
|
||||
end
|
||||
|
||||
local function pad_right(s, w) return s .. string.rep(" ", w - #s) end
|
||||
-- Column width for the `#define _atom_offset_F_T = N` alignment.
|
||||
local OFFSET_MACRO_COL = 44 ---@type integer
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Marker-call helpers (ported from offset_gen.meta.lua lines 148-205)
|
||||
-- Type declarations
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Extract comma-separated identifier args from a parenthesized group
|
||||
--- after a function-like macro call.
|
||||
local function extract_ident_args(token, after_ident)
|
||||
local arg_start = skip_ws_and_cmt(token, after_ident)
|
||||
if token:sub(arg_start, arg_start) ~= "(" then return {}, nil end
|
||||
local inner, after_paren = read_parens(token, arg_start)
|
||||
-- SourceFile, PassCtx, PassResult: see ps1_meta.lua
|
||||
|
||||
local args = {}
|
||||
local n = 1
|
||||
local len = #inner
|
||||
while n <= len do
|
||||
n = skip_ws_and_cmt(inner, n)
|
||||
if n > len then break end
|
||||
local ident, after = read_ident(inner, n)
|
||||
if ident and ident ~= "" then
|
||||
table.insert(args, ident)
|
||||
n = after
|
||||
else
|
||||
n = n + 1
|
||||
end
|
||||
n = skip_ws_and_cmt(inner, n)
|
||||
if n <= len and inner:sub(n, n) == "," then n = n + 1 end
|
||||
--- @class BranchOffset
|
||||
--- @field tag string -- Marker tag (e.g. "F" in `atom_offset(F, T)`)
|
||||
--- @field target string -- Target label name (e.g. "T" in `atom_offset(F, T)`)
|
||||
--- @field branch_word integer -- Branch word position within the atom body
|
||||
--- @field offset integer -- Computed per consuming instruction (see `compute_offsets`)
|
||||
--- @field consuming_encoder string|nil -- Instruction consuming the offset (e.g. "branch_le_zero", "jump", "call_addr")
|
||||
--- @field consuming_arg_pos integer|nil -- 1-based arg position within the consuming instruction's arg list
|
||||
|
||||
--- @class AtomData
|
||||
--- @field name string -- Atom name
|
||||
--- @field total_words integer -- Total word count of the atom body
|
||||
--- @field offsets BranchOffset[] -- Per-branch offset list
|
||||
|
||||
--- @class OffsetBranch
|
||||
--- @field tag string
|
||||
--- @field target string
|
||||
--- @field branch_word integer
|
||||
--- @field consuming_encoder string|nil
|
||||
--- @field consuming_arg_pos integer|nil
|
||||
--- @field line integer|nil
|
||||
|
||||
--- @class MarkerProjectState
|
||||
--- @field labels table<string, integer> -- bag: label name -> word index
|
||||
--- @field branches OffsetBranch[]
|
||||
|
||||
--- @class OffsetConst
|
||||
--- @field macro_name string
|
||||
--- @field enum_name string
|
||||
--- @field value integer
|
||||
|
||||
--- @class OffsetOutput
|
||||
--- @field offsets_h string
|
||||
|
||||
--- @class OffsetsPass
|
||||
--- @field run fun(ctx: PassCtx): PassResult
|
||||
|
||||
--- @class AtomEntry
|
||||
--- @field paths AtomPaths|nil
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Canonical marker projection
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- MARKER_PROJECTORS is the marker-kind data table.
|
||||
-- The emission-model pass already records marker word positions + consuming-instruction context;
|
||||
-- this pass only projects those records into the label/branch lookup shape needed by offset computation.
|
||||
local MARKER_PROJECTORS = { ---@type table<string, fun(state: MarkerProjectState, marker: EmissionMarker): nil>
|
||||
--- @param state MarkerProjectState
|
||||
--- @param marker EmissionMarker
|
||||
--- @return nil
|
||||
label = function(state, marker)
|
||||
state.labels[marker.name] = marker.word_index
|
||||
end,
|
||||
--- @param state MarkerProjectState
|
||||
--- @param marker EmissionMarker
|
||||
--- @return nil
|
||||
offset = function(state, marker)
|
||||
state.branches[#state.branches + 1] = {
|
||||
tag = marker.name,
|
||||
target = marker.target,
|
||||
branch_word = marker.word_index,
|
||||
consuming_encoder = marker.consuming_encoder,
|
||||
consuming_arg_pos = marker.consuming_arg_pos,
|
||||
}
|
||||
end,
|
||||
}
|
||||
|
||||
--- Project canonical marker records into the two lookup tables used by the offset renderer.
|
||||
--- No source text, body text, or body token is inspected.
|
||||
--- @param markers EmissionMarker[]
|
||||
--- @return table<string, integer>
|
||||
--- @return OffsetBranch[]
|
||||
local function project_markers(markers)
|
||||
local state = { labels = {}, branches = {} } ---@type MarkerProjectState
|
||||
for _, marker in ipairs(markers or {}) do ---@type integer, EmissionMarker
|
||||
local project = MARKER_PROJECTORS[marker.kind] ---@type (fun(state: MarkerProjectState, marker: EmissionMarker): nil)|nil
|
||||
if project then project(state, marker) end
|
||||
end
|
||||
|
||||
return args, after_paren
|
||||
end
|
||||
|
||||
--- Scan a single token for atom_label/atom_offset markers, walking through
|
||||
--- balanced groups transparently (so nested calls are found).
|
||||
local function scan_for_atom_markers(token, at_pos, labels, branches)
|
||||
local i = 1
|
||||
local len = #token
|
||||
while i <= len do
|
||||
i = skip_ws_and_cmt(token, i)
|
||||
if i > len then break end
|
||||
local c = token:sub(i, i)
|
||||
if is_alpha(c) then
|
||||
local ident, after = read_ident(token, i)
|
||||
if ident == "atom_label" then
|
||||
local args, after_paren = extract_ident_args(token, after)
|
||||
if #args >= 1 then labels[args[1]] = at_pos end
|
||||
if after_paren then i = after_paren else i = after end
|
||||
elseif ident == "atom_offset" then
|
||||
local args, after_paren = extract_ident_args(token, after)
|
||||
if #args >= 2 then table.insert(branches, {pos = at_pos, target = args[2], tag = args[1]}) end
|
||||
if after_paren then i = after_paren else i = after end
|
||||
else
|
||||
i = after
|
||||
end
|
||||
else
|
||||
local nx = skip_str_or_cmt(token, i)
|
||||
if nx > i then i = nx else i = i + 1 end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
--- Find the end position (just past the closing ')') of the first
|
||||
--- atom_label/atom_offset call in `tok`. Returns 0 if no such call.
|
||||
local function find_marker_call_end(tok)
|
||||
local i = 1
|
||||
local len = #tok
|
||||
while i <= len do
|
||||
i = skip_ws_and_cmt(tok, i)
|
||||
if i > len then break end
|
||||
local c = tok:sub(i, i)
|
||||
if is_space(c) then
|
||||
i = i + 1
|
||||
elseif c == "/" then
|
||||
-- comment — skip past it (delegated to duffle.skip_str_or_cmt)
|
||||
local nx = skip_str_or_cmt(tok, i)
|
||||
if nx > i then i = nx else i = i + 1 end
|
||||
else
|
||||
local ident, after = read_ident(tok, i)
|
||||
if ident == "atom_label" or ident == "atom_offset" then
|
||||
local j = skip_ws_and_cmt(tok, after)
|
||||
if tok:sub(j, j) == "(" then
|
||||
local _, end_paren = read_parens(tok, j)
|
||||
return end_paren - 1
|
||||
end
|
||||
return 0
|
||||
end
|
||||
i = after or (i + 1)
|
||||
end
|
||||
end
|
||||
return 0
|
||||
return state.labels, state.branches
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Atom scanner (ported from offset_gen.meta.lua lines 245-321)
|
||||
-- Offset computation + header generation
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Skip C qualifier keywords (static, const, etc.) and return the position
|
||||
--- past the last qualifier.
|
||||
local function skip_qualifiers(source, i)
|
||||
local keywords = {
|
||||
["static"] = true, ["const"] = true, ["volatile"] = true,
|
||||
["extern"] = true, ["register"] = true, ["auto"] = true,
|
||||
["inline"] = true, ["typedef"] = true,
|
||||
["internal"]= true, ["LP_"] = true, ["global"] = true, ["gkknown"] = true,
|
||||
}
|
||||
while true do
|
||||
i = skip_ws_and_cmt(source, i)
|
||||
local ident, after = read_ident(source, i)
|
||||
if not ident then return i end
|
||||
if keywords[ident] then i = after else return i end
|
||||
end
|
||||
end
|
||||
|
||||
--- Find every MipsAtom_(name) { ... } in a source.
|
||||
local function find_atoms(source_text)
|
||||
local atoms = {}
|
||||
local len = #source_text
|
||||
local i = 1
|
||||
|
||||
local function try_wrapped(after_pos)
|
||||
local paren_pos = skip_ws_and_cmt(source_text, after_pos)
|
||||
if source_text:sub(paren_pos, paren_pos) ~= "(" then return nil end
|
||||
local inner, after_paren = read_parens(source_text, paren_pos)
|
||||
local n = 1
|
||||
while n <= #inner and is_space(inner:sub(n, n)) do n = n + 1 end
|
||||
local ns = n
|
||||
while n <= #inner and is_alnum(inner:sub(n, n)) do n = n + 1 end
|
||||
local name = inner:sub(ns, n - 1)
|
||||
if name == "" then return nil end
|
||||
-- Find the brace after the parens.
|
||||
local brace_pos = duffle.scan_to_char(source_text, "{", after_paren)
|
||||
if not brace_pos then return nil end
|
||||
local body, after_brace = read_braces(source_text, brace_pos)
|
||||
return {name = name, body = body, after_brace = after_brace}
|
||||
end
|
||||
|
||||
local function try_raw(after_pos)
|
||||
local next_pos = skip_ws_and_cmt(source_text, after_pos)
|
||||
local next_ident, next_after = read_ident(source_text, next_pos)
|
||||
if not next_ident then return nil end
|
||||
if not starts_with(next_ident, "code_") then return nil end
|
||||
if #next_ident <= 5 then return nil end
|
||||
local atom_name = next_ident:sub(6)
|
||||
local brace_pos = duffle.scan_to_char(source_text, "{", next_after)
|
||||
if not brace_pos then return nil end
|
||||
local body, after_brace = read_braces(source_text, brace_pos)
|
||||
return {name = atom_name, body = body, after_brace = after_brace}
|
||||
end
|
||||
|
||||
while i <= len do
|
||||
i = skip_ws_and_cmt(source_text, i); if i > len then break end
|
||||
i = skip_qualifiers(source_text, i); if i > len then break end
|
||||
local ident, after = read_ident(source_text, i)
|
||||
if not ident then
|
||||
i = i + 1
|
||||
elseif ident == "MipsAtom_" then
|
||||
local atom = try_wrapped(after)
|
||||
if atom then
|
||||
table.insert(atoms, {name = atom.name, body = atom.body})
|
||||
i = atom.after_brace
|
||||
else
|
||||
i = i + 1
|
||||
end
|
||||
elseif ident == "MipsCode" then
|
||||
local atom = try_raw(after)
|
||||
if atom then
|
||||
table.insert(atoms, {name = atom.name, body = atom.body})
|
||||
i = atom.after_brace
|
||||
else
|
||||
i = after
|
||||
end
|
||||
else
|
||||
i = after
|
||||
end
|
||||
end
|
||||
return atoms
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Per-atom body scan (ported from offset_gen.meta.lua lines 207-239)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Scan an atom body for labels + branches, count total words.
|
||||
--- Returns (labels, branches, total_words).
|
||||
local function scan_atom_body(body, word_counts)
|
||||
local pos = 0
|
||||
local labels = {}
|
||||
local branches = {}
|
||||
for _, tok in ipairs(split_top_level_commas(body)) do
|
||||
local k = 1
|
||||
local tlen = #tok
|
||||
while k <= tlen and is_space(tok:sub(k, k)) do k = k + 1 end
|
||||
local leading_ident = read_ident(tok, k)
|
||||
if leading_ident == "atom_label" or leading_ident == "atom_offset" then
|
||||
-- Marker call: record at the current pos, do NOT advance pos.
|
||||
-- But the source pattern may bundle the marker with the next
|
||||
-- instruction on a new line (no top-level comma between them).
|
||||
-- In that case, the rest of `tok` after the marker call is
|
||||
-- a real instruction that must still be counted.
|
||||
scan_for_atom_markers(tok, pos, labels, branches)
|
||||
local marker_end = find_marker_call_end(tok)
|
||||
if marker_end > 0 and marker_end < #tok then
|
||||
local rest = trim(tok:sub(marker_end + 1))
|
||||
if rest ~= "" then
|
||||
local rest_words = count_token_words(rest, word_counts)
|
||||
pos = pos + rest_words
|
||||
end
|
||||
end
|
||||
else
|
||||
local words = count_token_words(tok, word_counts)
|
||||
scan_for_atom_markers(tok, pos, labels, branches)
|
||||
pos = pos + words
|
||||
end
|
||||
end
|
||||
return labels, branches, pos
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Offset computation + header generation (ported lines 327-383)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Compute branch offsets as (target_word - branch_word - 1).
|
||||
local function compute_offsets(labels, branches)
|
||||
local results = {}
|
||||
for _, br in ipairs(branches) do
|
||||
local target = labels[br.target]
|
||||
--- Compute branch offsets per consuming instruction.
|
||||
--- Disposition table:
|
||||
--- `branch_*` -> relative offset: `target_word - branch_word - 1` (MIPS branch-immediate encoding).
|
||||
--- `jump` / `call_addr` -> same value as `branch_*` (a relative word offset).
|
||||
--- The duffle headers' `enc_i` macro truncates the value to the immediate-field width (16 bits for branches, 26 bits for jumps).
|
||||
--- For tape-atom bodies within a single module, this works for `j`/`jal` because the linker's symbol resolution produces the correct 26-bit absolute target via standard `j` relocations.
|
||||
--- For cross-module `j`/`jal` (atom body in one module, target in another), the linker emits a `R_MIPS_26` relocation against the lower 26 bits; the upper 4 bits come from the PC of the delay slot following the `j`.
|
||||
--- The metaprogram doesn't know either at compile time, so the emitted value is the relative word offset that the duffle `enc_i` macro places in the immediate field; the toolchain handles the rest.
|
||||
--- `jump_reg` / `call_reg` / `jump_link` -> ERROR. Register-form jumps have no offset field; `atom_offset` is invalid.
|
||||
--- missing `consuming_encoder` -> ERROR. A lone top-level `atom_offset` is not a branch.
|
||||
--- @param labels table<string, integer>
|
||||
--- @param branches OffsetBranch[]
|
||||
--- @param errors Finding[]
|
||||
--- @return BranchOffset[]
|
||||
local function compute_offsets(labels, branches, errors)
|
||||
local results = {} ---@type BranchOffset[]
|
||||
for _, br in ipairs(branches) do ---@type integer, OffsetBranch
|
||||
local target = labels[br.target] ---@type integer|nil
|
||||
if not target then
|
||||
error("Branch target '" .. br.target .. "' has no atom_label (at word " .. br.pos .. ")")
|
||||
errors[#errors + 1] = {
|
||||
line = br.line or 0,
|
||||
msg = "Branch target '" .. br.target .. "' has no atom_label (at word " .. br.branch_word .. ")",
|
||||
}
|
||||
else
|
||||
local consuming = br.consuming_encoder ---@type string|nil
|
||||
if consuming == nil or consuming == "" then
|
||||
errors[#errors + 1] = {
|
||||
line = br.line or 0,
|
||||
msg = "atom_offset requires a consuming encoder (branch_*, jump, call_addr); top-level atom_offset is invalid; at word " .. br.branch_word,
|
||||
}
|
||||
elseif consuming == "jump_reg" or consuming == "call_reg" or consuming == "jump_link" then
|
||||
errors[#errors + 1] = {
|
||||
line = br.line or 0,
|
||||
msg = "atom_offset cannot be used with " .. consuming
|
||||
.. " (register-form jumps have no offset field); at word " .. br.branch_word,
|
||||
}
|
||||
else
|
||||
-- Consuming instructions with an offset field (`branch_*`, `jump`, `call_addr`) use the same relative offset value.
|
||||
-- The MIPS encoding differs per opcode but the duffle `enc_i` macro handles the truncation to the immediate-field width.
|
||||
results[#results + 1] = {
|
||||
target = br.target,
|
||||
tag = br.tag,
|
||||
branch_word = br.branch_word,
|
||||
offset = target - br.branch_word - 1,
|
||||
consuming_encoder = br.consuming_encoder,
|
||||
consuming_arg_pos = br.consuming_arg_pos,
|
||||
}
|
||||
end
|
||||
end
|
||||
table.insert(results, {target = br.target, tag = br.tag, offset = target - br.pos - 1})
|
||||
end
|
||||
return results
|
||||
end
|
||||
|
||||
--- Generate the per-source .offsets.h header.
|
||||
local function generate_header(source_path, atoms_data)
|
||||
local basename = basename_no_ext(source_path)
|
||||
--- Right-pad `s` with spaces to width `w`. If `s` is already `w` or wider, no padding is added.
|
||||
--- @param s string
|
||||
--- @param w integer
|
||||
--- @return string
|
||||
local function pad_right(s, w)
|
||||
return s .. string.rep(" ", math.max(0, w - #s))
|
||||
end
|
||||
|
||||
local lines = {}
|
||||
local function add(s) table.insert(lines, s) end
|
||||
--- (internal) Build a constant-table entry `{macro_name, enum_name, value}` from a BranchOffset.
|
||||
--- @param bo BranchOffset
|
||||
--- @return OffsetConst
|
||||
local function make_offset_const(bo)
|
||||
return {
|
||||
macro_name = OFFSET_MACRO_PREFIX .. bo.tag .. "_" .. bo.target,
|
||||
enum_name = OFFSET_ENUM_PREFIX .. bo.tag .. "_" .. bo.target,
|
||||
value = bo.offset,
|
||||
}
|
||||
end
|
||||
|
||||
--- (internal) Emit one atom's offset constants + enum into the lines buffer.
|
||||
--- @param add fun(s: string)
|
||||
--- @param atom AtomData
|
||||
--- @return nil
|
||||
local function emit_atom_offsets(add, atom)
|
||||
if #atom.offsets == 0 then return end
|
||||
add("// --- atom: " .. atom.name .. " (" .. atom.total_words .. " words) ---")
|
||||
add("")
|
||||
local consts = {} ---@type OffsetConst[]
|
||||
for _, r in ipairs(atom.offsets) do ---@type integer, BranchOffset
|
||||
consts[#consts + 1] = make_offset_const(r)
|
||||
end
|
||||
for _, c in ipairs(consts) do ---@type integer, OffsetConst
|
||||
add("#define " .. pad_right(c.macro_name, OFFSET_MACRO_COL) .. " " .. c.value)
|
||||
end
|
||||
add("")
|
||||
add("enum {")
|
||||
for _, c in ipairs(consts) do ---@type integer, OffsetConst
|
||||
add(" " .. c.enum_name .. " = " .. c.macro_name .. ",")
|
||||
end
|
||||
add("};")
|
||||
add("")
|
||||
end
|
||||
|
||||
--- Generate the per-directory .offsets.h header.
|
||||
--- @param dir string
|
||||
--- @param sources SourceFile[]
|
||||
--- @param atoms_data AtomData[]
|
||||
--- @return string
|
||||
local function generate_header(dir, sources, atoms_data)
|
||||
local dir_basename = duffle.basename_no_ext(dir) ---@type string
|
||||
|
||||
local lines = {} ---@type string[]
|
||||
--- @param s string
|
||||
--- @return nil
|
||||
local function add(s) lines[#lines + 1] = s end
|
||||
|
||||
add("// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT")
|
||||
add("// Source: " .. source_path)
|
||||
add("// Directory: " .. dir:gsub("/", "\\") .. "\\")
|
||||
for _, src in ipairs(sources) do ---@type integer, SourceFile
|
||||
add("// source: " .. src.path:gsub("/", "\\"))
|
||||
end
|
||||
add("#pragma once")
|
||||
add("")
|
||||
add("#pragma region " .. basename)
|
||||
add("#pragma region " .. dir_basename)
|
||||
add("")
|
||||
add("")
|
||||
for _, atom in ipairs(atoms_data) do
|
||||
if #atom.offsets > 0 then
|
||||
add("// --- atom: " .. atom.name .. " (" .. atom.total_words .. " words) ---")
|
||||
add("")
|
||||
local consts = {}
|
||||
for _, r in ipairs(atom.offsets) do
|
||||
table.insert(consts, {
|
||||
macro_name = "_atom_offset_" .. r.tag .. "_" .. r.target,
|
||||
enum_name = "atom_offset_" .. r.tag .. "_" .. r.target,
|
||||
value = r.offset,
|
||||
})
|
||||
end
|
||||
for _, c in ipairs(consts) do
|
||||
add("#define " .. pad_right(c.macro_name, 44) .. " " .. c.value)
|
||||
end
|
||||
add("")
|
||||
add("enum {")
|
||||
for _, c in ipairs(consts) do
|
||||
add(" " .. c.enum_name .. " = " .. c.macro_name .. ",")
|
||||
end
|
||||
add("};")
|
||||
add("")
|
||||
end
|
||||
for _, atom in ipairs(atoms_data) do ---@type integer, AtomData
|
||||
emit_atom_offsets(add, atom)
|
||||
end
|
||||
add("#pragma endregion " .. basename)
|
||||
add("#pragma endregion " .. dir_basename)
|
||||
add("")
|
||||
return table.concat(lines, "\n") .. "\n"
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- M.run — orchestrator entry
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
local M = {} ---@type OffsetsPass
|
||||
|
||||
--- @class M
|
||||
--- (internal) Aggregate atoms from every source in one directory, render the per-directory `offsets.h`.
|
||||
--- Returns the offsets_h path if a header was written, or nil.
|
||||
--- @param ctx PassCtx
|
||||
--- @param dir string
|
||||
--- @param sources SourceFile[]
|
||||
--- @param errors Finding[]
|
||||
--- @return string|nil
|
||||
local function process_directory(ctx, dir, sources, errors)
|
||||
local atoms_data = {} ---@type AtomData[]
|
||||
|
||||
local M = {}
|
||||
--- @param atom AtomEntry
|
||||
--- @return nil
|
||||
local function append_atom(atom)
|
||||
local paths = atom and atom.paths ---@type AtomPaths|nil
|
||||
if not paths then return end
|
||||
local labels, branches = project_markers(paths.markers) ---@type table<string, integer>, OffsetBranch[]
|
||||
atoms_data[#atoms_data + 1] = {
|
||||
name = atom.raw_name or atom.name,
|
||||
total_words = #(paths.word_events or {}),
|
||||
offsets = compute_offsets(labels, branches, errors),
|
||||
}
|
||||
end
|
||||
|
||||
for _, src in ipairs(sources) do ---@type integer, SourceFile
|
||||
local scan = src.scan or {} ---@type SourceScan
|
||||
for _, atom in ipairs(scan.atoms or {}) do append_atom(atom) end ---@type integer, AtomEntry
|
||||
for _, atom in ipairs(scan.raw_atoms or {}) do append_atom(atom) end ---@type integer, AtomEntry
|
||||
end
|
||||
if #atoms_data == 0 then return nil end
|
||||
|
||||
local out_path = dir .. "/gen/offsets.h" ---@type string
|
||||
duffle.ensure_dir(duffle.dirname(out_path))
|
||||
duffle.write_file(out_path, generate_header(dir, sources, atoms_data))
|
||||
return out_path
|
||||
end
|
||||
|
||||
--- Run the offsets pass.
|
||||
--- For each canonical source-directory, emits a per-directory `gen/offsets.h`
|
||||
--- containing constants for every marker recorded in atom.paths across every source in that directory.
|
||||
--- @param ctx PassCtx
|
||||
--- @return PassResult
|
||||
function M.run(ctx)
|
||||
local outputs = {}
|
||||
local errors = {}
|
||||
local warnings = {}
|
||||
local outputs = {} ---@type OffsetOutput[]
|
||||
local errors = {} ---@type Finding[]
|
||||
local warnings = {} ---@type Finding[]
|
||||
|
||||
for _, src in ipairs(ctx.sources) do
|
||||
local atoms = find_atoms(src.text)
|
||||
if #atoms > 0 then
|
||||
local atoms_data = {}
|
||||
for _, atom in ipairs(atoms) do
|
||||
local labels, branches, total = scan_atom_body(atom.body, ctx.shared.word_counts)
|
||||
local offsets = compute_offsets(labels, branches)
|
||||
table.insert(atoms_data, {
|
||||
name = atom.name,
|
||||
total_words = total,
|
||||
offsets = offsets,
|
||||
})
|
||||
end
|
||||
local corpus = ctx.shared and ctx.shared.corpus ---@type Corpus|nil
|
||||
if type(corpus) ~= "table" then
|
||||
error("offsets.run requires ctx.shared.corpus", 0)
|
||||
end
|
||||
if type(corpus.source_order) ~= "table" then
|
||||
error("offsets.run requires ctx.shared.corpus.source_order.", 0)
|
||||
end
|
||||
|
||||
local out_path = src.dir .. "/gen/" .. basename_no_ext(src.dir) .. ".offsets.h"
|
||||
if not ctx.dry_run then
|
||||
ensure_dir(dirname(out_path))
|
||||
write_file(out_path, generate_header(src.path, atoms_data))
|
||||
end
|
||||
table.insert(outputs, { offsets_h = out_path })
|
||||
-- Per-directory aggregation: every source in the same directory contributes to one `gen/offsets.h`.
|
||||
local sources_by_dir = corpus.sources_by_dir or duffle.group_sources_by_dir(corpus.source_order) ---@type table<string, SourceFile[]>
|
||||
for dir, sources in pairs(sources_by_dir) do ---@type string, SourceFile[]
|
||||
local out_path = process_directory(ctx, dir, sources, errors) ---@type string|nil
|
||||
if out_path then
|
||||
outputs[#outputs + 1] = { offsets_h = out_path }
|
||||
end
|
||||
end
|
||||
|
||||
|
||||
+1107
-164
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,112 @@
|
||||
--- word_count_eval.lua — Word-counting logic for the tape-atom metaprogram pipeline.
|
||||
---
|
||||
--- Two responsibilities:
|
||||
--- 1. **Public utility** `M.count_token_words(token, wc)`: Used by `passes/offsets.lua`, `passes/annotation.lua`, and other passes.
|
||||
--- 2. **Pass entry** `M.run(ctx)`: Loads the authored `word_count.metadata.h` into `ctx.shared.corpus.word_counts` for downstream passes.
|
||||
--- The generated `.macs.h` files are OUTPUT artifacts and are NOT inputs to this pass;
|
||||
--- Current component counts are owned by `passes/components.lua` (which populates `corpus.word_counts` and `corpus.components`
|
||||
--- AFTER computing each current count from the just-built body + `corpus.word_counts`).
|
||||
---
|
||||
--- **Canonical contract**:
|
||||
--- * `ctx.shared.corpus.word_counts` is the count table.
|
||||
--- * `corpus.word_counts` is the sole count table. Consumers read `corpus.word_counts` directly.
|
||||
--- * `ctx.shared.components` is NOT created by this pass (projections only).
|
||||
--- * No `.macs.h` recursive discovery (no `scan_dir`, no scan cache, no `_invalidate_scan_cache`).
|
||||
---
|
||||
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
|
||||
--- Lua 5.3 compatible.
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Module-scope requires + package.path setup
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
||||
-- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator.
|
||||
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" ---@type string
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") ---@type DuffleExport
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Type declarations
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- @class WordCounts
|
||||
--- @field [string] integer -- bag: macro name -> word count
|
||||
|
||||
--- @class WordCountEval
|
||||
--- @field count_token_words fun(token: string, wc: WordCounts): integer
|
||||
--- @field run fun(ctx: PassCtx): PassResult
|
||||
|
||||
-- SourceFile, PassCtx, PassResult: see ps1_meta.lua
|
||||
-- DuffleExport: see duffle.lua (facade returned by duffle_paths.lua)
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Module exports
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
local M = {} ---@type WordCountEval
|
||||
|
||||
-- ┌────────────────────────────────────────────────────────────────────┐
|
||||
-- │ Shared utility: count_token_words │
|
||||
-- └────────────────────────────────────────────────────────────────────┘
|
||||
|
||||
--- Count words emitted by a single comma-separated token inside an atom body.
|
||||
--- For most tokens (regular MIPS instructions) this returns 1.
|
||||
--- For `mac_X(...)` calls, this returns the resolved word count from `wc` (recursively if needed). For `nop2` etc., returns wc[name].
|
||||
--- For unknown macros, returns 1 and (optionally) warns.
|
||||
--- @param token string -- a single token from split_top_level_commas
|
||||
--- @param wc WordCounts -- the shared word-count table
|
||||
--- @return integer
|
||||
function M.count_token_words(token, wc)
|
||||
local s = duffle.trim(token) ---@type string
|
||||
if s == "" then return 0 end
|
||||
local name, after = duffle.read_ident(s, 1) ---@type string|nil, integer
|
||||
if not name then return 1 end
|
||||
if wc[name] then return wc[name] end
|
||||
local paren_pos = duffle.skip_ws_and_cmt(s, after) ---@type integer
|
||||
if s:sub(paren_pos, paren_pos) == "(" then
|
||||
io.stderr:write(" warning: unknown macro '" .. name .. "', assuming 1 word\n")
|
||||
end
|
||||
return 1
|
||||
end
|
||||
|
||||
-- ┌────────────────────────────────────────────────────────────────────┐
|
||||
-- │ Pass entry: M.run(ctx) — "word-counts" pass │
|
||||
-- └────────────────────────────────────────────────────────────────────┘
|
||||
|
||||
--- Load the authored `word_count.metadata.h` into `ctx.shared.corpus.word_counts`.
|
||||
--- Generated `.macs.h` files are OUTPUT artifacts and are NOT scanned as inputs.
|
||||
--- Current component counts are computed and inserted by `passes/components.lua`
|
||||
--- after the components pass iterates `corpus.source_order` and writes each source-directory's `gen/macs.h` file.
|
||||
---
|
||||
--- Contract:
|
||||
--- * `ctx.shared.corpus` MUST exist (canonical corpus ownership).
|
||||
--- * `ctx.metadata_path` MUST be a readable file path to the authored `word_count.metadata.h`.
|
||||
--- * The pass assigns exactly one table to `corpus.word_counts`.
|
||||
--- Consumers read the corpus-owned table directly.
|
||||
--- Consumers must read `corpus.word_counts` directly.
|
||||
--- @param ctx PassCtx
|
||||
--- @return PassResult
|
||||
function M.run(ctx)
|
||||
-- 1. Canonical-corpus ownership gate.
|
||||
local corpus = ctx.shared and ctx.shared.corpus ---@type Corpus|nil
|
||||
if type(corpus) ~= "table" then
|
||||
error("word_count_eval.run requires ctx.shared.corpus (canonical corpus). The fixture must install the corpus before running this pass.", 0)
|
||||
end
|
||||
|
||||
-- 2. metadata_path gate.
|
||||
if type(ctx.metadata_path) ~= "string" or ctx.metadata_path == "" then
|
||||
error("word_count_eval.run requires ctx.metadata_path (path to the authored word_count.metadata.h).", 0)
|
||||
end
|
||||
|
||||
-- 3. Load authored metadata. Generated .macs.h files are NOT scanned
|
||||
-- (the pass computes their counts from the just-built bodies after disk emission; see passes/components.lua).
|
||||
local wc = duffle.load_word_counts(ctx.metadata_path) ---@type WordCounts
|
||||
|
||||
-- 4. Assign the count table. ONE assignment, no copy. The assignment creates no secondary alias.
|
||||
corpus.word_counts = wc
|
||||
|
||||
return { outputs = {}, errors = {}, warnings = {} }
|
||||
end
|
||||
|
||||
return M
|
||||
Binary file not shown.
@@ -0,0 +1,51 @@
|
||||
-- autoexec.lua - pcsx_debug_helper plugin entry point.
|
||||
-- Packaged in scripts/pcsx_debug_helper.zip. Loaded by pcsx-redux via the -archive CLI flag (see scripts/launch_pcsx_debug.ps1).
|
||||
--
|
||||
-- Registers two web handlers for external CLI tools:
|
||||
-- /api/v1/lua/gte - full GTE state (32 data + 32 control regs + PC)
|
||||
-- /api/v1/lua/gp - GP state summary (screenshot endpoint + VRAM endpoint refs)
|
||||
--
|
||||
-- The GTE handler reads COP2 regs via PCSX.getRegisters().CP2D/CP2C.
|
||||
-- The pcsx-redux gdb stub doesn't expose COP2, so this is the only way for external tools to see GTE state.
|
||||
--
|
||||
-- The GP handler is a thin pointer:
|
||||
-- pcsx-redux's Lua API exposes only PCSX.GPU.takeScreenShot() (no GPUSTAT, no GP0/GP1 command log, no display state). For richer GP state, the existing web endpoints are the practical path:
|
||||
-- /api/v1/state/still - PNG screenshot
|
||||
-- /api/v1/gpu/vram/raw - VRAM raw bytes (1MB)
|
||||
--
|
||||
-- Companion: scripts/gdb/gdb_tape_atoms.gdb (covers GPRs + atom-aware stepping).
|
||||
|
||||
local function register_handlers()
|
||||
if not PCSX.WebServer then PCSX.WebServer = {} end
|
||||
if not PCSX.WebServer.Handlers then PCSX.WebServer.Handlers = {} end
|
||||
|
||||
-- ── GTE state ──
|
||||
PCSX.WebServer.Handlers.gte = function(req)
|
||||
local r = PCSX.getRegisters()
|
||||
local out = { "pc=0x" .. string.format("%x", r.pc) }
|
||||
for i = 0, 31 do
|
||||
out[#out + 1] = string.format("D[%d]=0x%08x C[%d]=0x%08x",
|
||||
i, r.CP2D.r[i], i, r.CP2C.r[i])
|
||||
end
|
||||
return table.concat(out, "\n")
|
||||
end
|
||||
|
||||
-- ── GP state (pointer to existing endpoints) ──
|
||||
-- pcsx-redux's Lua GPU API exposes only takeScreenShot(); no GPUSTAT / GP0 / GP1 command log / display state.
|
||||
-- We point to the existing web endpoints that DO expose those (when the emulator is actually rendering. Paused-at-BP frames won't have a fresh frame).
|
||||
PCSX.WebServer.Handlers.gp = function(req)
|
||||
local out = {
|
||||
"gpu_screenshot_png=http://localhost:8080/api/v1/state/still",
|
||||
"vram_raw=http://localhost:8080/api/v1/gpu/vram/raw (1MB VRAM)",
|
||||
"gpustat=NOT_AVAILABLE_VIA_LUA",
|
||||
"gp_command_log=NOT_AVAILABLE_VIA_LUA (use pcsx-redux Debug > GPU Logger)",
|
||||
"hint_run_emulator_unpaused_for_screenshot",
|
||||
}
|
||||
return table.concat(out, "\n")
|
||||
end
|
||||
end
|
||||
|
||||
local ok, err = pcall(register_handlers)
|
||||
if ok then print("[pcsx_debug_helper] handlers registered: gte, gp")
|
||||
else print("[pcsx_debug_helper] registration failed: " .. tostring(err))
|
||||
end
|
||||
+633
-359
File diff suppressed because it is too large
Load Diff
+204
-7
@@ -4,27 +4,31 @@ $path_code = join-path $path_root 'code'
|
||||
$path_scripts = join-path $path_root 'scripts'
|
||||
$path_toolchain = join-path $path_root 'toolchain'
|
||||
|
||||
# Halt on any error (instead of PowerShell's default `Continue`).
|
||||
$ErrorActionPreference = 'Stop'
|
||||
|
||||
$misc = join-path $PSScriptRoot 'helpers/misc.ps1'
|
||||
. $misc
|
||||
|
||||
# TODO(Ed): Review usage of these deps
|
||||
# I orgiinally cloned them when starting to get to the C runtime usage of the course
|
||||
# However, based on the heavy reliance of the PSX.Dev extension I might fallback; also
|
||||
# The gdb server doesn't need the full repo and were only using the src/mips
|
||||
# which has a standalone repo (nuggets)
|
||||
# armips may not be used at all but I'm not sure...
|
||||
|
||||
$url_armips = 'https://github.com/Kingcom/armips.git'
|
||||
$url_pcsx_redux = 'https://github.com/grumpycoders/pcsx-redux.git'
|
||||
$url_psyq_iwyu = 'https://github.com/johnbaumann/psyq_include_what_you_use.git'
|
||||
$url_lpeg = 'https://github.com/roberto-ieru/LPeg.git'
|
||||
# $url_mkpsxiso = 'https://github.com/Lameguy64/mkpsxiso.git'
|
||||
|
||||
$url_mkpsxiso_win64 = 'https://github.com/Lameguy64/mkpsxiso/releases/download/v2.30/mkpsxiso-2.30-win64.zip'
|
||||
|
||||
$path_armips = join-path $path_toolchain 'armips'
|
||||
$path_pcsx_redux = join-path $path_toolchain 'pcsx-redux'
|
||||
$path_psyq_iwyu = join-path $path_toolchain 'psyq_iwyu'
|
||||
$path_lpeg = join-path $path_toolchain 'lpeg'
|
||||
$path_mkpsxiso = join-path $path_toolchain 'mkpsxiso'
|
||||
|
||||
clone-gitrepo $path_armips $url_armips
|
||||
clone-gitrepo $path_lpeg $url_lpeg
|
||||
clone-gitrepo $path_pcsx_redux $url_pcsx_redux
|
||||
clone-gitrepo $path_psyq_iwyu $url_psyq_iwyu
|
||||
# clone-gitrepo $path_mkpsxiso $url_mkpsxiso
|
||||
|
||||
$path_armips_build = join-path $path_armips 'build'
|
||||
verify-path $path_armips_build
|
||||
@@ -37,3 +41,196 @@ pop-location
|
||||
# $path_pcsx_redux_binaries = join-path $path_pcsx_redux_vsprojects 'x64/Release'
|
||||
|
||||
# $psyq_obj_parser = join-path $path_pcsx_redux_binaries 'psyq-obj-parser.exe'
|
||||
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
# PCSX-Redux — built via MSBuild (VS2022)
|
||||
# Requires: Visual Studio 2022 with the C++ desktop workload.
|
||||
# Output: toolchain\pcsx-redux\vsprojects\x64\Debug\pcsx-redux.exe
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
# Locate MSBuild from the VS2022 install (no hardcoded path — uses vswhere).
|
||||
$vswhere = "${env:ProgramFiles(x86)}\Microsoft Visual Studio\Installer\vswhere.exe"
|
||||
if (-not (Test-Path $vswhere)) {
|
||||
write-error "vswhere not found at '$vswhere'. Install Visual Studio 2022 with the C++ desktop workload."
|
||||
exit 1
|
||||
}
|
||||
$msbuild_exe = & $vswhere -latest -products * -requires Microsoft.Component.MSBuild -find "MSBuild\**\Bin\MSBuild.exe" 2>$null | Select-Object -First 1
|
||||
if (-not $msbuild_exe) {
|
||||
write-error "MSBuild not found via vswhere. Install Visual Studio 2022 with the C++ desktop workload."
|
||||
exit 1
|
||||
}
|
||||
|
||||
$path_pcsx_sln = join-path $path_pcsx_redux 'vsprojects\pcsx-redux.sln'
|
||||
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
# NuGet restore — required before MSBuild.
|
||||
# pcsx-redux's .vcxproj files use the legacy packages.config style with
|
||||
# hardcoded `<Import Project="..\packages\{id}.{ver}\...">` directives.
|
||||
# MSBuild's `/t:Restore` won't fetch missing packages here (the local
|
||||
# packages\ dir is checked but no package-source lookup happens), and
|
||||
# `dotnet restore` errors on packages.config projects, so we walk every
|
||||
# packages.config, parse out the <package id version/> entries, and pull
|
||||
# any missing .nupkg directly from api.nuget.org's flat container.
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
$path_pcsx_packages = join-path $path_pcsx_redux 'vsprojects\packages'
|
||||
$nuget_flat_container = 'https://api.nuget.org/v3-flatcontainer'
|
||||
|
||||
# Collect required (id, version) pairs from every packages.config.
|
||||
$required_packages = @{}
|
||||
Get-ChildItem -Path (join-path $path_pcsx_redux 'vsprojects') -Filter 'packages.config' -Recurse -ErrorAction SilentlyContinue |
|
||||
ForEach-Object {
|
||||
[xml]$xml = Get-Content -LiteralPath $_.FullName -Raw
|
||||
foreach ($pkg in $xml.packages.package) {
|
||||
$key = '{0}|{1}' -f $pkg.id, $pkg.version
|
||||
$required_packages[$key] = @{ id = $pkg.id; version = $pkg.version }
|
||||
}
|
||||
}
|
||||
|
||||
# Ensure the packages root exists.
|
||||
if (-not (Test-Path -LiteralPath $path_pcsx_packages)) {
|
||||
New-Item -ItemType Directory -Path $path_pcsx_packages -Force | Out-Null
|
||||
}
|
||||
|
||||
# Download anything missing.
|
||||
# Skip the package entirely if its dir already has any contents (the legacy packages.config style means the targets file location varies per package
|
||||
# — `luajit.native` puts it at build/native/, `glfw` puts it elsewhere — so we can't probe a specific path; just check whether the dir is non-empty).
|
||||
Add-Type -AssemblyName System.IO.Compression.FileSystem
|
||||
foreach ($pkg in $required_packages.Values) {
|
||||
$pkgDir = Join-Path $path_pcsx_packages ('{0}.{1}' -f $pkg.id, $pkg.version)
|
||||
if ((Test-Path -LiteralPath $pkgDir) -and `
|
||||
(@(Get-ChildItem -LiteralPath $pkgDir -Recurse -ErrorAction SilentlyContinue).Count -gt 0)) {
|
||||
continue
|
||||
}
|
||||
$url = '{0}/{1}/{2}/{1}.{2}.nupkg' -f $nuget_flat_container, $pkg.id, $pkg.version
|
||||
$nupkg = Join-Path $pkgDir ('{0}.{1}.nupkg' -f $pkg.id, $pkg.version)
|
||||
New-Item -ItemType Directory -Path $pkgDir -Force | Out-Null
|
||||
Write-Host "Fetching NuGet package: $($pkg.id) $($pkg.version)"
|
||||
try {
|
||||
Invoke-WebRequest -Uri $url -OutFile $nupkg -UseBasicParsing -ErrorAction Stop
|
||||
[System.IO.Compression.ZipFile]::ExtractToDirectory($nupkg, $pkgDir)
|
||||
Remove-Item -LiteralPath $nupkg -Force
|
||||
} catch {
|
||||
$msg = $_.Exception.Message
|
||||
if ($msg -match '404') {
|
||||
Write-Host " Not on nuget.org (vendored?) — skipping $url"
|
||||
} else {
|
||||
Write-Warning "Failed to fetch $url — $msg"
|
||||
}
|
||||
if (Test-Path -LiteralPath $nupkg) { Remove-Item -LiteralPath $nupkg -Force }
|
||||
}
|
||||
}
|
||||
|
||||
# ════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════
|
||||
# isoffi.lua size guard — `core.vcxproj` #includes src/core/isoffi.lua into luaiso.cc via the `-- lualoader, R"EOF(...)EOF"` trick.
|
||||
# The raw string literal between R"EOF(-- and -- )EOF" must stay under ~16,379 bytes or MSVC (19.44) fails with C2026 (its actual raw-string limit is 16,384, minus 5 bytes for the `-- lualoader, ` prefix).
|
||||
# If the upstream file grows past that, trim it: remove license header, trailing whitespace, blank separators, inline comments, and shrink 4-space indent to 2-space.
|
||||
# Idempotent — only writes when the raw string exceeds the limit.
|
||||
# ════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════
|
||||
$path_isoffi = join-path $path_pcsx_redux 'src\core\isoffi.lua'
|
||||
if (Test-Path -LiteralPath $path_isoffi) {
|
||||
$content = Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8
|
||||
$startMarker = $content.IndexOf('R"EOF(--')
|
||||
$endMarker = $content.IndexOf('-- )EOF"')
|
||||
$literalLen = if ($startMarker -ge 0 -and $endMarker -gt $startMarker) { $endMarker - ($startMarker + 8) } else { -1 }
|
||||
# Effective MSVC raw-string limit for the lualoader prefix is 16379 bytes.
|
||||
if ($literalLen -gt 16379) {
|
||||
Write-Host "isoffi.lua raw string is $literalLen bytes (>16379); trimming for MSVC C2026 limit."
|
||||
$lines = $content -split "`n"
|
||||
$markerIdx = -1
|
||||
for ($i = 0; $i -lt $lines.Length; $i++) {
|
||||
if ($lines[$i] -match '^-- \)EOF"') { $markerIdx = $i; break }
|
||||
}
|
||||
$newLines = @()
|
||||
for ($i = 0; $i -lt $lines.Length; $i++) {
|
||||
$lineNum = $i + 1
|
||||
$line = $lines[$i]
|
||||
# Keep the first line and the EOF-marker line untouched.
|
||||
if ($i -eq 0 -or $i -eq $markerIdx) { $newLines += $line; continue }
|
||||
# Drop the GPL license header (lines 2-17).
|
||||
if ($lineNum -ge 2 -and $lineNum -le 17) { continue }
|
||||
# Drop blank separator lines.
|
||||
if ($line -match '^\s*$') { continue }
|
||||
# Drop trailing whitespace.
|
||||
$line = $line -replace '\s+$', ''
|
||||
# Drop inline comments (anything from `--` to end of line).
|
||||
$line = $line -replace '\s*--.*$', ''
|
||||
# Shrink 4-space indent to 2-space.
|
||||
$line = $line -replace '^( )', ' '
|
||||
if ($line -match '^\s*$') { continue }
|
||||
$newLines += $line
|
||||
}
|
||||
($newLines -join "`n") | Out-File -LiteralPath $path_isoffi -Encoding utf8 -NoNewline
|
||||
$newLen = ((Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8) -replace '.*R"EOF\(--', '' -replace '-- \)EOF".*', '').Length
|
||||
Write-Host "isoffi.lua trimmed: $literalLen -> $newLen bytes of raw string content."
|
||||
}
|
||||
}
|
||||
|
||||
& $msbuild_exe $path_pcsx_sln /p:Configuration=Release /p:Platform=x64 /p:PlatformToolset=v143 /m /v:minimal
|
||||
|
||||
# Locate luajit via scoop. `luajit.exe` is on PATH via scoop's shim;
|
||||
# we use `scoop prefix` to find the install root for the include dir (needed to compile lpeg against luajit's headers).
|
||||
# If scoop or luajit is missing, fail fast with an actionable message.
|
||||
$luajit_prefix = & scoop prefix luajit 2>$null
|
||||
if (-not $luajit_prefix -or -not (Test-Path (Join-Path $luajit_prefix 'bin/luajit.exe'))) {
|
||||
write-error "luajit not found via 'scoop prefix luajit'. Install via: scoop install luajit"
|
||||
exit 1
|
||||
}
|
||||
|
||||
# Discover the luajit include dir by globbing `include/luajit-*`.
|
||||
# This avoids hardcoding a specific version (e.g. `luajit-2.1`).
|
||||
$luajit_include_root = Join-Path $luajit_prefix 'include'
|
||||
$lua_inc_dir = Get-ChildItem -Path $luajit_include_root -Directory -Filter 'luajit-*' -ErrorAction SilentlyContinue |
|
||||
Select-Object -First 1 -ExpandProperty FullName
|
||||
if (-not $lua_inc_dir) {
|
||||
write-error "No 'luajit-*' include dir found under '$luajit_include_root'. The scoop luajit install may be broken."
|
||||
exit 1
|
||||
}
|
||||
|
||||
# Generate lpeg.dll by compiling the 6 source files directly.
|
||||
# `gcc` is on PATH (scoop's shim puts it there).
|
||||
# The source files: lpcap.c lpcode.c lpcset.c lpprint.c lptree.c lpvm.c
|
||||
# Link against luajit's import library (`libluajit-5.1.a`) for the Lua C API symbols (lua_*, luaL_*).
|
||||
$luajit_lib_dir = Join-Path $luajit_prefix 'lib'
|
||||
$lpeg_sources = @('lpcap.c', 'lpcode.c', 'lpcset.c', 'lpprint.c', 'lptree.c', 'lpvm.c')
|
||||
$lpeg_compile_args = @(
|
||||
'-O2', '-shared',
|
||||
"-I$lua_inc_dir",
|
||||
"-L$luajit_lib_dir",
|
||||
'-o', 'lpeg.dll'
|
||||
) + $lpeg_sources + @('-lluajit-5.1')
|
||||
push-location $path_lpeg
|
||||
& gcc @lpeg_compile_args
|
||||
pop-location
|
||||
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
# lfs (LuaFileSystem) — compiled from pcsx-redux's vendored luafilesystem source.
|
||||
# Source: toolchain/pcsx-redux/third_party/luafilesystem/src/lfs.c
|
||||
# Output: toolchain/lfs/lfs.dll
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
$path_lfs = join-path $path_toolchain 'lfs'
|
||||
verify-path $path_lfs
|
||||
$lfs_src = join-path $path_pcsx_redux 'third_party\luafilesystem\src\lfs.c'
|
||||
$lfs_dll = join-path $path_lfs 'lfs.dll'
|
||||
$lfs_dll_import = join-path $luajit_lib_dir 'libluajit-5.1.dll.a'
|
||||
& gcc -O2 -shared "-I$lua_inc_dir" -o $lfs_dll $lfs_src $lfs_dll_import
|
||||
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
# OpenBIOS — built from the PCSX-Redux source tree via make + mipsel-none-elf
|
||||
# Output: toolchain\pcsx-redux\src\mips\openbios\openbios.bin
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
$path_openbios = join-path $path_pcsx_redux 'src\mips\openbios'
|
||||
|
||||
# Wipe stale *.dep files across src\mips.
|
||||
# These cache absolute paths to the GCC headers directory; if the toolchain was upgraded (e.g. v14.2.0 → v16.1.0)
|
||||
# Make reads the stale paths and aborts with "no rule to make target .../stddef.h".
|
||||
# `make clean` in openbios only clears its own dir — subdirs like common/crt0/, modplayer/, and shell/ keep their stale .dep files.
|
||||
# Easier to just delete the lot before each build than to teach every Makefile about deepclean recursion.
|
||||
Get-ChildItem -Path (join-path $path_pcsx_redux 'src\mips') -Recurse -Filter '*.dep' -ErrorAction SilentlyContinue |
|
||||
ForEach-Object { Remove-Item -LiteralPath $_.FullName -Force }
|
||||
|
||||
push-location $path_openbios
|
||||
& make clean
|
||||
& make
|
||||
pop-location
|
||||
|
||||
@@ -1,246 +0,0 @@
|
||||
-- word_count_eval.lua
|
||||
--
|
||||
-- Word-counting logic for the tape-atom metaprogram pipeline.
|
||||
-- Used by:
|
||||
-- - passes/components.lua (compute_component_word_count)
|
||||
-- - passes/offsets.lua (scan_atom_body)
|
||||
-- - passes/annotation.lua (TAPE_WORDS <-> WORD_COUNT drift check)
|
||||
--
|
||||
-- This module ALSO exposes M.run(ctx) — the "word-counts" pass entry in
|
||||
-- the PASSES table — which loads metadata.h + scans for existing
|
||||
-- *.macs.h files into ctx.shared.word_counts.
|
||||
--
|
||||
-- Coding standard: tabs (1/level), EmmyLua annotations, no regex,
|
||||
-- Lua 5.3 compatible.
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Module-scope requires + package.path setup
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
local script_path = arg and arg[0] or "?"
|
||||
local last_sep = 0
|
||||
for i = 1, #script_path do
|
||||
local c = script_path:sub(i, i)
|
||||
if c == "/" or c == "\\" then last_sep = i end
|
||||
end
|
||||
local script_dir = last_sep == 0 and "./" or script_path:sub(1, last_sep)
|
||||
package.path = script_dir .. "?.lua;" .. script_dir .. "?/init.lua;" .. package.path
|
||||
package.cpath = "C:\\projects\\Pikuma\\ps1\\toolchain\\luajit-2.1\\lib\\lua\\5.1\\?.dll;" .. package.cpath
|
||||
|
||||
local duffle = require("duffle")
|
||||
local trim = duffle.trim
|
||||
local read_ident = duffle.read_ident
|
||||
local skip_ws_and_cmt = duffle.skip_ws_and_cmt
|
||||
local load_word_counts = duffle.load_word_counts
|
||||
local split_top_level_commas = duffle.split_top_level_commas
|
||||
local is_space = duffle.is_space
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Type declarations
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- @class WordCounts
|
||||
--- @field [string] integer -- macro name -> word count
|
||||
|
||||
--- @class PassCtx
|
||||
--- @field sources SourceFile[]
|
||||
--- @field metadata_path string
|
||||
--- @field shared table
|
||||
--- @field shared.word_counts WordCounts
|
||||
--- @field out_root string
|
||||
--- @field project_root string
|
||||
--- @field upstream table<string, table>
|
||||
--- @field flags table
|
||||
--- @field dry_run boolean
|
||||
--- @field verbose boolean
|
||||
|
||||
--- @class PassResult
|
||||
--- @field outputs table[]
|
||||
--- @field errors table[]
|
||||
--- @field warnings table[]
|
||||
|
||||
--- @class SourceFile
|
||||
--- @field path string
|
||||
--- @field text string
|
||||
--- @field dir string
|
||||
--- @field basename string
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Module exports
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
local M = {}
|
||||
|
||||
-- ┌────────────────────────────────────────────────────────────────────┐
|
||||
-- │ Shared utility: count_token_words │
|
||||
-- └────────────────────────────────────────────────────────────────────┘
|
||||
|
||||
--- Count words emitted by a single comma-separated token inside an atom body.
|
||||
--- For most tokens (regular MIPS instructions) this returns 1.
|
||||
--- For `mac_X(...)` calls, this returns the resolved word count from `wc`
|
||||
--- (recursively if needed). For `nop2` etc., returns wc[name].
|
||||
--- For unknown macros, returns 1 and (optionally) warns.
|
||||
---
|
||||
--- PORT NOTE: taken verbatim from tape_atom.offset_gen.meta.lua:130-141
|
||||
--- (`word_count_of_token`). Behavior is identical to preserve the
|
||||
--- branch-offset fix from commit 98e27c2.
|
||||
---
|
||||
--- @param token string -- a single token from split_top_level_commas
|
||||
--- @param wc WordCounts -- the shared word-count table
|
||||
--- @return integer
|
||||
function M.count_token_words(token, wc)
|
||||
local s = trim(token)
|
||||
if s == "" then return 0 end
|
||||
local name, after = read_ident(s, 1)
|
||||
if not name then return 1 end
|
||||
if wc[name] then return wc[name] end
|
||||
local j = skip_ws_and_cmt(s, after)
|
||||
if s:sub(j, j) == "(" then
|
||||
io.stderr:write(" warning: unknown macro '" .. name .. "', assuming 1 word\n")
|
||||
end
|
||||
return 1
|
||||
end
|
||||
|
||||
-- ┌────────────────────────────────────────────────────────────────────┐
|
||||
-- │ Shared utility: scan_dir │
|
||||
-- └────────────────────────────────────────────────────────────────────┘
|
||||
|
||||
--- Recursively scan a directory for files matching a glob suffix.
|
||||
--- No regex per the no_regex constraint — uses plain byte matching
|
||||
--- via `dir /b /s` on Windows.
|
||||
---
|
||||
--- PORT NOTE: taken from tape_atom.offset_gen.meta.lua:432-443
|
||||
--- (`scan_dir`). Adapted: removed the hardcoded project_root derivation;
|
||||
--- the caller passes `dir` explicitly.
|
||||
---
|
||||
--- @param dir string -- directory to scan (absolute or relative)
|
||||
--- @param suffix string -- file pattern, e.g. "*.macs.h"
|
||||
--- @return string[]
|
||||
function M.scan_dir(dir, suffix)
|
||||
local results = {}
|
||||
local p = io.popen('dir /b /s "' .. dir .. '\\' .. suffix .. '" 2>nul')
|
||||
if not p then return results end
|
||||
for raw_line in p:lines() do
|
||||
local path = raw_line:gsub("\\", "/")
|
||||
results[#results + 1] = path
|
||||
end
|
||||
p:close()
|
||||
return results
|
||||
end
|
||||
|
||||
-- ┌────────────────────────────────────────────────────────────────────┐
|
||||
-- │ Shared utility: count_body_words │
|
||||
-- └────────────────────────────────────────────────────────────────────┘
|
||||
|
||||
--- Count words emitted by an entire atom body (a brace-delimited block).
|
||||
--- Splits by top-level commas; for each token, delegates to count_token_words.
|
||||
--- Handles `atom_label(name)` / `atom_offset(tag, name)` markers (record at
|
||||
--- current pos, do NOT advance pos; if the marker call bundles an instruction
|
||||
--- after it, count that instruction too).
|
||||
---
|
||||
--- PORT NOTE: taken verbatim from tape_atom.offset_gen.meta.lua:207-239
|
||||
--- (`scan_atom_body`). Behavior is identical to preserve the branch-offset
|
||||
--- fix from commit 98e27c2.
|
||||
---
|
||||
--- @param body string -- brace-delimited atom body (without braces)
|
||||
--- @param wc WordCounts -- the shared word-count table
|
||||
--- @return integer -- total words
|
||||
function M.count_body_words(body, wc)
|
||||
local pos = 0
|
||||
for _, tok in ipairs(split_top_level_commas(body)) do
|
||||
local k = 1
|
||||
local tlen = #tok
|
||||
while k <= tlen and is_space(tok:sub(k, k)) do k = k + 1 end
|
||||
local leading_ident = read_ident(tok, k)
|
||||
if leading_ident == "atom_label" or leading_ident == "atom_offset" then
|
||||
-- Marker call: record at current pos, do NOT advance pos.
|
||||
-- But the source pattern may bundle the marker with the next
|
||||
-- instruction on a new line (no top-level comma between them).
|
||||
-- In that case, the rest of `tok` after the marker call is
|
||||
-- a real instruction that must still be counted.
|
||||
local marker_end = M.find_marker_call_end(tok)
|
||||
if marker_end > 0 and marker_end < #tok then
|
||||
local rest = trim(tok:sub(marker_end + 1))
|
||||
if rest ~= "" then
|
||||
local rest_words = M.count_token_words(rest, wc)
|
||||
pos = pos + rest_words
|
||||
end
|
||||
end
|
||||
else
|
||||
pos = pos + M.count_token_words(tok, wc)
|
||||
end
|
||||
end
|
||||
return pos
|
||||
end
|
||||
|
||||
--- Find the end position (just past the closing ')') of the first
|
||||
--- atom_label/atom_offset call in `tok`. Returns 0 if no such call.
|
||||
--- Internal helper for count_body_words.
|
||||
---
|
||||
--- PORT NOTE: taken from tape_atom.offset_gen.meta.lua:181-205
|
||||
--- (`find_marker_call_end`).
|
||||
---
|
||||
--- @param tok string
|
||||
--- @return integer -- 0 if no marker call found
|
||||
function M.find_marker_call_end(tok)
|
||||
local i = 1
|
||||
local len = #tok
|
||||
while i <= len do
|
||||
i = skip_ws_and_cmt(tok, i)
|
||||
if i > len then break end
|
||||
local c = tok:sub(i, i)
|
||||
if is_space(c) then
|
||||
i = i + 1
|
||||
elseif c == "/" then
|
||||
-- comment — skip past it (delegated to duffle.skip_str_or_cmt)
|
||||
local nx = duffle.skip_str_or_cmt(tok, i)
|
||||
if nx > i then i = nx else i = i + 1 end
|
||||
else
|
||||
local ident, after = read_ident(tok, i)
|
||||
if ident == "atom_label" or ident == "atom_offset" then
|
||||
local j = skip_ws_and_cmt(tok, after)
|
||||
if tok:sub(j, j) == "(" then
|
||||
local _, end_paren = duffle.read_parens(tok, j)
|
||||
return end_paren - 1
|
||||
end
|
||||
return 0
|
||||
end
|
||||
i = after or (i + 1)
|
||||
end
|
||||
end
|
||||
return 0
|
||||
end
|
||||
|
||||
-- ┌────────────────────────────────────────────────────────────────────┐
|
||||
-- │ Pass entry: M.run(ctx) — "word-counts" pass │
|
||||
-- └────────────────────────────────────────────────────────────────────┘
|
||||
|
||||
--- Load metadata.h + scan for existing *.macs.h files into
|
||||
--- ctx.shared.word_counts. Loading the .macs.h files is idempotent:
|
||||
--- entries from later (current-build) .macs.h files override
|
||||
--- metadata.h entries of the same name.
|
||||
---
|
||||
--- @param ctx PassCtx
|
||||
--- @return PassResult
|
||||
function M.run(ctx)
|
||||
local wc = {}
|
||||
|
||||
-- 1. Load metadata.h (the encoding-macro source of truth).
|
||||
local meta_counts = load_word_counts(ctx.metadata_path)
|
||||
for name, count in pairs(meta_counts) do wc[name] = count end
|
||||
|
||||
-- 2. Scan project_root recursively for *.macs.h files (component-macro source).
|
||||
local macs_files = M.scan_dir(ctx.project_root, "*.macs.h")
|
||||
for _, macs_path in ipairs(macs_files) do
|
||||
local ok, mc = pcall(load_word_counts, macs_path)
|
||||
if ok and type(mc) == "table" then
|
||||
for name, count in pairs(mc) do wc[name] = count end
|
||||
end
|
||||
end
|
||||
|
||||
ctx.shared.word_counts = wc
|
||||
|
||||
return { outputs = {}, errors = {}, warnings = {} }
|
||||
end
|
||||
|
||||
return M
|
||||
Reference in New Issue
Block a user