mirror of
https://github.com/Ed94/pikuma_ps1.git
synced 2026-09-10 18:29:06 +00:00
Compare commits
185
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a37ffe6f58 | ||
|
|
b61610d819 | ||
|
|
1b950ab5b5 | ||
|
|
b2858b3c73 | ||
|
|
f1801343e2 | ||
|
|
85b2205603 | ||
|
|
2d754650c9 | ||
|
|
e2ffe538b6 | ||
|
|
223d1832eb | ||
|
|
de13bc3ce9 | ||
|
|
2a087f735e | ||
|
|
449216967b | ||
|
|
c226e8a7d3 | ||
|
|
81f37e0098 | ||
|
|
bde829bf59 | ||
|
|
cf78cfa120 | ||
|
|
3440c9b59e | ||
|
|
1cbddc6708 | ||
|
|
b345ccd60e | ||
|
|
bbda5efaea | ||
|
|
290bb0e07a | ||
|
|
86fe189b4e | ||
|
|
da007d342e | ||
|
|
5a4bfb1224 | ||
|
|
d4795cf9de | ||
|
|
e79c364b40 | ||
|
|
18b1d5a04b | ||
|
|
581b00b960 | ||
|
|
3faccfc283 | ||
|
|
1a0d417649 | ||
|
|
3301826f5c | ||
|
|
d9b9241e2c | ||
|
|
a16c727db2 | ||
|
|
8a825a59c7 | ||
|
|
f8b28be02e | ||
|
|
ffc66052f8 | ||
|
|
7764612325 | ||
|
|
1a5b618484 | ||
|
|
d23b6a2a36 | ||
|
|
7ec778a68e | ||
|
|
9ca865d5db | ||
|
|
764ded4557 | ||
|
|
67a84d34f3 | ||
|
|
baaff12f33 | ||
|
|
b695056b9a | ||
|
|
3a4d6304dd | ||
|
|
a535d381ed | ||
|
|
c447bfa877 | ||
|
|
d88e0d0487 | ||
|
|
9a6eca6047 | ||
|
|
5c9c61720f | ||
|
|
b8e31123e4 | ||
|
|
ea3e30a11e | ||
|
|
37f4712237 | ||
|
|
b699b47b28 | ||
|
|
640dab7e61 | ||
|
|
4688566767 | ||
|
|
5ebaa6e083 | ||
|
|
7f0bdefbcb | ||
|
|
d5f28b83ea | ||
|
|
3ea3e8d105 | ||
|
|
6b60cef2e8 | ||
|
|
77f19321cd | ||
|
|
2e07665920 | ||
|
|
9501bbbcc2 | ||
|
|
9b6b5535f5 | ||
|
|
7807047dc0 | ||
|
|
7daeec0ee3 | ||
|
|
3f3b691ac0 | ||
|
|
a2d79d65eb | ||
|
|
bebcc6a585 | ||
|
|
ece21ed368 | ||
|
|
144c605ad8 | ||
|
|
4afd1af0fd | ||
|
|
004a7eff19 | ||
|
|
e42c75a26a | ||
|
|
69f2c0d036 | ||
|
|
b045856dd6 | ||
|
|
68b87f1c8b | ||
|
|
917b764d95 | ||
|
|
773aa44013 | ||
|
|
2b6fe53ce8 | ||
|
|
01f7ceba7c | ||
|
|
6f2eff920d | ||
|
|
f25765a7b7 | ||
|
|
2757aa4330 | ||
|
|
748b58c5c5 | ||
|
|
6441dbc23e | ||
|
|
57fdb9e037 | ||
|
|
b5953a723b | ||
|
|
888ffce859 | ||
|
|
7289e7c89c | ||
|
|
54a5bb9a31 | ||
|
|
e0f4ac873d | ||
|
|
f17fa9165e | ||
|
|
8282f8e902 | ||
|
|
9eb696ece8 | ||
|
|
858e57f293 | ||
|
|
afcd9b86f0 | ||
|
|
43cd4e0344 | ||
|
|
09dde54030 | ||
|
|
315e1b2c5e | ||
|
|
02658d3609 | ||
|
|
dbc459b7e0 | ||
|
|
a704341fc6 | ||
|
|
7421b32fd7 | ||
|
|
e2eb74be19 | ||
|
|
338f1fe46e | ||
|
|
27a9038e0d | ||
|
|
8c8d2e54aa | ||
|
|
80a35aa23a | ||
|
|
f247d56c32 | ||
|
|
590ff1e2ec | ||
|
|
653e18ee28 | ||
|
|
ebb876fe89 | ||
|
|
1b40b16c0e | ||
|
|
9ffd6592bc | ||
|
|
d56adab38f | ||
|
|
08af73d0d2 | ||
|
|
67d54debfa | ||
|
|
3c25306070 | ||
|
|
c3cf05950e | ||
|
|
f6b4d9895e | ||
|
|
e70361b548 | ||
|
|
ed3eb45b1d | ||
|
|
d7770b6e1d | ||
|
|
137549b1c8 | ||
|
|
7d5b13aadb | ||
|
|
2d901003f9 | ||
|
|
b43d22008e | ||
|
|
904889b483 | ||
|
|
f7aa7b75e7 | ||
|
|
aca6e30e20 | ||
|
|
8b0fb1d4e4 | ||
|
|
9f7a4a00ce | ||
|
|
277af1c901 | ||
|
|
97d2f66c5a | ||
|
|
d9406553b3 | ||
|
|
e662d175ab | ||
|
|
5387a07b84 | ||
|
|
65d805e3ba | ||
|
|
987f4dee1e | ||
|
|
df723c691d | ||
|
|
45ac85c038 | ||
|
|
072231c46b | ||
|
|
2b00956862 | ||
|
|
1ffad6cf98 | ||
|
|
318516a354 | ||
|
|
91a91b3495 | ||
|
|
a0d22700db | ||
|
|
51bdf7106b | ||
|
|
531e1cbd58 | ||
|
|
541e52de2b | ||
|
|
eccf17d21c | ||
|
|
0d94632edf | ||
|
|
798807a9c2 | ||
|
|
e9f26f89b8 | ||
|
|
a226b45d18 | ||
|
|
c22e4baa41 | ||
|
|
fa598a41c6 | ||
|
|
a928d06ac9 | ||
|
|
91c2218471 | ||
|
|
27a5f8029f | ||
|
|
7a168137fc | ||
|
|
2ceb2f2a05 | ||
|
|
6103f47f05 | ||
|
|
c824c998eb | ||
|
|
9d066ae292 | ||
|
|
c9b7f8c08b | ||
|
|
59903546d7 | ||
|
|
1ffdda45e5 | ||
|
|
1209172649 | ||
|
|
98e27c2815 | ||
|
|
88aa1b8b59 | ||
|
|
ca3dc4aff0 | ||
|
|
1fb4883138 | ||
|
|
0ad609e7c2 | ||
|
|
ccdf1b832b | ||
|
|
4d177bc34d | ||
|
|
32a754cd06 | ||
|
|
407c7d352a | ||
|
|
8541713d0c | ||
|
|
602a0b46d8 | ||
|
|
74f390c3b1 | ||
|
|
5e7da32387 |
+9
-2
@@ -1,8 +1,9 @@
|
|||||||
build
|
build
|
||||||
toolchain/armips
|
toolchain/armips
|
||||||
|
toolchain/luajit-2.1
|
||||||
toolchain/pcsx-redux
|
toolchain/pcsx-redux
|
||||||
# toolchain/psyq_iwyu
|
toolchain/psyq_iwyu
|
||||||
# toolchain/PSn00bSDK
|
toolchain/PSn00bSDK
|
||||||
|
|
||||||
*.exe
|
*.exe
|
||||||
*.elf
|
*.elf
|
||||||
@@ -14,3 +15,9 @@ toolchain/pcsx-redux
|
|||||||
*.a
|
*.a
|
||||||
.sentry-native
|
.sentry-native
|
||||||
.vscode/settings.json
|
.vscode/settings.json
|
||||||
|
toolchain/lfs
|
||||||
|
toolchain/lpeg
|
||||||
|
|
||||||
|
scratch
|
||||||
|
toolchain/libpsn00b
|
||||||
|
scripts/pcsx_debug_helper.zip
|
||||||
|
|||||||
Vendored
+26
@@ -0,0 +1,26 @@
|
|||||||
|
# Cozy and Windy
|
||||||
|
|
||||||
|
Editor theme ported from the Rider scheme of the same name.
|
||||||
|
|
||||||
|
It colors the editor surface, C/C++ syntax, and tape-atom DSL keywords
|
||||||
|
emitted by `local.tape-atom-syntax`. It does not change workbench chrome.
|
||||||
|
|
||||||
|
## Install
|
||||||
|
|
||||||
|
```powershell
|
||||||
|
cd C:\projects\Pikuma\ps1\.vscode\cozy-and-windy
|
||||||
|
npm run package
|
||||||
|
code --install-extension .\cozy-and-windy-0.1.0.vsix --force
|
||||||
|
```
|
||||||
|
|
||||||
|
Reload the window. Select **Cozy and Windy** as the color theme, or set
|
||||||
|
`workbench.colorTheme` to `Cozy and Windy` in the PS1 workspace settings.
|
||||||
|
|
||||||
|
Keep `local.tape-atom-syntax` installed. This theme colors those token
|
||||||
|
types; it does not classify them.
|
||||||
|
|
||||||
|
## Inspect
|
||||||
|
|
||||||
|
Open `hello_camera.atom.c` and run **Developer: Inspect Editor Tokens and Scopes**
|
||||||
|
on `MipsAtom_`, an atom name, `atom_info`, `R_PrimCursor`, a `gte_*` call,
|
||||||
|
and a `mac_*` call.
|
||||||
Binary file not shown.
Vendored
+25
@@ -0,0 +1,25 @@
|
|||||||
|
{
|
||||||
|
"name": "cozy-and-windy",
|
||||||
|
"displayName": "Cozy and Windy",
|
||||||
|
"description": "Editor theme ported from the Rider Cozy and Windy scheme. Colors C/C++ and tape-atom DSL keywords.",
|
||||||
|
"publisher": "local",
|
||||||
|
"version": "0.1.0",
|
||||||
|
"engines": {
|
||||||
|
"vscode": "^1.80.0"
|
||||||
|
},
|
||||||
|
"categories": [
|
||||||
|
"Themes"
|
||||||
|
],
|
||||||
|
"scripts": {
|
||||||
|
"package": "npx --yes @vscode/vsce@3.6.1 package --allow-missing-repository --skip-license --out cozy-and-windy-0.1.0.vsix"
|
||||||
|
},
|
||||||
|
"contributes": {
|
||||||
|
"themes": [
|
||||||
|
{
|
||||||
|
"label": "Cozy and Windy",
|
||||||
|
"uiTheme": "vs-dark",
|
||||||
|
"path": "./themes/cozy-and-windy-color-theme.json"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,132 @@
|
|||||||
|
{
|
||||||
|
"name": "Cozy and Windy",
|
||||||
|
"type": "dark",
|
||||||
|
"semanticHighlighting": true,
|
||||||
|
"colors": {
|
||||||
|
// 121212
|
||||||
|
// 111212
|
||||||
|
// 211f1e
|
||||||
|
// 191817
|
||||||
|
"editor.background": "#191817",
|
||||||
|
"editor.foreground": "#dfc6ba",
|
||||||
|
"editor.lineHighlightBackground": "#1c1c1c",
|
||||||
|
"editor.selectionBackground": "#164371",
|
||||||
|
"editor.selectionForeground": "#c8c8c8",
|
||||||
|
"editorLineNumber.foreground": "#43c3c3",
|
||||||
|
"editorLineNumber.activeForeground": "#00fff4",
|
||||||
|
"editorIndentGuide.background1": "#181818",
|
||||||
|
"editorIndentGuide.activeBackground1": "#202020",
|
||||||
|
"editorRuler.foreground": "#505050",
|
||||||
|
"editorGutter.background": "#211f1e",
|
||||||
|
"editorBracketMatch.background": "#3b514d",
|
||||||
|
"editor.foldBackground": "#0c0c0c6a",
|
||||||
|
"editor.wordHighlightBackground": "#211f1e4d",
|
||||||
|
"editor.wordHighlightStrongBackground": "#303030",
|
||||||
|
"editorCursor.foreground": "#00fff4",
|
||||||
|
"editorWhitespace.foreground": "#181818",
|
||||||
|
// "editorLineHighlightBorder": "#1c1c1c",
|
||||||
|
"editorWidget.background": "#211f1e",
|
||||||
|
"editorSuggestWidget.background": "#2c334b",
|
||||||
|
"editorHoverWidget.background": "#2c334b"
|
||||||
|
},
|
||||||
|
"semanticTokenColors": {
|
||||||
|
"comment": { "foreground": "#868686", "fontStyle": "italic" },
|
||||||
|
"keyword": { "foreground": "#d8bd5b" },
|
||||||
|
"string": { "foreground": "#d46a54" },
|
||||||
|
"number": { "foreground": "#b5cea8" },
|
||||||
|
"operator": { "foreground": "#be8e78" },
|
||||||
|
"class": { "foreground": "#54a4d6" },
|
||||||
|
"struct": { "foreground": "#54a4d6" },
|
||||||
|
"enum": { "foreground": "#54a4d6" },
|
||||||
|
"type": { "foreground": "#54a4d6" },
|
||||||
|
"interface": { "foreground": "#7984ab" },
|
||||||
|
"function": { "foreground": "#cccab5" },
|
||||||
|
// "function": { "foreground": "#6090a9" },
|
||||||
|
"method": { "foreground": "#6090a9" },
|
||||||
|
"variable": { "foreground": "#bc966c" },
|
||||||
|
"parameter": { "foreground": "#ce8365" },
|
||||||
|
"property": { "foreground": "#acb8c8" },
|
||||||
|
"*.static": { "foreground": "#9e95c6" },
|
||||||
|
"macro": { "foreground": "#5ea852" },
|
||||||
|
"namespace": { "foreground": "#8e8e8e" },
|
||||||
|
"typeParameter": { "foreground": "#b8d7a3" },
|
||||||
|
"enumMember": { "foreground": "#a373b0" },
|
||||||
|
"label": { "foreground": "#c8c8c8", "fontStyle": "bold" },
|
||||||
|
"tapeAtomKeyword": { "foreground": "#d8bd5b", "fontStyle": "bold" },
|
||||||
|
"tapeAtomName": { "foreground": "#b1b7d6", "fontStyle": "bold" },
|
||||||
|
"tapeComponentKeyword": { "foreground": "#d68a36", "fontStyle": "bold" },
|
||||||
|
"tapeComponentName": { "foreground": "#b1b7d6", "fontStyle": "bold" },
|
||||||
|
"tapeAnnotation": { "foreground": "#d8bd5b" },
|
||||||
|
"tapeBindType": { "foreground": "#54a4d6" },
|
||||||
|
"tapePhase": { "foreground": "#b8d7a3", "fontStyle": "italic" },
|
||||||
|
"tapeLabel": { "foreground": "#959595", "fontStyle": "bold" },
|
||||||
|
// "tapeCpuInstruction": { "foreground": "#6d9aa0" },
|
||||||
|
// "tapeCpuInstruction": { "foreground": "#cf7539" },
|
||||||
|
// "tapeCpuInstruction": { "foreground": "#d16b3a" },
|
||||||
|
"tapeCpuInstruction": { "foreground": "#d5895a" },
|
||||||
|
"tapeGteInstruction": { "foreground": "#988bcb" },
|
||||||
|
"tapeGpuInstruction": { "foreground": "#bf7dac" },
|
||||||
|
"tapeComponentInstruction": { "foreground": "#8baa5d" },
|
||||||
|
// "tapeGprRegister": { "foreground": "#92d4d9" },
|
||||||
|
"tapeGprRegister": { "foreground": "#a2bfa8" },
|
||||||
|
"tapeCop2Register": { "foreground": "#945cd9" },
|
||||||
|
"tapeDuffleType": { "foreground": "#54a4d6" },
|
||||||
|
"tapeAttribute": { "foreground": "#73a07c" },
|
||||||
|
"tapeGprRegister.tapeRead": { "foreground": "#5bb8b0", "fontStyle": "italic" },
|
||||||
|
"tapeGprRegister.tapeWrite": { "foreground": "#2d8f8c", "fontStyle": "bold" },
|
||||||
|
"tapeCop2Register.tapeRead": { "foreground": "#b08ae0", "fontStyle": "italic" },
|
||||||
|
"tapeCop2Register.tapeWrite": { "foreground": "#7b3ec4", "fontStyle": "bold" },
|
||||||
|
// "*.tapeAuto": { },
|
||||||
|
"tapeControlFlow": { "foreground": "#63d169", "fontStyle": "bold" },
|
||||||
|
"tapeDelaySlot": { "foreground": "#ff5647" }
|
||||||
|
},
|
||||||
|
"tokenColors": [
|
||||||
|
{ "scope": ["comment", "comment.block", "comment.line", "comment.block.documentation"], "settings": { "foreground": "#868686", "fontStyle": "italic" } },
|
||||||
|
{ "scope": ["keyword", "keyword.control", "keyword.other"], "settings": { "foreground": "#d8bd5b" } },
|
||||||
|
{ "scope": ["string", "string.quoted"], "settings": { "foreground": "#d46a54" } },
|
||||||
|
{ "scope": ["string.quoted.other"], "settings": { "foreground": "#d69d85" } },
|
||||||
|
{ "scope": ["constant.numeric"], "settings": { "foreground": "#b5cea8" } },
|
||||||
|
{ "scope": ["punctuation", "keyword.operator"], "settings": { "foreground": "#be8e78" } },
|
||||||
|
{ "scope": ["keyword.operator.overload"], "settings": { "foreground": "#b87e76" } },
|
||||||
|
{ "scope": ["entity.name.type", "entity.name.type.class", "entity.name.type.struct", "entity.name.type.enum"], "settings": { "foreground": "#54a4d6" } },
|
||||||
|
{ "scope": ["entity.name.type.interface"], "settings": { "foreground": "#7984ab" } },
|
||||||
|
{ "scope": ["entity.name.function"], "settings": { "foreground": "#cccab5" } },
|
||||||
|
{ "scope": ["entity.name.function.member"], "settings": { "foreground": "#6090a9" } },
|
||||||
|
{ "scope": ["variable.other.local"], "settings": { "foreground": "#bc966c" } },
|
||||||
|
{ "scope": ["variable.parameter"], "settings": { "foreground": "#ce8365" } },
|
||||||
|
{ "scope": ["variable.other.property"], "settings": { "foreground": "#acb8c8" } },
|
||||||
|
{ "scope": ["variable.other.constant"], "settings": { "foreground": "#9e95c6" } },
|
||||||
|
{ "scope": ["variable.other.global", "variable.other.defaultLibrary"], "settings": { "foreground": "#bf7dac" } },
|
||||||
|
{ "scope": ["support.type", "support.function"], "settings": { "foreground": "#8baa5d" } },
|
||||||
|
{ "scope": ["meta.preprocessor"], "settings": { "foreground": "#5ea852" } },
|
||||||
|
// { "scope": ["variable.parameter.preprocessor"], "settings": { "foreground": "#636363" } },
|
||||||
|
{ "scope": ["variable.parameter.preprocessor"], "settings": { "foreground": "#bc966c" } },
|
||||||
|
{ "scope": ["keyword.control.directive"], "settings": { "foreground": "#d68a36" } },
|
||||||
|
{ "scope": ["entity.name.namespace"], "settings": { "foreground": "#8e8e8e" } },
|
||||||
|
{ "scope": ["entity.name.type.parameter"], "settings": { "foreground": "#b8d7a3" } },
|
||||||
|
{ "scope": ["variable.other.enummember"], "settings": { "foreground": "#a373b0" } },
|
||||||
|
{ "scope": ["entity.name.type.concept"], "settings": { "foreground": "#76ff7d" } },
|
||||||
|
{ "scope": ["entity.name.type.dependent"], "settings": { "foreground": "#448b5a", "fontStyle": "bold" } },
|
||||||
|
{ "scope": ["entity.name.label"], "settings": { "foreground": "#c8c8c8", "fontStyle": "bold" } },
|
||||||
|
{ "scope": ["invalid"], "settings": { "foreground": "#ff5647" } },
|
||||||
|
{ "scope": ["keyword.codetag.todo"], "settings": { "foreground": "#c10000", "fontStyle": "bold italic" } },
|
||||||
|
{ "scope": ["keyword.control.duffle.atom"], "settings": { "foreground": "#d8bd5b", "fontStyle": "bold" } },
|
||||||
|
{ "scope": ["entity.name.function.duffle.atom"], "settings": { "foreground": "#cccab5", "fontStyle": "bold" } },
|
||||||
|
{ "scope": ["keyword.control.duffle.component"], "settings": { "foreground": "#d68a36", "fontStyle": "bold" } },
|
||||||
|
{ "scope": ["entity.name.function.duffle.component"], "settings": { "foreground": "#6090a9", "fontStyle": "bold" } },
|
||||||
|
{ "scope": ["support.function.duffle.annotation"], "settings": { "foreground": "#d8bd5b" } },
|
||||||
|
{ "scope": ["entity.name.type.duffle.bind"], "settings": { "foreground": "#54a4d6" } },
|
||||||
|
{ "scope": ["entity.name.tag.duffle.phase"], "settings": { "foreground": "#b8d7a3", "fontStyle": "italic" } },
|
||||||
|
{ "scope": ["entity.name.label.duffle.atom"], "settings": { "foreground": "#c8c8c8", "fontStyle": "bold" } },
|
||||||
|
{ "scope": ["support.function.duffle.cpu"], "settings": { "foreground": "#6d9aa0" } },
|
||||||
|
{ "scope": ["support.function.duffle.gte"], "settings": { "foreground": "#988bcb" } },
|
||||||
|
{ "scope": ["support.function.duffle.gpu"], "settings": { "foreground": "#bf7dac" } },
|
||||||
|
{ "scope": ["support.function.duffle.component"], "settings": { "foreground": "#8baa5d" } },
|
||||||
|
{ "scope": ["keyword.control.duffle.branch"], "settings": { "foreground": "#76ff7d", "fontStyle": "bold" } },
|
||||||
|
{ "scope": ["keyword.operator.duffle.delayslot"], "settings": { "foreground": "#ff5647" } },
|
||||||
|
{ "scope": ["variable.other.constant.duffle.gpr"], "settings": { "foreground": "#3fa8a6" } },
|
||||||
|
{ "scope": ["variable.other.constant.duffle.cop2"], "settings": { "foreground": "#945cd9" } },
|
||||||
|
{ "scope": ["storage.type.duffle.type"], "settings": { "foreground": "#54a4d6" } },
|
||||||
|
{ "scope": ["storage.modifier.duffle.attr"], "settings": { "foreground": "#73a07c" } }
|
||||||
|
]
|
||||||
|
}
|
||||||
Vendored
+43
@@ -0,0 +1,43 @@
|
|||||||
|
# Package and install the local VS Code Insiders extensions under .vscode/.
|
||||||
|
# Usage:
|
||||||
|
# .\install_extensions.ps1
|
||||||
|
# .\install_extensions.ps1 -SkipPackage
|
||||||
|
|
||||||
|
param([switch] $SkipPackage)
|
||||||
|
|
||||||
|
$path_vscode = $PSScriptRoot
|
||||||
|
$code_insiders = "C:\apps\Microsoft VS Code Insiders\bin\code-insiders.cmd"
|
||||||
|
if (-not (test-path -literalpath $code_insiders)) {
|
||||||
|
$found = get-command code-insiders -erroraction silentlycontinue
|
||||||
|
if ($found) { $code_insiders = $found.source }
|
||||||
|
}
|
||||||
|
|
||||||
|
if (-not (test-path -literalpath $code_insiders)) { throw "code-insiders not found. Install VS Code Insiders or add it to PATH." }
|
||||||
|
|
||||||
|
$extensions = @(
|
||||||
|
(join-path $path_vscode "tape-atom-syntax"),
|
||||||
|
(join-path $path_vscode "cozy-and-windy")
|
||||||
|
)
|
||||||
|
|
||||||
|
foreach ($extension in $extensions) {
|
||||||
|
$package_json = join-path $extension "package.json"
|
||||||
|
if (-not (test-path -literalpath $package_json)) { throw "missing $package_json" }
|
||||||
|
|
||||||
|
$manifest = get-content -literalpath $package_json -raw | convertfrom-json
|
||||||
|
$vsix = join-path $extension ("{0}-{1}.vsix" -f $manifest.name, $manifest.version)
|
||||||
|
|
||||||
|
if (-not $SkipPackage) {
|
||||||
|
if (-not $manifest.scripts.package) { throw "$package_json has no scripts.package" }
|
||||||
|
write-host "packaging $($manifest.displayName) ($($manifest.name)@$($manifest.version))"
|
||||||
|
& npm --prefix $extension run package
|
||||||
|
if ($LASTEXITCODE -ne 0) { throw "npm run package failed for $extension" }
|
||||||
|
}
|
||||||
|
|
||||||
|
if (-not (test-path -literalpath $vsix)) { throw "missing $vsix" }
|
||||||
|
|
||||||
|
write-host "installing $vsix"
|
||||||
|
& $code_insiders --install-extension $vsix --force
|
||||||
|
if ($LASTEXITCODE -ne 0) { throw "install failed for $vsix" }
|
||||||
|
}
|
||||||
|
|
||||||
|
write-host "done. reload the Insiders window (Developer: Reload Window)."
|
||||||
Vendored
+130
-27
@@ -4,7 +4,7 @@
|
|||||||
// For more information, visit: https://go.microsoft.com/fwlink/?linkid=830387
|
// For more information, visit: https://go.microsoft.com/fwlink/?linkid=830387
|
||||||
"version": "0.2.0",
|
"version": "0.2.0",
|
||||||
"configurations": [
|
"configurations": [
|
||||||
{
|
{
|
||||||
"name": "Debug: Hello Psy-Q!",
|
"name": "Debug: Hello Psy-Q!",
|
||||||
"type": "gdb",
|
"type": "gdb",
|
||||||
"request": "attach",
|
"request": "attach",
|
||||||
@@ -12,6 +12,10 @@
|
|||||||
"remote": true,
|
"remote": true,
|
||||||
"cwd": "${workspaceRoot}/build",
|
"cwd": "${workspaceRoot}/build",
|
||||||
"valuesFormatting": "parseText",
|
"valuesFormatting": "parseText",
|
||||||
|
"registerLimit": "1-32",
|
||||||
|
"frameFilters": false,
|
||||||
|
"showDevDebugOutput": false,
|
||||||
|
"printCalls": false,
|
||||||
"stopAtConnect": true,
|
"stopAtConnect": true,
|
||||||
"gdbpath": "gdb-multiarch",
|
"gdbpath": "gdb-multiarch",
|
||||||
"windows": {
|
"windows": {
|
||||||
@@ -20,10 +24,17 @@
|
|||||||
"osx": {
|
"osx": {
|
||||||
"gdbpath": "gdb"
|
"gdbpath": "gdb"
|
||||||
},
|
},
|
||||||
"executable": "${workspaceRoot}/build/hello_psyq.elf",
|
"executable": "${workspaceRoot}/build/hello_gte.elf",
|
||||||
|
"setupCommands": [
|
||||||
|
{ "text": "set mi-async off" },
|
||||||
|
{ "text": "set remotetimeout 0" },
|
||||||
|
{ "text": "set logging file build/gen/hello_gte.gdb.log" },
|
||||||
|
{ "text": "set logging redirect on" }
|
||||||
|
],
|
||||||
"autorun": [
|
"autorun": [
|
||||||
"monitor reset shellhalt",
|
"monitor reset shellhalt",
|
||||||
"load hello_psyq.elf",
|
"load hello_gte.elf",
|
||||||
|
"source scripts/gdb/gdb_tape_atoms.gdb",
|
||||||
"tbreak main",
|
"tbreak main",
|
||||||
"continue"
|
"continue"
|
||||||
]
|
]
|
||||||
@@ -36,30 +47,10 @@
|
|||||||
"remote": true,
|
"remote": true,
|
||||||
"cwd": "${workspaceRoot}/build",
|
"cwd": "${workspaceRoot}/build",
|
||||||
"valuesFormatting": "parseText",
|
"valuesFormatting": "parseText",
|
||||||
"stopAtConnect": true,
|
"registerLimit": "1-32",
|
||||||
"gdbpath": "gdb-multiarch",
|
"frameFilters": false,
|
||||||
"windows": {
|
"showDevDebugOutput": false,
|
||||||
"gdbpath": "gdb-multiarch.exe"
|
"printCalls": false,
|
||||||
},
|
|
||||||
"osx": {
|
|
||||||
"gdbpath": "gdb"
|
|
||||||
},
|
|
||||||
"executable": "${workspaceRoot}/build/hello_gpu.elf",
|
|
||||||
"autorun": [
|
|
||||||
"monitor reset shellhalt",
|
|
||||||
"load hello_gpu.elf",
|
|
||||||
"tbreak main",
|
|
||||||
"continue"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"name": "Debug: Hello GTE Psy-Q!",
|
|
||||||
"type": "gdb",
|
|
||||||
"request": "attach",
|
|
||||||
"target": "localhost:3333",
|
|
||||||
"remote": true,
|
|
||||||
"cwd": "${workspaceRoot}/build",
|
|
||||||
"valuesFormatting": "parseText",
|
|
||||||
"stopAtConnect": true,
|
"stopAtConnect": true,
|
||||||
"gdbpath": "gdb-multiarch",
|
"gdbpath": "gdb-multiarch",
|
||||||
"windows": {
|
"windows": {
|
||||||
@@ -69,12 +60,124 @@
|
|||||||
"gdbpath": "gdb"
|
"gdbpath": "gdb"
|
||||||
},
|
},
|
||||||
"executable": "${workspaceRoot}/build/hello_gte.elf",
|
"executable": "${workspaceRoot}/build/hello_gte.elf",
|
||||||
|
"setupCommands": [
|
||||||
|
{ "text": "set mi-async off" },
|
||||||
|
{ "text": "set remotetimeout 0" },
|
||||||
|
{ "text": "set logging file build/gen/hello_gte.gdb.log" },
|
||||||
|
{ "text": "set logging redirect on" }
|
||||||
|
],
|
||||||
"autorun": [
|
"autorun": [
|
||||||
"monitor reset shellhalt",
|
"monitor reset shellhalt",
|
||||||
"load hello_gte.elf",
|
"load hello_gte.elf",
|
||||||
"tbreak main",
|
"tbreak main",
|
||||||
"continue"
|
"continue"
|
||||||
]
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"name": "Debug: Hello GTE!",
|
||||||
|
"type": "gdb",
|
||||||
|
"request": "attach",
|
||||||
|
"target": "localhost:3333",
|
||||||
|
"remote": true,
|
||||||
|
"cwd": "${workspaceRoot}",
|
||||||
|
"valuesFormatting": "parseText",
|
||||||
|
"registerLimit": "1-32",
|
||||||
|
"frameFilters": false,
|
||||||
|
"showDevDebugOutput": false,
|
||||||
|
"printCalls": false,
|
||||||
|
"stopAtConnect": true,
|
||||||
|
"gdbpath": "gdb-multiarch",
|
||||||
|
"windows": {
|
||||||
|
"gdbpath": "gdb-multiarch.exe"
|
||||||
|
},
|
||||||
|
"osx": {
|
||||||
|
"gdbpath": "gdb"
|
||||||
|
},
|
||||||
|
"executable": "${workspaceRoot}/build/hello_gte.dwarf-injected.elf",
|
||||||
|
"setupCommands": [
|
||||||
|
{ "text": "set mi-async off" },
|
||||||
|
{ "text": "set remotetimeout 0" },
|
||||||
|
{ "text": "set logging file build/gen/hello_gte.gdb.log" },
|
||||||
|
{ "text": "set logging redirect on" }
|
||||||
|
],
|
||||||
|
"autorun": [
|
||||||
|
"monitor reset shellhalt",
|
||||||
|
"load build/hello_gte.dwarf-injected.elf",
|
||||||
|
"source scripts/gdb/gdb_tape_atoms.gdb",
|
||||||
|
"tbreak main",
|
||||||
|
"continue"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"name": "Debug: Hello Joypad!",
|
||||||
|
"type": "gdb",
|
||||||
|
"request": "attach",
|
||||||
|
"target": "localhost:3333",
|
||||||
|
"remote": true,
|
||||||
|
"cwd": "${workspaceRoot}",
|
||||||
|
"valuesFormatting": "parseText",
|
||||||
|
"registerLimit": "1-32",
|
||||||
|
"frameFilters": false,
|
||||||
|
"showDevDebugOutput": false,
|
||||||
|
"printCalls": false,
|
||||||
|
"stopAtConnect": true,
|
||||||
|
"gdbpath": "gdb-multiarch",
|
||||||
|
"windows": {
|
||||||
|
"gdbpath": "gdb-multiarch.exe"
|
||||||
|
},
|
||||||
|
"osx": {
|
||||||
|
"gdbpath": "gdb"
|
||||||
|
},
|
||||||
|
"executable": "${workspaceRoot}/build/hello_joypad.dwarf-injected.elf",
|
||||||
|
"setupCommands": [
|
||||||
|
{ "text": "set mi-async off" },
|
||||||
|
{ "text": "set remotetimeout 0" },
|
||||||
|
{ "text": "set logging file build/gen/hello_joypad.gdb.log" },
|
||||||
|
{ "text": "set logging redirect on" }
|
||||||
|
],
|
||||||
|
"autorun": [
|
||||||
|
"monitor reset shellhalt",
|
||||||
|
"load build/hello_joypad.dwarf-injected.elf",
|
||||||
|
"source scripts/gdb/gdb_tape_atoms.gdb",
|
||||||
|
"tbreak main",
|
||||||
|
"continue"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"name": "Debug: Hello Camera!",
|
||||||
|
"type": "gdb",
|
||||||
|
"request": "attach",
|
||||||
|
"target": "localhost:3333",
|
||||||
|
"remote": true,
|
||||||
|
"cwd": "${workspaceRoot}",
|
||||||
|
"valuesFormatting": "parseText",
|
||||||
|
"registerLimit": "1-32",
|
||||||
|
"frameFilters": false,
|
||||||
|
"showDevDebugOutput": false,
|
||||||
|
"printCalls": false,
|
||||||
|
"stopAtConnect": true,
|
||||||
|
"gdbpath": "gdb-multiarch",
|
||||||
|
"windows": {
|
||||||
|
"gdbpath": "gdb-multiarch.exe"
|
||||||
|
},
|
||||||
|
"osx": {
|
||||||
|
"gdbpath": "gdb"
|
||||||
|
},
|
||||||
|
"executable": "${workspaceRoot}/build/hello_camera.dwarf-injected.elf",
|
||||||
|
"setupCommands": [
|
||||||
|
{ "text": "set mi-async off" },
|
||||||
|
{ "text": "set remotetimeout 0" },
|
||||||
|
{ "text": "set logging file build/gen/hello_camera.gdb.log" },
|
||||||
|
{ "text": "set logging redirect on" }
|
||||||
|
],
|
||||||
|
"autorun": [
|
||||||
|
"monitor reset shellhalt",
|
||||||
|
"load build/hello_camera.dwarf-injected.elf",
|
||||||
|
"source scripts/gdb/gdb_tape_atoms.gdb",
|
||||||
|
"tbreak main",
|
||||||
|
"continue"
|
||||||
|
]
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
BIN
Binary file not shown.
+222
@@ -0,0 +1,222 @@
|
|||||||
|
"use strict";
|
||||||
|
|
||||||
|
const { nearestCall } = require("./lexer");
|
||||||
|
const { mergeIndexes, scanSource } = require("./source-index");
|
||||||
|
|
||||||
|
const TOKEN_TYPES = [
|
||||||
|
"tapeAtomKeyword",
|
||||||
|
"tapeAtomName",
|
||||||
|
"tapeComponentKeyword",
|
||||||
|
"tapeComponentName",
|
||||||
|
"tapeAnnotation",
|
||||||
|
"tapeBindType",
|
||||||
|
"tapePhase",
|
||||||
|
"tapeLabel",
|
||||||
|
"tapeCpuInstruction",
|
||||||
|
"tapeControlFlow",
|
||||||
|
"tapeGteInstruction",
|
||||||
|
"tapeGpuInstruction",
|
||||||
|
"tapeComponentInstruction",
|
||||||
|
"tapeDelaySlot",
|
||||||
|
"tapeGprRegister",
|
||||||
|
"tapeCop2Register",
|
||||||
|
"tapeDuffleType",
|
||||||
|
"tapeAttribute",
|
||||||
|
"keyword",
|
||||||
|
"macro",
|
||||||
|
];
|
||||||
|
|
||||||
|
const TOKEN_MODIFIERS = ["declaration", "tapeRead", "tapeWrite", "tapeAuto"];
|
||||||
|
const TOKEN_TYPE_INDEX = new Map(TOKEN_TYPES.map((name, index) => [name, index]));
|
||||||
|
const TOKEN_MODIFIER_INDEX = new Map(TOKEN_MODIFIERS.map((name, index) => [name, index]));
|
||||||
|
|
||||||
|
const ATOM_KEYWORDS = new Set(["MipsAtom_", "MipsAtom_Proc_"]);
|
||||||
|
const COMPONENT_KEYWORDS = new Set(["MipsAtomComp_", "MipsAtomComp_Proc_"]);
|
||||||
|
const ANNOTATIONS = new Set([
|
||||||
|
"atom_info", "atom_bind", "atom_reads", "atom_writes", "atom_label",
|
||||||
|
"atom_offset", "atom_reg", "atom_type", "atom_ctx", "atom_phase",
|
||||||
|
"atom_auto_reg", "phase_auto_reg", "atom_dbg_skip",
|
||||||
|
]);
|
||||||
|
|
||||||
|
const DSL_KEYWORDS = new Set([
|
||||||
|
"FI_", "I_", "NI_", "Relative_", "Struct_", "Enum_", "Union_", "Array_",
|
||||||
|
"Slice_", "TypeR_", "TypeV_", "align_", "internal", "local_persist", "global",
|
||||||
|
"RO_", "LP_", "gknown", "expect_", "cexpr_",
|
||||||
|
"asm", "asm_words", "asm_rpins", "asm_clobber",
|
||||||
|
"O_", "S_", "C_", "T_", "tmpl", "glue", "r_", "v_", "tr_", "tv_",
|
||||||
|
"rgcc", "r_use", "r_set", "r_mod", "r_imm", "r_mem",
|
||||||
|
"u1_", "u2_", "u4_", "u8_", "s1_", "s2_", "s4_", "s8_",
|
||||||
|
"u1_r", "u2_r", "u4_r", "u8_r", "u1_v", "u2_v", "u4_v", "u8_v",
|
||||||
|
]);
|
||||||
|
|
||||||
|
const DELAY_SLOT_KEYWORDS = new Set(["LdSlot_", "BdSlot_", "DmaSlot_", "GteDelay_"]);
|
||||||
|
|
||||||
|
const CONTROL_FLOW_PREFIXES = /^(?:branch_|jump_|call_)/;
|
||||||
|
|
||||||
|
const ROLE_TO_TYPE = {
|
||||||
|
atomName: "tapeAtomName",
|
||||||
|
componentName: "tapeComponentName",
|
||||||
|
bindType: "tapeBindType",
|
||||||
|
duffleType: "tapeDuffleType",
|
||||||
|
gprRegister: "tapeGprRegister",
|
||||||
|
cop2Register: "tapeCop2Register",
|
||||||
|
};
|
||||||
|
|
||||||
|
function registerType(name, index) {
|
||||||
|
const kind = index.registers.get(name);
|
||||||
|
if (kind === "gpr" || /^R_[A-Za-z0-9_]+$/.test(name)) return "tapeGprRegister";
|
||||||
|
if (kind === "cop2" || /^(?:C2_|gte_cr_)[A-Za-z0-9_]+$/.test(name)) return "tapeCop2Register";
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
function instructionType(name, index) {
|
||||||
|
const domain = index.macros.get(name);
|
||||||
|
if (domain === "control") return "tapeControlFlow";
|
||||||
|
if (domain === "cpu") return "tapeCpuInstruction";
|
||||||
|
if (domain === "gte") return "tapeGteInstruction";
|
||||||
|
if (domain === "gpu") return "tapeGpuInstruction";
|
||||||
|
if (domain === "component") {
|
||||||
|
if (/^mac_gte_/.test(name)) return "tapeGteInstruction";
|
||||||
|
if (/^mac_gp/.test(name)) return "tapeGpuInstruction";
|
||||||
|
if (/^mac_/.test(name)) return "tapeComponentInstruction";
|
||||||
|
return "macro";
|
||||||
|
}
|
||||||
|
if (domain === "utility") return "macro";
|
||||||
|
if (/^gte_(?!cr_)/.test(name)) return "tapeGteInstruction";
|
||||||
|
if (/^gp[01]_/.test(name)) return "tapeGpuInstruction";
|
||||||
|
if (/^mac_gte_/.test(name)) return "tapeGteInstruction";
|
||||||
|
if (/^mac_gp/.test(name)) return "tapeGpuInstruction";
|
||||||
|
if (/^mac_/.test(name)) return "tapeComponentInstruction";
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
function modifierMask(modifiers) {
|
||||||
|
let mask = 0;
|
||||||
|
for (const modifier of modifiers) {
|
||||||
|
const index = TOKEN_MODIFIER_INDEX.get(modifier);
|
||||||
|
if (index !== undefined) mask |= (1 << index);
|
||||||
|
}
|
||||||
|
return mask;
|
||||||
|
}
|
||||||
|
|
||||||
|
function isRegUseAccess(tokens, tokenIndex) {
|
||||||
|
const prev = tokens[tokenIndex - 1];
|
||||||
|
if (!prev || prev.text !== ".") return false;
|
||||||
|
const prevPrev = tokens[tokenIndex - 2];
|
||||||
|
if (!prevPrev || prevPrev.kind !== "identifier") return false;
|
||||||
|
const next = tokens[tokenIndex + 1];
|
||||||
|
if (next && next.text === ".") return false;
|
||||||
|
if (prevPrev.text === "r") return true;
|
||||||
|
const prev3 = tokens[tokenIndex - 3];
|
||||||
|
const prev4 = tokens[tokenIndex - 4];
|
||||||
|
if (prev3 && prev3.text === "." && prev4 && prev4.kind === "identifier" && prev4.text === "r") return true;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
function classifyDocument(source, filePath, workspaceIndex, shouldCancel = () => false) {
|
||||||
|
const scanned = scanSource(source, filePath);
|
||||||
|
const index = mergeIndexes(workspaceIndex, scanned.index);
|
||||||
|
const spans = [];
|
||||||
|
|
||||||
|
for (let tokenIndex = 0; tokenIndex < scanned.tokens.length; tokenIndex += 1) {
|
||||||
|
if (shouldCancel()) break;
|
||||||
|
const token = scanned.tokens[tokenIndex];
|
||||||
|
if (token.kind !== "identifier") continue;
|
||||||
|
|
||||||
|
let type = null;
|
||||||
|
let modifiers = [];
|
||||||
|
const declaration = scanned.declarations.get(token.start);
|
||||||
|
const context = nearestCall(scanned.contexts, tokenIndex);
|
||||||
|
|
||||||
|
if (declaration) {
|
||||||
|
type = ROLE_TO_TYPE[declaration.role] || null;
|
||||||
|
modifiers = declaration.modifiers.slice();
|
||||||
|
} else if (ATOM_KEYWORDS.has(token.text)) {
|
||||||
|
type = "tapeAtomKeyword";
|
||||||
|
} else if (COMPONENT_KEYWORDS.has(token.text)) {
|
||||||
|
type = "keyword";
|
||||||
|
} else if (ANNOTATIONS.has(token.text)) {
|
||||||
|
type = "tapeAnnotation";
|
||||||
|
} else if (context && context.callee === "atom_bind" && context.argIndex === 0) {
|
||||||
|
type = "tapeBindType";
|
||||||
|
} else if (context && context.callee === "atom_phase" && context.argIndex === 0) {
|
||||||
|
type = "tapePhase";
|
||||||
|
modifiers = ["declaration"];
|
||||||
|
} else if (context && context.callee === "atom_ctx" && context.argIndex === 0) {
|
||||||
|
type = "tapeAtomName";
|
||||||
|
} else if (context && context.callee === "atom_label" && context.argIndex === 0) {
|
||||||
|
type = "tapeLabel";
|
||||||
|
modifiers = ["declaration"];
|
||||||
|
} else if (context && context.callee === "atom_offset" && context.argIndex <= 1) {
|
||||||
|
type = "tapeLabel";
|
||||||
|
} else if (context && context.callee === "atom_reads") {
|
||||||
|
type = registerType(token.text, index);
|
||||||
|
if (type) modifiers = ["tapeRead"];
|
||||||
|
} else if (context && context.callee === "atom_writes") {
|
||||||
|
type = registerType(token.text, index);
|
||||||
|
if (type) modifiers = ["tapeWrite"];
|
||||||
|
} else if (context && context.callee === "atom_auto_reg") {
|
||||||
|
if (context.argIndex === 0) type = "tapeAtomName";
|
||||||
|
if (context.argIndex === 1) {
|
||||||
|
type = "tapeGprRegister";
|
||||||
|
modifiers = ["declaration", "tapeAuto"];
|
||||||
|
}
|
||||||
|
} else if (context && context.callee === "phase_auto_reg") {
|
||||||
|
if (context.argIndex === 0) type = "tapePhase";
|
||||||
|
if (context.argIndex === 1) {
|
||||||
|
type = "tapeGprRegister";
|
||||||
|
modifiers = ["declaration", "tapeAuto"];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!type && index.bindTypes.has(token.text)) type = "tapeBindType";
|
||||||
|
if (!type && DSL_KEYWORDS.has(token.text)) type = "keyword";
|
||||||
|
if (!type && index.types.has(token.text)) type = "tapeDuffleType";
|
||||||
|
if (!type && index.attributes.has(token.text)) type = "tapeAttribute";
|
||||||
|
if (!type) type = registerType(token.text, index);
|
||||||
|
if (!type && DELAY_SLOT_KEYWORDS.has(token.text)) type = "tapeDelaySlot";
|
||||||
|
if (!type) {
|
||||||
|
const domain = index.macros.get(token.text);
|
||||||
|
if (domain === "control" || (domain && CONTROL_FLOW_PREFIXES.test(token.text))) {
|
||||||
|
type = "tapeControlFlow";
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (!type && isRegUseAccess(scanned.tokens, tokenIndex)) type = "tapeGprRegister";
|
||||||
|
if (!type) type = instructionType(token.text, index);
|
||||||
|
if (!type && /^(?:Slice_|A[0-9]+_)/.test(token.text)) type = "tapeDuffleType";
|
||||||
|
if (!type && /_[RV]$/.test(token.text)) type = "tapeDuffleType";
|
||||||
|
if (!type && index.atoms.has(token.text)) type = "tapeAtomName";
|
||||||
|
if (!type && index.components.has(token.text)) type = "tapeComponentName";
|
||||||
|
if (!type && index.phases.has(token.text)) type = "tapePhase";
|
||||||
|
if (!type && index.labels.has(token.text)) type = "tapeLabel";
|
||||||
|
if (!type) continue;
|
||||||
|
|
||||||
|
spans.push({
|
||||||
|
text: token.text,
|
||||||
|
type,
|
||||||
|
typeIndex: TOKEN_TYPE_INDEX.get(type),
|
||||||
|
modifiers,
|
||||||
|
modifierMask: modifierMask(modifiers),
|
||||||
|
start: token.start,
|
||||||
|
length: token.end - token.start,
|
||||||
|
line: token.line,
|
||||||
|
character: token.character,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
spans.sort((left, right) => left.start - right.start || left.length - right.length);
|
||||||
|
const nonOverlapping = [];
|
||||||
|
for (const span of spans) {
|
||||||
|
const previous = nonOverlapping[nonOverlapping.length - 1];
|
||||||
|
if (!previous || previous.start + previous.length <= span.start) nonOverlapping.push(span);
|
||||||
|
}
|
||||||
|
|
||||||
|
return { spans: nonOverlapping, errors: scanned.errors };
|
||||||
|
}
|
||||||
|
|
||||||
|
module.exports = {
|
||||||
|
TOKEN_MODIFIERS,
|
||||||
|
TOKEN_TYPES,
|
||||||
|
classifyDocument,
|
||||||
|
modifierMask,
|
||||||
|
};
|
||||||
+111
@@ -0,0 +1,111 @@
|
|||||||
|
"use strict";
|
||||||
|
|
||||||
|
const vscode = require("vscode");
|
||||||
|
const { TOKEN_MODIFIERS, TOKEN_TYPES, classifyDocument } = require("./classifier");
|
||||||
|
const { createIndex, mergeIndexes, scanSource } = require("./source-index");
|
||||||
|
|
||||||
|
const SOURCE_GLOB = "**/*.{c,h,cc,cpp,cxx,hh,hpp,hxx}";
|
||||||
|
const EXCLUDE_GLOB = "**/{gen,build,.slop_cache,toolchain,node_modules}/**";
|
||||||
|
const EXCLUDED_SEGMENTS = new Set(["gen", "build", ".slop_cache", "toolchain", "node_modules"]);
|
||||||
|
|
||||||
|
function isExcluded(uri) {
|
||||||
|
const segments = uri.fsPath.replaceAll("\\", "/").split("/");
|
||||||
|
return segments.some((segment) => EXCLUDED_SEGMENTS.has(segment));
|
||||||
|
}
|
||||||
|
|
||||||
|
function formatError(filePath, error) {
|
||||||
|
return `${filePath}:${error.offset}: ${error.kind}`;
|
||||||
|
}
|
||||||
|
|
||||||
|
async function activate(context) {
|
||||||
|
const output = vscode.window.createOutputChannel("Tape Atom DSL");
|
||||||
|
const emitter = new vscode.EventEmitter();
|
||||||
|
const legend = new vscode.SemanticTokensLegend(TOKEN_TYPES, TOKEN_MODIFIERS);
|
||||||
|
let workspaceIndex = createIndex();
|
||||||
|
let rebuildGeneration = 0;
|
||||||
|
let debounceHandle = null;
|
||||||
|
|
||||||
|
async function rebuildIndex() {
|
||||||
|
const generation = ++rebuildGeneration;
|
||||||
|
const files = await vscode.workspace.findFiles(SOURCE_GLOB, EXCLUDE_GLOB);
|
||||||
|
let nextIndex = createIndex();
|
||||||
|
|
||||||
|
for (const uri of files) {
|
||||||
|
if (generation !== rebuildGeneration) return;
|
||||||
|
if (isExcluded(uri)) continue;
|
||||||
|
try {
|
||||||
|
const bytes = await vscode.workspace.fs.readFile(uri);
|
||||||
|
const source = Buffer.from(bytes).toString("utf8");
|
||||||
|
const result = scanSource(source, uri.fsPath);
|
||||||
|
nextIndex = mergeIndexes(nextIndex, result.index);
|
||||||
|
for (const error of result.errors) output.appendLine(formatError(uri.fsPath, error));
|
||||||
|
} catch (error) {
|
||||||
|
output.appendLine(`${uri.fsPath}: ${error.stack || error.message || error}`);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (generation !== rebuildGeneration) return;
|
||||||
|
workspaceIndex = nextIndex;
|
||||||
|
emitter.fire();
|
||||||
|
}
|
||||||
|
|
||||||
|
function scheduleRebuild(uri) {
|
||||||
|
if (uri && isExcluded(uri)) return;
|
||||||
|
if (debounceHandle !== null) clearTimeout(debounceHandle);
|
||||||
|
debounceHandle = setTimeout(() => {
|
||||||
|
debounceHandle = null;
|
||||||
|
rebuildIndex().catch((error) => output.appendLine(error.stack || String(error)));
|
||||||
|
}, 100);
|
||||||
|
}
|
||||||
|
|
||||||
|
const provider = {
|
||||||
|
onDidChangeSemanticTokens: emitter.event,
|
||||||
|
provideDocumentSemanticTokens(document, cancellationToken) {
|
||||||
|
try {
|
||||||
|
const result = classifyDocument(
|
||||||
|
document.getText(),
|
||||||
|
document.uri.fsPath,
|
||||||
|
workspaceIndex,
|
||||||
|
() => cancellationToken.isCancellationRequested
|
||||||
|
);
|
||||||
|
const builder = new vscode.SemanticTokensBuilder(legend);
|
||||||
|
for (const span of result.spans) {
|
||||||
|
if (cancellationToken.isCancellationRequested) break;
|
||||||
|
builder.push(span.line, span.character, span.length, span.typeIndex, span.modifierMask);
|
||||||
|
}
|
||||||
|
for (const error of result.errors) {
|
||||||
|
output.appendLine(formatError(document.uri.fsPath || document.uri.toString(), error));
|
||||||
|
}
|
||||||
|
return builder.build();
|
||||||
|
} catch (error) {
|
||||||
|
output.appendLine(`${document.uri}: ${error.stack || error.message || error}`);
|
||||||
|
return new vscode.SemanticTokensBuilder(legend).build();
|
||||||
|
}
|
||||||
|
},
|
||||||
|
};
|
||||||
|
|
||||||
|
const selector = [
|
||||||
|
{ language: "c", scheme: "file" },
|
||||||
|
{ language: "c", scheme: "untitled" },
|
||||||
|
{ language: "cpp", scheme: "file" },
|
||||||
|
{ language: "cpp", scheme: "untitled" },
|
||||||
|
];
|
||||||
|
const watcher = vscode.workspace.createFileSystemWatcher(SOURCE_GLOB);
|
||||||
|
|
||||||
|
context.subscriptions.push(
|
||||||
|
output,
|
||||||
|
emitter,
|
||||||
|
watcher,
|
||||||
|
watcher.onDidCreate(scheduleRebuild),
|
||||||
|
watcher.onDidChange(scheduleRebuild),
|
||||||
|
watcher.onDidDelete(scheduleRebuild),
|
||||||
|
vscode.languages.registerDocumentSemanticTokensProvider(selector, provider, legend),
|
||||||
|
{ dispose() { if (debounceHandle !== null) clearTimeout(debounceHandle); } }
|
||||||
|
);
|
||||||
|
|
||||||
|
await rebuildIndex();
|
||||||
|
}
|
||||||
|
|
||||||
|
function deactivate() {}
|
||||||
|
|
||||||
|
module.exports = { activate, deactivate };
|
||||||
Vendored
+186
@@ -0,0 +1,186 @@
|
|||||||
|
"use strict";
|
||||||
|
|
||||||
|
function isIdentifierStart(code) {
|
||||||
|
return code === 95 ||
|
||||||
|
(code >= 65 && code <= 90) ||
|
||||||
|
(code >= 97 && code <= 122);
|
||||||
|
}
|
||||||
|
|
||||||
|
function isIdentifierContinue(code) {
|
||||||
|
return isIdentifierStart(code) || (code >= 48 && code <= 57);
|
||||||
|
}
|
||||||
|
|
||||||
|
function lex(source) {
|
||||||
|
if (typeof source !== "string") throw new TypeError("source must be a string");
|
||||||
|
|
||||||
|
const tokens = [];
|
||||||
|
const errors = [];
|
||||||
|
let offset = 0;
|
||||||
|
let line = 0;
|
||||||
|
let character = 0;
|
||||||
|
|
||||||
|
function advance() {
|
||||||
|
if (source[offset] === "\r" && source[offset + 1] === "\n") {
|
||||||
|
offset += 2;
|
||||||
|
line += 1;
|
||||||
|
character = 0;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (source[offset] === "\n") {
|
||||||
|
offset += 1;
|
||||||
|
line += 1;
|
||||||
|
character = 0;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
offset += 1;
|
||||||
|
character += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
function pushToken(kind, start, startLine, startCharacter) {
|
||||||
|
tokens.push({
|
||||||
|
kind,
|
||||||
|
text: source.slice(start, offset),
|
||||||
|
start,
|
||||||
|
end: offset,
|
||||||
|
line: startLine,
|
||||||
|
character: startCharacter,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
while (offset < source.length) {
|
||||||
|
const ch = source[offset];
|
||||||
|
|
||||||
|
if (/\s/.test(ch)) {
|
||||||
|
advance();
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (ch === "/" && source[offset + 1] === "/") {
|
||||||
|
while (offset < source.length && source[offset] !== "\r" && source[offset] !== "\n") advance();
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (ch === "/" && source[offset + 1] === "*") {
|
||||||
|
const start = offset;
|
||||||
|
advance();
|
||||||
|
advance();
|
||||||
|
let closed = false;
|
||||||
|
while (offset < source.length) {
|
||||||
|
if (source[offset] === "*" && source[offset + 1] === "/") {
|
||||||
|
advance();
|
||||||
|
advance();
|
||||||
|
closed = true;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
advance();
|
||||||
|
}
|
||||||
|
if (!closed) errors.push({ kind: "unterminated-block-comment", offset: start });
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (ch === "\"" || ch === "'") {
|
||||||
|
const quote = ch;
|
||||||
|
const start = offset;
|
||||||
|
advance();
|
||||||
|
let closed = false;
|
||||||
|
while (offset < source.length) {
|
||||||
|
if (source[offset] === "\\") {
|
||||||
|
advance();
|
||||||
|
if (offset < source.length) advance();
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (source[offset] === quote) {
|
||||||
|
advance();
|
||||||
|
closed = true;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if (source[offset] === "\n" || source[offset] === "\r") break;
|
||||||
|
advance();
|
||||||
|
}
|
||||||
|
if (!closed) errors.push({ kind: "unterminated-literal", offset: start });
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
const code = source.charCodeAt(offset);
|
||||||
|
if (isIdentifierStart(code)) {
|
||||||
|
const start = offset;
|
||||||
|
const startLine = line;
|
||||||
|
const startCharacter = character;
|
||||||
|
advance();
|
||||||
|
while (offset < source.length && isIdentifierContinue(source.charCodeAt(offset))) advance();
|
||||||
|
pushToken("identifier", start, startLine, startCharacter);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
const start = offset;
|
||||||
|
const startLine = line;
|
||||||
|
const startCharacter = character;
|
||||||
|
advance();
|
||||||
|
pushToken("punctuation", start, startLine, startCharacter);
|
||||||
|
}
|
||||||
|
|
||||||
|
return { tokens, errors };
|
||||||
|
}
|
||||||
|
|
||||||
|
function buildCallContexts(tokens) {
|
||||||
|
const contexts = Array.from({ length: tokens.length }, () => []);
|
||||||
|
const calls = [];
|
||||||
|
const errors = [];
|
||||||
|
const stack = [];
|
||||||
|
|
||||||
|
for (let tokenIndex = 0; tokenIndex < tokens.length; tokenIndex += 1) {
|
||||||
|
const token = tokens[tokenIndex];
|
||||||
|
|
||||||
|
if (token.text === ")") {
|
||||||
|
if (stack.length === 0) {
|
||||||
|
errors.push({ kind: "unmatched-close-paren", offset: token.start });
|
||||||
|
} else {
|
||||||
|
const frame = stack.pop();
|
||||||
|
if (frame.callee !== null) calls.push({ ...frame, closeTokenIndex: tokenIndex });
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
contexts[tokenIndex] = stack
|
||||||
|
.filter((frame) => frame.callee !== null)
|
||||||
|
.map((frame) => ({
|
||||||
|
callee: frame.callee,
|
||||||
|
calleeTokenIndex: frame.calleeTokenIndex,
|
||||||
|
openTokenIndex: frame.openTokenIndex,
|
||||||
|
argIndex: frame.argIndex,
|
||||||
|
}));
|
||||||
|
|
||||||
|
if (token.text === "(") {
|
||||||
|
const previous = tokens[tokenIndex - 1];
|
||||||
|
const hasCallee = previous && previous.kind === "identifier";
|
||||||
|
stack.push({
|
||||||
|
callee: hasCallee ? previous.text : null,
|
||||||
|
calleeTokenIndex: hasCallee ? tokenIndex - 1 : -1,
|
||||||
|
openTokenIndex: tokenIndex,
|
||||||
|
argIndex: 0,
|
||||||
|
});
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (token.text === "," && stack.length > 0) {
|
||||||
|
const frame = stack[stack.length - 1];
|
||||||
|
if (frame.callee !== null) frame.argIndex += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
for (const frame of stack) {
|
||||||
|
errors.push({ kind: "unmatched-open-paren", offset: tokens[frame.openTokenIndex].start });
|
||||||
|
}
|
||||||
|
|
||||||
|
return { contexts, calls, errors };
|
||||||
|
}
|
||||||
|
|
||||||
|
function nearestCall(contexts, tokenIndex, callee) {
|
||||||
|
const entries = contexts[tokenIndex] || [];
|
||||||
|
for (let contextIndex = entries.length - 1; contextIndex >= 0; contextIndex -= 1) {
|
||||||
|
const entry = entries[contextIndex];
|
||||||
|
if (callee === undefined || entry.callee === callee) return entry;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
module.exports = { buildCallContexts, lex, nearestCall };
|
||||||
+85
@@ -0,0 +1,85 @@
|
|||||||
|
{
|
||||||
|
"name": "atomasm-psx",
|
||||||
|
"displayName": "AtomAsm-PSX",
|
||||||
|
"description": "Semantic highlighting for the PS1 Tape/Atom MIPS macro DSL",
|
||||||
|
"publisher": "local",
|
||||||
|
"version": "0.3.0",
|
||||||
|
"engines": { "vscode": "^1.80.0" },
|
||||||
|
"categories": ["Programming Languages"],
|
||||||
|
"activationEvents": ["onLanguage:c", "onLanguage:cpp"],
|
||||||
|
"main": "./extension.js",
|
||||||
|
"files": [
|
||||||
|
"classifier.js",
|
||||||
|
"extension.js",
|
||||||
|
"lexer.js",
|
||||||
|
"source-index.js",
|
||||||
|
"syntaxes/tape_atom.tmLanguage.json"
|
||||||
|
],
|
||||||
|
"scripts": {
|
||||||
|
"test": "node --test test/*.test.js",
|
||||||
|
"package": "npx --yes @vscode/vsce@3.6.1 package --allow-missing-repository --skip-license --out atomasm-psx-0.3.0.vsix"
|
||||||
|
},
|
||||||
|
"contributes": {
|
||||||
|
"semanticTokenTypes": [
|
||||||
|
{ "id": "tapeAtomKeyword", "superType": "keyword", "description": "Tape atom declaration keyword" },
|
||||||
|
{ "id": "tapeAtomName", "superType": "function", "description": "Tape atom name" },
|
||||||
|
{ "id": "tapeComponentKeyword", "superType": "keyword", "description": "Tape atom component declaration keyword" },
|
||||||
|
{ "id": "tapeComponentName", "superType": "function", "description": "Tape atom component name" },
|
||||||
|
{ "id": "tapeAnnotation", "superType": "macro", "description": "Tape atom annotation" },
|
||||||
|
{ "id": "tapeBindType", "superType": "type", "description": "Tape bind structure type" },
|
||||||
|
{ "id": "tapePhase", "superType": "label", "description": "Tape atom phase" },
|
||||||
|
{ "id": "tapeLabel", "superType": "label", "description": "Tape atom branch label" },
|
||||||
|
{ "id": "tapeCpuInstruction", "superType": "macro", "description": "MIPS CPU instruction emitter" },
|
||||||
|
{ "id": "tapeControlFlow", "superType": "keyword", "description": "MIPS branch or jump instruction" },
|
||||||
|
{ "id": "tapeGteInstruction", "superType": "macro", "description": "GTE instruction emitter" },
|
||||||
|
{ "id": "tapeGpuInstruction", "superType": "macro", "description": "GPU command emitter" },
|
||||||
|
{ "id": "tapeComponentInstruction", "superType": "macro", "description": "Tape atom component invocation" },
|
||||||
|
{ "id": "tapeDelaySlot", "superType": "keyword", "description": "Load or branch delay slot annotation" },
|
||||||
|
{ "id": "tapeGprRegister", "superType": "variable", "description": "MIPS GPR alias" },
|
||||||
|
{ "id": "tapeCop2Register", "superType": "variable", "description": "COP2 data or control register alias" },
|
||||||
|
{ "id": "tapeDuffleType", "superType": "type", "description": "Duffle type or type constructor" },
|
||||||
|
{ "id": "tapeAttribute", "superType": "keyword", "description": "Duffle linkage or storage attribute" },
|
||||||
|
{ "id": "keyword", "description": "Standard keyword (DSL built-in macros)" },
|
||||||
|
{ "id": "macro", "description": "Standard macro (utility #define with no instruction domain)" }
|
||||||
|
],
|
||||||
|
"semanticTokenModifiers": [
|
||||||
|
{ "id": "tapeRead", "description": "Register declared in atom_reads" },
|
||||||
|
{ "id": "tapeWrite", "description": "Register declared in atom_writes" },
|
||||||
|
{ "id": "tapeAuto", "description": "Auto-allocated register" }
|
||||||
|
],
|
||||||
|
"semanticTokenScopes": [
|
||||||
|
{
|
||||||
|
"language": "c",
|
||||||
|
"scopes": {
|
||||||
|
"tapeAtomKeyword": ["keyword.control.duffle.atom"],
|
||||||
|
"tapeAtomName": ["entity.name.function.duffle.atom"],
|
||||||
|
"tapeComponentKeyword": ["keyword.control.duffle.component"],
|
||||||
|
"tapeComponentName": ["entity.name.function.duffle.component"],
|
||||||
|
"tapeAnnotation": ["support.function.duffle.annotation"],
|
||||||
|
"tapeBindType": ["entity.name.type.duffle.bind"],
|
||||||
|
"tapePhase": ["entity.name.tag.duffle.phase"],
|
||||||
|
"tapeLabel": ["entity.name.label.duffle.atom"],
|
||||||
|
"tapeCpuInstruction": ["support.function.duffle.cpu"],
|
||||||
|
"tapeControlFlow": ["keyword.control.duffle.branch"],
|
||||||
|
"tapeGteInstruction": ["support.function.duffle.gte"],
|
||||||
|
"tapeGpuInstruction": ["support.function.duffle.gpu"],
|
||||||
|
"tapeComponentInstruction": ["support.function.duffle.component"],
|
||||||
|
"tapeDelaySlot": ["keyword.operator.duffle.delayslot"],
|
||||||
|
"tapeGprRegister": ["variable.other.constant.duffle.gpr"],
|
||||||
|
"tapeCop2Register": ["variable.other.constant.duffle.cop2"],
|
||||||
|
"tapeDuffleType": ["storage.type.duffle.type"],
|
||||||
|
"tapeAttribute": ["storage.modifier.duffle.attr"],
|
||||||
|
"keyword": ["keyword"],
|
||||||
|
"macro": ["entity.name.function.preprocessor"]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"grammars": [
|
||||||
|
{
|
||||||
|
"scopeName": "tape_atom.injection",
|
||||||
|
"path": "./syntaxes/tape_atom.tmLanguage.json",
|
||||||
|
"injectTo": ["source.c", "source.cpp"]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
+341
@@ -0,0 +1,341 @@
|
|||||||
|
"use strict";
|
||||||
|
|
||||||
|
const path = require("node:path");
|
||||||
|
const { buildCallContexts, lex, nearestCall } = require("./lexer");
|
||||||
|
|
||||||
|
const BASE_TYPES = [
|
||||||
|
"B1", "B2", "B4", "B8", "F4", "F8", "S1", "S2", "S4", "S8",
|
||||||
|
"U1", "U2", "U4", "U8", "MipsAtom", "MipsCode", "Reg",
|
||||||
|
];
|
||||||
|
|
||||||
|
const C_BUILTINS = new Set([
|
||||||
|
"void", "type", "char", "short", "int", "long", "float", "double",
|
||||||
|
"unsigned", "signed", "bool", "size_t", "uint8_t", "uint16_t", "uint32_t",
|
||||||
|
"int8_t", "int16_t", "int32_t",
|
||||||
|
]);
|
||||||
|
|
||||||
|
const BASE_ATTRIBUTES = [
|
||||||
|
"FI_", "I_", "NI_", "Relative_", "Struct_", "Enum_", "Union_", "Array_",
|
||||||
|
"Slice_", "TypeR_", "TypeV_", "align_", "internal", "local_persist", "global",
|
||||||
|
"RO_", "LP_", "gknown", "expect_", "cexpr_",
|
||||||
|
"asm", "asm_words", "asm_rpins", "asm_clobber",
|
||||||
|
"O_", "S_", "C_", "T_", "tmpl", "glue", "r_", "v_", "tr_", "tv_",
|
||||||
|
"rgcc", "r_use", "r_set", "r_mod", "r_imm", "r_mem",
|
||||||
|
"u1_", "u2_", "u4_", "u8_", "s1_", "s2_", "s4_", "s8_",
|
||||||
|
"u1_r", "u2_r", "u4_r", "u8_r", "u1_v", "u2_v", "u4_v", "u8_v",
|
||||||
|
];
|
||||||
|
|
||||||
|
function createIndex() {
|
||||||
|
return {
|
||||||
|
atoms: new Set(),
|
||||||
|
components: new Set(),
|
||||||
|
componentAliases: new Set(),
|
||||||
|
macros: new Map(),
|
||||||
|
registers: new Map(),
|
||||||
|
bindTypes: new Set(),
|
||||||
|
types: new Set(BASE_TYPES),
|
||||||
|
phases: new Set(),
|
||||||
|
labels: new Set(),
|
||||||
|
attributes: new Set(BASE_ATTRIBUTES),
|
||||||
|
componentCallees: new Map(),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
function cloneIndex(source) {
|
||||||
|
const result = createIndex();
|
||||||
|
for (const key of ["atoms", "components", "componentAliases", "bindTypes", "types", "phases", "labels", "attributes"]) {
|
||||||
|
for (const value of source[key]) result[key].add(value);
|
||||||
|
}
|
||||||
|
for (const [name, domain] of source.macros) result.macros.set(name, domain);
|
||||||
|
for (const [name, domain] of source.registers) result.registers.set(name, domain);
|
||||||
|
for (const [name, callees] of source.componentCallees) result.componentCallees.set(name, callees.slice());
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
function mergeIndexes(...sources) {
|
||||||
|
const result = createIndex();
|
||||||
|
for (const source of sources) {
|
||||||
|
if (!source) continue;
|
||||||
|
for (const key of ["atoms", "components", "componentAliases", "bindTypes", "types", "phases", "labels", "attributes"]) {
|
||||||
|
for (const value of source[key]) result[key].add(value);
|
||||||
|
}
|
||||||
|
for (const [name, domain] of source.macros) {
|
||||||
|
const existing = result.macros.get(name);
|
||||||
|
if (!existing || domainRank(domain) >= domainRank(existing)) result.macros.set(name, domain);
|
||||||
|
}
|
||||||
|
for (const [name, domain] of source.registers) result.registers.set(name, domain);
|
||||||
|
for (const [name, callees] of source.componentCallees) {
|
||||||
|
const existing = result.componentCallees.get(name) || [];
|
||||||
|
result.componentCallees.set(name, existing.concat(callees));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return resolveComponentDomains(result);
|
||||||
|
}
|
||||||
|
|
||||||
|
function domainFromPath(filePath) {
|
||||||
|
const base = path.basename(filePath.replaceAll("\\", "/")).toLowerCase();
|
||||||
|
if (base === "mips.h") return "cpu";
|
||||||
|
if (base === "gte.h") return "gte";
|
||||||
|
if (base === "gp.h") return "gpu";
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
function prefixDomain(name) {
|
||||||
|
if (/^(?:branch_|jump_|call_)/.test(name)) return "control";
|
||||||
|
if (/^gte_(?!cr_)/.test(name) || name.startsWith("mac_gte_") || name.startsWith("ac_gte_")) return "gte";
|
||||||
|
if (/^gp[01]_/.test(name) || name.startsWith("mac_gp_") || name.startsWith("ac_gp_")) return "gpu";
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
function collectBraceIdentifiers(tokens, openBraceIndex) {
|
||||||
|
const names = [];
|
||||||
|
let depth = 0;
|
||||||
|
for (let tokenIndex = openBraceIndex; tokenIndex < tokens.length; tokenIndex += 1) {
|
||||||
|
if (tokens[tokenIndex].text === "{") depth += 1;
|
||||||
|
if (tokens[tokenIndex].text === "}") {
|
||||||
|
depth -= 1;
|
||||||
|
if (depth === 0) break;
|
||||||
|
}
|
||||||
|
if (tokens[tokenIndex].kind === "identifier") names.push(tokens[tokenIndex].text);
|
||||||
|
}
|
||||||
|
return names;
|
||||||
|
}
|
||||||
|
|
||||||
|
function resolveComponentDomains(index) {
|
||||||
|
const hardwareRank = { cpu: 1, gpu: 2, gte: 3, control: 4 };
|
||||||
|
let changed = true;
|
||||||
|
while (changed) {
|
||||||
|
changed = false;
|
||||||
|
for (const [alias, callees] of index.componentCallees) {
|
||||||
|
let best = index.macros.get(alias) || "component";
|
||||||
|
let bestRank = hardwareRank[best] || 0;
|
||||||
|
for (const callee of callees) {
|
||||||
|
const domain = prefixDomain(callee) || index.macros.get(callee);
|
||||||
|
const rank = hardwareRank[domain] || 0;
|
||||||
|
if (rank > bestRank) {
|
||||||
|
best = domain;
|
||||||
|
bestRank = rank;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (bestRank > 0 && index.macros.get(alias) !== best) {
|
||||||
|
index.macros.set(alias, best);
|
||||||
|
changed = true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return index;
|
||||||
|
}
|
||||||
|
|
||||||
|
function domainRank(domain) {
|
||||||
|
if (domain === "control") return 4;
|
||||||
|
if (domain === "cpu" || domain === "gte" || domain === "gpu") return 3;
|
||||||
|
if (domain === "component") return 2;
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
function registerKind(name) {
|
||||||
|
if (/^R_[A-Za-z0-9_]+$/.test(name)) return "gpr";
|
||||||
|
if (/^(?:C2_|gte_cr_)[A-Za-z0-9_]+$/.test(name)) return "cop2";
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
function componentAlias(name) {
|
||||||
|
return name.startsWith("ac_") ? `mac_${name.slice(3)}` : null;
|
||||||
|
}
|
||||||
|
|
||||||
|
function findFunctionNameBefore(tokens, calleeTokenIndex) {
|
||||||
|
let closeIndex = calleeTokenIndex - 1;
|
||||||
|
while (closeIndex >= 0 && tokens[closeIndex].kind === "identifier" && tokens[closeIndex].text === "atom_dbg_skip") {
|
||||||
|
closeIndex -= 1;
|
||||||
|
}
|
||||||
|
if (!tokens[closeIndex] || tokens[closeIndex].text !== ")") return null;
|
||||||
|
|
||||||
|
let depth = 1;
|
||||||
|
for (let tokenIndex = closeIndex - 1; tokenIndex >= 0; tokenIndex -= 1) {
|
||||||
|
if (tokens[tokenIndex].text === ")") depth += 1;
|
||||||
|
if (tokens[tokenIndex].text === "(") depth -= 1;
|
||||||
|
if (depth !== 0) continue;
|
||||||
|
const name = tokens[tokenIndex - 1];
|
||||||
|
return name && name.kind === "identifier" ? name : null;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
function scanSource(source, filePath) {
|
||||||
|
const lexical = lex(source);
|
||||||
|
const balanced = buildCallContexts(lexical.tokens);
|
||||||
|
const tokens = lexical.tokens;
|
||||||
|
const contexts = balanced.contexts;
|
||||||
|
const index = createIndex();
|
||||||
|
const declarations = new Map();
|
||||||
|
const domain = domainFromPath(filePath);
|
||||||
|
|
||||||
|
function mark(token, role, modifiers = ["declaration"]) {
|
||||||
|
declarations.set(token.start, { role, modifiers });
|
||||||
|
}
|
||||||
|
|
||||||
|
function addComponent(token) {
|
||||||
|
index.components.add(token.text);
|
||||||
|
mark(token, "componentName");
|
||||||
|
const alias = componentAlias(token.text);
|
||||||
|
if (alias) {
|
||||||
|
index.componentAliases.add(alias);
|
||||||
|
index.macros.set(alias, prefixDomain(alias) || prefixDomain(token.text) || "component");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
function bindComponentCallees(alias, callees) {
|
||||||
|
if (!alias) return;
|
||||||
|
index.componentAliases.add(alias);
|
||||||
|
index.componentCallees.set(alias, callees);
|
||||||
|
if (!index.macros.has(alias)) index.macros.set(alias, "component");
|
||||||
|
}
|
||||||
|
|
||||||
|
for (let tokenIndex = 0; tokenIndex < tokens.length; tokenIndex += 1) {
|
||||||
|
const token = tokens[tokenIndex];
|
||||||
|
if (token.kind !== "identifier") continue;
|
||||||
|
|
||||||
|
const kind = registerKind(token.text);
|
||||||
|
if (kind) {
|
||||||
|
index.registers.set(token.text, kind);
|
||||||
|
if (tokens[tokenIndex + 1] && tokens[tokenIndex + 1].text === "=") {
|
||||||
|
mark(token, kind === "gpr" ? "gprRegister" : "cop2Register");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const context = nearestCall(contexts, tokenIndex);
|
||||||
|
if (context && context.argIndex === 0) {
|
||||||
|
if (context.callee === "MipsAtom_") {
|
||||||
|
index.atoms.add(token.text);
|
||||||
|
mark(token, "atomName");
|
||||||
|
}
|
||||||
|
if (context.callee === "MipsAtomComp_") addComponent(token);
|
||||||
|
if (context.callee === "atom_bind") index.bindTypes.add(token.text);
|
||||||
|
if (context.callee === "atom_phase" || context.callee === "phase_auto_reg") index.phases.add(token.text);
|
||||||
|
if (context.callee === "atom_label" || context.callee === "atom_offset") index.labels.add(token.text);
|
||||||
|
}
|
||||||
|
|
||||||
|
const isWrappedType = context && (
|
||||||
|
((context.callee === "Struct_" || context.callee === "Union_") && context.argIndex === 0) ||
|
||||||
|
(context.callee === "Enum_" && context.argIndex === 1)
|
||||||
|
);
|
||||||
|
if (isWrappedType) {
|
||||||
|
index.types.add(token.text);
|
||||||
|
mark(token, token.text.startsWith("Binds_") ? "bindType" : "duffleType");
|
||||||
|
if (token.text.startsWith("Binds_")) index.bindTypes.add(token.text);
|
||||||
|
}
|
||||||
|
|
||||||
|
if (context && context.callee === "atom_offset" && context.argIndex === 1) index.labels.add(token.text);
|
||||||
|
|
||||||
|
if (context && context.callee === "atom_auto_reg") {
|
||||||
|
if (context.argIndex === 0) index.atoms.add(token.text);
|
||||||
|
if (context.argIndex === 1) {
|
||||||
|
index.registers.set(token.text, "gpr");
|
||||||
|
mark(token, "gprRegister", ["declaration", "tapeAuto"]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (context && context.callee === "phase_auto_reg" && context.argIndex === 1) {
|
||||||
|
index.registers.set(token.text, "gpr");
|
||||||
|
mark(token, "gprRegister", ["declaration", "tapeAuto"]);
|
||||||
|
}
|
||||||
|
|
||||||
|
if (token.text === "define" && tokens[tokenIndex - 1] && tokens[tokenIndex - 1].text === "#") {
|
||||||
|
const name = tokens[tokenIndex + 1];
|
||||||
|
if (name && name.kind === "identifier" && name.line === token.line) {
|
||||||
|
if (/^(?:RegUse_|Struct_|Enum_|Union_|TypeR_|TypeV_|Relative_|Binds_)/.test(name.text)) {
|
||||||
|
index.types.add(name.text);
|
||||||
|
} else if (/^(?:ac_|mac_)/.test(name.text)) {
|
||||||
|
const alias = name.text.startsWith("ac_") ? componentAlias(name.text) : name.text;
|
||||||
|
const rest = [];
|
||||||
|
for (let restIndex = tokenIndex + 2; restIndex < tokens.length && tokens[restIndex].line === name.line; restIndex += 1) {
|
||||||
|
if (tokens[restIndex].kind === "identifier") rest.push(tokens[restIndex].text);
|
||||||
|
}
|
||||||
|
if (alias) {
|
||||||
|
index.componentAliases.add(alias);
|
||||||
|
index.macros.set(alias, prefixDomain(alias) || "component");
|
||||||
|
if (rest.length) index.componentCallees.set(alias, rest);
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
index.macros.set(name.text, domain || "utility");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (token.text === "typedef") {
|
||||||
|
let endIndex = tokenIndex + 1;
|
||||||
|
let hasBrace = false;
|
||||||
|
let lastIdentifier = null;
|
||||||
|
while (endIndex < tokens.length && tokens[endIndex].text !== ";") {
|
||||||
|
if (tokens[endIndex].text === "{") hasBrace = true;
|
||||||
|
if (tokens[endIndex].kind === "identifier" && !C_BUILTINS.has(tokens[endIndex].text)) lastIdentifier = tokens[endIndex];
|
||||||
|
endIndex += 1;
|
||||||
|
}
|
||||||
|
if (!hasBrace && lastIdentifier && !C_BUILTINS.has(lastIdentifier.text)) {
|
||||||
|
index.types.add(lastIdentifier.text);
|
||||||
|
mark(lastIdentifier, "duffleType");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (token.text === "MipsAtom_Proc_") {
|
||||||
|
const functionName = findFunctionNameBefore(tokens, tokenIndex);
|
||||||
|
if (functionName) {
|
||||||
|
const atomName = functionName.text.endsWith("_proc")
|
||||||
|
? functionName.text.slice(0, -5)
|
||||||
|
: functionName.text;
|
||||||
|
index.atoms.add(atomName);
|
||||||
|
index.atoms.add(functionName.text);
|
||||||
|
mark(functionName, "atomName");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (token.text === "MipsAtomComp_Proc_") {
|
||||||
|
const functionName = findFunctionNameBefore(tokens, tokenIndex);
|
||||||
|
if (functionName) addComponent(functionName);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
for (const call of balanced.calls) {
|
||||||
|
if (call.callee === "MipsAtomComp_") {
|
||||||
|
const name = tokens[call.openTokenIndex + 1];
|
||||||
|
const brace = tokens[call.closeTokenIndex + 1];
|
||||||
|
if (name && name.kind === "identifier" && brace && brace.text === "{") {
|
||||||
|
bindComponentCallees(componentAlias(name.text), collectBraceIdentifiers(tokens, call.closeTokenIndex + 1));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (call.callee === "MipsAtomComp_Proc_") {
|
||||||
|
const functionName = findFunctionNameBefore(tokens, call.calleeTokenIndex);
|
||||||
|
let braceIndex = -1;
|
||||||
|
for (let tokenIndex = call.openTokenIndex + 1; tokenIndex < call.closeTokenIndex; tokenIndex += 1) {
|
||||||
|
if (tokens[tokenIndex].text === "{") {
|
||||||
|
braceIndex = tokenIndex;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (functionName && braceIndex >= 0) {
|
||||||
|
bindComponentCallees(componentAlias(functionName.text), collectBraceIdentifiers(tokens, braceIndex));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (!domain) continue;
|
||||||
|
const name = tokens[call.calleeTokenIndex];
|
||||||
|
const after = tokens[call.closeTokenIndex + 1];
|
||||||
|
if (!name || !after || after.text !== "{") continue;
|
||||||
|
if (/^(?:gp0_|gp1_|gte_|mac_)/.test(name.text)) index.macros.set(name.text, domain);
|
||||||
|
}
|
||||||
|
|
||||||
|
return {
|
||||||
|
index: resolveComponentDomains(cloneIndex(index)),
|
||||||
|
declarations,
|
||||||
|
tokens,
|
||||||
|
contexts,
|
||||||
|
errors: [...lexical.errors, ...balanced.errors],
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
module.exports = {
|
||||||
|
createIndex,
|
||||||
|
domainFromPath,
|
||||||
|
mergeIndexes,
|
||||||
|
resolveComponentDomains,
|
||||||
|
scanSource,
|
||||||
|
};
|
||||||
@@ -0,0 +1,71 @@
|
|||||||
|
{
|
||||||
|
"scopeName": "tape_atom.injection",
|
||||||
|
"injectionSelector": "L:source.c -comment -string, L:source.cpp -comment -string",
|
||||||
|
"patterns": [
|
||||||
|
{ "include": "#atom-declarations" },
|
||||||
|
{ "include": "#component-declarations" },
|
||||||
|
{ "include": "#annotation-arguments" },
|
||||||
|
{ "include": "#annotations" },
|
||||||
|
{ "include": "#delay-slots" },
|
||||||
|
{ "include": "#types" },
|
||||||
|
{ "include": "#attributes" }
|
||||||
|
],
|
||||||
|
"repository": {
|
||||||
|
"atom-declarations": {
|
||||||
|
"patterns": [
|
||||||
|
{
|
||||||
|
"match": "\\b(MipsAtom_)\\s*\\(\\s*([A-Za-z_][A-Za-z0-9_]*)",
|
||||||
|
"captures": {
|
||||||
|
"1": { "name": "keyword.control.duffle.atom" },
|
||||||
|
"2": { "name": "entity.name.function.duffle.atom" }
|
||||||
|
}
|
||||||
|
},
|
||||||
|
{ "match": "\\bMipsAtom_Proc_\\b", "name": "keyword.control.duffle.atom" },
|
||||||
|
{ "match": "\\b[A-Za-z_][A-Za-z0-9_]*_proc\\b", "name": "entity.name.function.duffle.atom" }
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"component-declarations": {
|
||||||
|
"patterns": [
|
||||||
|
{
|
||||||
|
"match": "\\b(MipsAtomComp_)\\s*\\(\\s*(ac_[A-Za-z0-9_]*)",
|
||||||
|
"captures": {
|
||||||
|
"1": { "name": "keyword" },
|
||||||
|
"2": { "name": "entity.name.function.duffle.component" }
|
||||||
|
}
|
||||||
|
},
|
||||||
|
{ "match": "\\bMipsAtomComp_Proc_\\b", "name": "keyword" }
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"annotation-arguments": {
|
||||||
|
"patterns": [
|
||||||
|
{
|
||||||
|
"match": "\\b(atom_offset)\\s*\\(\\s*([A-Za-z_][A-Za-z0-9_]*)\\s*,\\s*([A-Za-z_][A-Za-z0-9_]*)",
|
||||||
|
"captures": {
|
||||||
|
"1": { "name": "support.function.duffle.annotation" },
|
||||||
|
"2": { "name": "entity.name.label.duffle.atom" },
|
||||||
|
"3": { "name": "entity.name.label.duffle.atom" }
|
||||||
|
}
|
||||||
|
},
|
||||||
|
{ "match": "(?<=\\batom_bind\\()\\s*Binds_[A-Za-z0-9_]+", "name": "entity.name.type.duffle.bind" },
|
||||||
|
{ "match": "(?<=\\batom_phase\\()\\s*[A-Za-z_][A-Za-z0-9_]*", "name": "entity.name.tag.duffle.phase" },
|
||||||
|
{ "match": "(?<=\\batom_label\\()\\s*[A-Za-z_][A-Za-z0-9_]*", "name": "entity.name.label.duffle.atom" }
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"annotations": {
|
||||||
|
"match": "\\b(atom_info|atom_bind|atom_reads|atom_writes|atom_label|atom_offset|atom_reg|atom_type|atom_ctx|atom_phase|atom_auto_reg|phase_auto_reg|atom_dbg_skip)\\b",
|
||||||
|
"name": "support.function.duffle.annotation"
|
||||||
|
},
|
||||||
|
"delay-slots": {
|
||||||
|
"match": "\\b(LdSlot_|BdSlot_|DmaSlot_|GteDelay_)\\b",
|
||||||
|
"name": "keyword.operator.duffle.delayslot"
|
||||||
|
},
|
||||||
|
"types": {
|
||||||
|
"match": "\\b(?:Binds_[A-Za-z0-9_]+|RegUse_[A-Za-z0-9_]+)\\b",
|
||||||
|
"name": "storage.type.duffle.type"
|
||||||
|
},
|
||||||
|
"attributes": {
|
||||||
|
"match": "\\b(?:FI_|I_|NI_|Relative_|Struct_|Enum_|Union_|Array_|Slice_|TypeR_|TypeV_|align_|internal|local_persist|global|RO_|LP_|gknown|expect_|cexpr_|asm|asm_words|asm_rpins|asm_clobber|O_|S_|C_|T_|tmpl|glue|r_|v_|tr_|tv_|rgcc|r_use|r_set|r_mod|r_imm|r_mem|u[1248]_|u[1248]_r|u[1248]_v|s[1248]_)\\b",
|
||||||
|
"name": "keyword"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
Binary file not shown.
+128
@@ -0,0 +1,128 @@
|
|||||||
|
"use strict";
|
||||||
|
|
||||||
|
const assert = require("node:assert/strict");
|
||||||
|
const test = require("node:test");
|
||||||
|
|
||||||
|
const { classifyDocument } = require("../classifier");
|
||||||
|
const { createIndex } = require("../source-index");
|
||||||
|
|
||||||
|
function byText(result, text) {
|
||||||
|
return result.spans.filter((span) => span.text === text);
|
||||||
|
}
|
||||||
|
|
||||||
|
test("classifyDocument distinguishes declaration, annotation, phase, bind, and label roles", () => {
|
||||||
|
const source = [
|
||||||
|
"typedef Struct_(Binds_CubeTri) { U4 PrimCursor; };",
|
||||||
|
"MipsAtom_(cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4),",
|
||||||
|
"\tatom_reads(R_PrimCursor), atom_writes(R_FaceCursor)) {",
|
||||||
|
"\tbranch_le_zero(R_T0, atom_offset(cull, exit)),",
|
||||||
|
"\tatom_label(exit)",
|
||||||
|
"};",
|
||||||
|
].join("\n");
|
||||||
|
|
||||||
|
const result = classifyDocument(source, "C:/x/code/hello_camera/hello_camera.atom.c", createIndex());
|
||||||
|
|
||||||
|
assert.equal(byText(result, "MipsAtom_")[0].type, "tapeAtomKeyword");
|
||||||
|
assert.deepEqual(byText(result, "cube_g4_face")[0].modifiers, ["declaration"]);
|
||||||
|
assert.equal(byText(result, "atom_bind")[0].type, "tapeAnnotation");
|
||||||
|
assert.equal(byText(result, "Binds_CubeTri").at(-1).type, "tapeBindType");
|
||||||
|
assert.equal(byText(result, "cube_g4")[0].type, "tapePhase");
|
||||||
|
assert.deepEqual(byText(result, "cube_g4")[0].modifiers, ["declaration"]);
|
||||||
|
assert.equal(byText(result, "cull")[0].type, "tapeLabel");
|
||||||
|
assert.equal(byText(result, "exit").every((span) => span.type === "tapeLabel"), true);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("classifyDocument applies read and write modifiers to GPRs", () => {
|
||||||
|
const source = "atom_info(atom_reads(R_PrimCursor), atom_writes(R_FaceCursor))";
|
||||||
|
const result = classifyDocument(source, "C:/x/code/test.atom.c", createIndex());
|
||||||
|
|
||||||
|
assert.deepEqual(byText(result, "R_PrimCursor")[0].modifiers, ["tapeRead"]);
|
||||||
|
assert.deepEqual(byText(result, "R_FaceCursor")[0].modifiers, ["tapeWrite"]);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("classifyDocument separates CPU, GTE, GPU, and component domains", () => {
|
||||||
|
const workspace = createIndex();
|
||||||
|
workspace.macros.set("load_word", "cpu");
|
||||||
|
workspace.macros.set("gte_cmdw_rtpt", "gte");
|
||||||
|
workspace.macros.set("gp1_word_DisplayOn", "gpu");
|
||||||
|
workspace.macros.set("mac_yield", "control");
|
||||||
|
workspace.componentAliases.add("mac_yield");
|
||||||
|
|
||||||
|
const source = "load_word(R_T0, R_T1, 0), gte_cmdw_rtpt, gp1_word_DisplayOn(), mac_yield(), C2_MAC0, gte_cr_OFX_Code";
|
||||||
|
const result = classifyDocument(source, "C:/x/code/test.c", workspace);
|
||||||
|
|
||||||
|
assert.equal(byText(result, "load_word")[0].type, "tapeCpuInstruction");
|
||||||
|
assert.equal(byText(result, "gte_cmdw_rtpt")[0].type, "tapeGteInstruction");
|
||||||
|
assert.equal(byText(result, "gp1_word_DisplayOn")[0].type, "tapeGpuInstruction");
|
||||||
|
assert.equal(byText(result, "mac_yield")[0].type, "tapeControlFlow");
|
||||||
|
assert.equal(byText(result, "C2_MAC0")[0].type, "tapeCop2Register");
|
||||||
|
assert.equal(byText(result, "gte_cr_OFX_Code")[0].type, "tapeCop2Register");
|
||||||
|
});
|
||||||
|
|
||||||
|
test("component invocations keep the domain resolved from their emitted instructions", () => {
|
||||||
|
const workspace = createIndex();
|
||||||
|
workspace.macros.set("mac_load_word_imm", "cpu");
|
||||||
|
workspace.macros.set("mac_gcmd_push", "gpu");
|
||||||
|
workspace.macros.set("mac_gte_store_f3", "gte");
|
||||||
|
workspace.macros.set("mac_load_v3s4", "cpu");
|
||||||
|
|
||||||
|
const source = "mac_load_word_imm(dst, imm), mac_gcmd_push(cmd), mac_gte_store_f3(cursor), mac_load_v3s4()";
|
||||||
|
const result = classifyDocument(source, "C:/x/code/hello_camera/hello_camera.atom.c", workspace);
|
||||||
|
|
||||||
|
assert.equal(byText(result, "mac_load_word_imm")[0].type, "tapeCpuInstruction");
|
||||||
|
assert.equal(byText(result, "mac_gcmd_push")[0].type, "tapeGpuInstruction");
|
||||||
|
assert.equal(byText(result, "mac_gte_store_f3")[0].type, "tapeGteInstruction");
|
||||||
|
assert.equal(byText(result, "mac_load_v3s4")[0].type, "tapeCpuInstruction");
|
||||||
|
});
|
||||||
|
|
||||||
|
test("utility macros without a hardware domain use the standard macro token", () => {
|
||||||
|
const workspace = createIndex();
|
||||||
|
workspace.macros.set("load_word", "cpu");
|
||||||
|
workspace.macros.set("assert", "utility");
|
||||||
|
workspace.macros.set("stringify", "utility");
|
||||||
|
workspace.macros.set("u4_hi", "utility");
|
||||||
|
|
||||||
|
const source = "load_word(R_T0, R_T1, 0), assert(ok), stringify(name), u4_hi(imm)";
|
||||||
|
const result = classifyDocument(source, "C:/x/code/hello_camera/hello_camera.c", workspace);
|
||||||
|
|
||||||
|
assert.equal(byText(result, "load_word")[0].type, "tapeCpuInstruction");
|
||||||
|
assert.equal(byText(result, "assert")[0].type, "macro");
|
||||||
|
assert.equal(byText(result, "stringify")[0].type, "macro");
|
||||||
|
assert.equal(byText(result, "u4_hi")[0].type, "macro");
|
||||||
|
});
|
||||||
|
|
||||||
|
test("document-local declarations override an empty workspace index", () => {
|
||||||
|
const source = [
|
||||||
|
"MipsAtomComp_(ac_new_component) { nop };",
|
||||||
|
"MipsAtomComp_Proc_(ab, { nop })",
|
||||||
|
"mac_new_component(),",
|
||||||
|
].join("\n");
|
||||||
|
const result = classifyDocument(source, "C:/x/code/duffle/math.atom.c", createIndex());
|
||||||
|
|
||||||
|
assert.equal(byText(result, "MipsAtomComp_")[0].type, "keyword");
|
||||||
|
assert.equal(byText(result, "MipsAtomComp_Proc_")[0].type, "keyword");
|
||||||
|
assert.equal(byText(result, "ac_new_component")[0].type, "tapeComponentName");
|
||||||
|
assert.equal(byText(result, "mac_new_component")[0].type, "tapeComponentInstruction");
|
||||||
|
});
|
||||||
|
|
||||||
|
test("delay slot markers share the tapeDelaySlot token", () => {
|
||||||
|
const source = "LdSlot_ nop, BdSlot_ nop, DmaSlot_ nop2, GteDelay_ nop";
|
||||||
|
const result = classifyDocument(source, "C:/x/code/duffle/gte.atom.c", createIndex());
|
||||||
|
|
||||||
|
assert.equal(byText(result, "LdSlot_")[0].type, "tapeDelaySlot");
|
||||||
|
assert.equal(byText(result, "BdSlot_")[0].type, "tapeDelaySlot");
|
||||||
|
assert.equal(byText(result, "DmaSlot_")[0].type, "tapeDelaySlot");
|
||||||
|
assert.equal(byText(result, "GteDelay_")[0].type, "tapeDelaySlot");
|
||||||
|
});
|
||||||
|
|
||||||
|
test("classifier returns ordered non-overlapping spans and partial malformed output", () => {
|
||||||
|
const source = "atom_reads(R_A /* broken";
|
||||||
|
const result = classifyDocument(source, "C:/x/code/test.atom.c", createIndex());
|
||||||
|
|
||||||
|
assert.equal(result.errors.some((error) => error.kind === "unterminated-block-comment"), true);
|
||||||
|
for (let spanIndex = 1; spanIndex < result.spans.length; spanIndex += 1) {
|
||||||
|
const previous = result.spans[spanIndex - 1];
|
||||||
|
const current = result.spans[spanIndex];
|
||||||
|
assert.equal(previous.start + previous.length <= current.start, true);
|
||||||
|
}
|
||||||
|
});
|
||||||
+88
@@ -0,0 +1,88 @@
|
|||||||
|
"use strict";
|
||||||
|
|
||||||
|
const assert = require("node:assert/strict");
|
||||||
|
const fs = require("node:fs");
|
||||||
|
const path = require("node:path");
|
||||||
|
const test = require("node:test");
|
||||||
|
|
||||||
|
const { TOKEN_MODIFIERS, TOKEN_TYPES } = require("../classifier");
|
||||||
|
|
||||||
|
const ROOT = path.resolve(__dirname, "..");
|
||||||
|
|
||||||
|
function readJson(filePath) {
|
||||||
|
const raw = fs.readFileSync(filePath, "utf8");
|
||||||
|
const stripped = raw.replace(/\/\/.*$/gm, "").replace(/,\s*([}\]])/g, "$1");
|
||||||
|
return JSON.parse(stripped);
|
||||||
|
}
|
||||||
|
|
||||||
|
function collectScopeNames(value, output = new Set()) {
|
||||||
|
if (Array.isArray(value)) {
|
||||||
|
for (const entry of value) collectScopeNames(entry, output);
|
||||||
|
return output;
|
||||||
|
}
|
||||||
|
if (!value || typeof value !== "object") return output;
|
||||||
|
if (typeof value.name === "string") output.add(value.name);
|
||||||
|
for (const child of Object.values(value)) collectScopeNames(child, output);
|
||||||
|
return output;
|
||||||
|
}
|
||||||
|
|
||||||
|
test("package semantic legend matches classifier exports", () => {
|
||||||
|
const packageJson = readJson(path.join(ROOT, "package.json"));
|
||||||
|
const contributedTypes = packageJson.contributes.semanticTokenTypes.map((entry) => entry.id);
|
||||||
|
const contributedModifiers = packageJson.contributes.semanticTokenModifiers.map((entry) => entry.id);
|
||||||
|
|
||||||
|
assert.equal(packageJson.version, "0.3.0");
|
||||||
|
assert.deepEqual(contributedTypes, TOKEN_TYPES);
|
||||||
|
assert.deepEqual(contributedModifiers, TOKEN_MODIFIERS.filter((name) => name !== "declaration"));
|
||||||
|
});
|
||||||
|
|
||||||
|
test("package includes runtime files only and acknowledges local-only metadata", () => {
|
||||||
|
const packageJson = readJson(path.join(ROOT, "package.json"));
|
||||||
|
|
||||||
|
assert.deepEqual(packageJson.files, [
|
||||||
|
"classifier.js",
|
||||||
|
"extension.js",
|
||||||
|
"lexer.js",
|
||||||
|
"source-index.js",
|
||||||
|
"syntaxes/tape_atom.tmLanguage.json",
|
||||||
|
]);
|
||||||
|
assert.equal(packageJson.scripts.package.includes("--allow-missing-repository"), true);
|
||||||
|
assert.equal(packageJson.scripts.package.includes("--skip-license"), true);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("every semantic token has a scope mapping; DSL-specific tokens also have grammar scopes", () => {
|
||||||
|
const packageJson = readJson(path.join(ROOT, "package.json"));
|
||||||
|
const grammar = readJson(path.join(ROOT, "syntaxes", "tape_atom.tmLanguage.json"));
|
||||||
|
const mappings = packageJson.contributes.semanticTokenScopes[0].scopes;
|
||||||
|
const grammarScopes = collectScopeNames(grammar);
|
||||||
|
|
||||||
|
const grammarRequired = new Set([
|
||||||
|
"tapeAtomKeyword", "tapeAtomName", "tapeComponentName",
|
||||||
|
"tapeAnnotation", "tapeBindType", "tapePhase", "tapeLabel",
|
||||||
|
"tapeDelaySlot", "tapeDuffleType", "keyword",
|
||||||
|
]);
|
||||||
|
|
||||||
|
for (const tokenType of TOKEN_TYPES) {
|
||||||
|
assert.equal(Array.isArray(mappings[tokenType]), true, `missing scope mapping: ${tokenType}`);
|
||||||
|
if (grammarRequired.has(tokenType)) {
|
||||||
|
assert.equal(mappings[tokenType].some((scope) => grammarScopes.has(scope)), true, `grammar does not emit: ${tokenType}`);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
test("TextMate offset labels stay scoped to atom_offset calls", () => {
|
||||||
|
const grammar = readJson(path.join(ROOT, "syntaxes", "tape_atom.tmLanguage.json"));
|
||||||
|
const serialized = JSON.stringify(grammar);
|
||||||
|
const offsetRule = grammar.repository["annotation-arguments"].patterns
|
||||||
|
.find((rule) => rule.match.includes("atom_offset"));
|
||||||
|
|
||||||
|
assert.equal(serialized.includes("(?<=,)"), false);
|
||||||
|
assert.equal(offsetRule.captures[1].name, "support.function.duffle.annotation");
|
||||||
|
assert.equal(offsetRule.captures[2].name, "entity.name.label.duffle.atom");
|
||||||
|
assert.equal(offsetRule.captures[3].name, "entity.name.label.duffle.atom");
|
||||||
|
});
|
||||||
|
|
||||||
|
test("workspace enables semantic highlighting", () => {
|
||||||
|
const settings = readJson(path.resolve(ROOT, "..", "settings.json"));
|
||||||
|
assert.equal(settings["editor.semanticHighlighting.enabled"], true);
|
||||||
|
});
|
||||||
+77
@@ -0,0 +1,77 @@
|
|||||||
|
"use strict";
|
||||||
|
|
||||||
|
const assert = require("node:assert/strict");
|
||||||
|
const test = require("node:test");
|
||||||
|
|
||||||
|
const { buildCallContexts, lex, nearestCall } = require("../lexer");
|
||||||
|
|
||||||
|
test("lex skips comments, strings, and character literals", () => {
|
||||||
|
const source = [
|
||||||
|
"MipsAtom_(visible)",
|
||||||
|
"// MipsAtom_(line_comment)",
|
||||||
|
"const char *s = \"atom_reads(R_Hidden)\";",
|
||||||
|
"char c = '\\''; /* gte_cmdw_hidden */",
|
||||||
|
"atom_reads(R_Visible)",
|
||||||
|
].join("\n");
|
||||||
|
|
||||||
|
const result = lex(source);
|
||||||
|
const identifiers = result.tokens
|
||||||
|
.filter((token) => token.kind === "identifier")
|
||||||
|
.map((token) => token.text);
|
||||||
|
|
||||||
|
assert.deepEqual(result.errors, []);
|
||||||
|
assert.equal(identifiers.includes("visible"), true);
|
||||||
|
assert.equal(identifiers.includes("R_Visible"), true);
|
||||||
|
assert.equal(identifiers.includes("line_comment"), false);
|
||||||
|
assert.equal(identifiers.includes("R_Hidden"), false);
|
||||||
|
assert.equal(identifiers.includes("gte_cmdw_hidden"), false);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("lex reports unterminated block comments without returning comment tokens", () => {
|
||||||
|
const result = lex("R_Visible /* atom_reads(R_Hidden)");
|
||||||
|
|
||||||
|
assert.equal(result.tokens.some((token) => token.text === "R_Visible"), true);
|
||||||
|
assert.equal(result.tokens.some((token) => token.text === "R_Hidden"), false);
|
||||||
|
assert.deepEqual(result.errors.map((error) => error.kind), ["unterminated-block-comment"]);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("line comments stop at CRLF boundaries", () => {
|
||||||
|
const result = lex("// atom_reads(R_Hidden)\r\natom_reads(R_Visible)\r\n");
|
||||||
|
const identifiers = result.tokens
|
||||||
|
.filter((token) => token.kind === "identifier")
|
||||||
|
.map((token) => token.text);
|
||||||
|
|
||||||
|
assert.equal(identifiers.includes("R_Hidden"), false);
|
||||||
|
assert.equal(identifiers.includes("R_Visible"), true);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("balanced contexts retain multiline nesting and argument indexes", () => {
|
||||||
|
const source = [
|
||||||
|
"atom_info(",
|
||||||
|
"\tatom_phase(cube_g4),",
|
||||||
|
"\tatom_reads(R_A, nested(R_B, R_C)),",
|
||||||
|
"\tatom_writes(R_D)",
|
||||||
|
")",
|
||||||
|
].join("\n");
|
||||||
|
const lexical = lex(source);
|
||||||
|
const balanced = buildCallContexts(lexical.tokens);
|
||||||
|
|
||||||
|
const byText = new Map();
|
||||||
|
lexical.tokens.forEach((token, index) => {
|
||||||
|
if (token.kind === "identifier") byText.set(token.text, index);
|
||||||
|
});
|
||||||
|
|
||||||
|
assert.equal(nearestCall(balanced.contexts, byText.get("cube_g4")).callee, "atom_phase");
|
||||||
|
assert.equal(nearestCall(balanced.contexts, byText.get("R_A")).callee, "atom_reads");
|
||||||
|
assert.equal(nearestCall(balanced.contexts, byText.get("R_A")).argIndex, 0);
|
||||||
|
assert.equal(nearestCall(balanced.contexts, byText.get("R_C")).callee, "nested");
|
||||||
|
assert.equal(nearestCall(balanced.contexts, byText.get("R_D")).callee, "atom_writes");
|
||||||
|
assert.deepEqual(balanced.errors, []);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("balanced contexts report unmatched parentheses", () => {
|
||||||
|
const lexical = lex("atom_reads(R_A");
|
||||||
|
const balanced = buildCallContexts(lexical.tokens);
|
||||||
|
|
||||||
|
assert.deepEqual(balanced.errors.map((error) => error.kind), ["unmatched-open-paren"]);
|
||||||
|
});
|
||||||
+134
@@ -0,0 +1,134 @@
|
|||||||
|
"use strict";
|
||||||
|
|
||||||
|
const assert = require("node:assert/strict");
|
||||||
|
const test = require("node:test");
|
||||||
|
|
||||||
|
const {
|
||||||
|
createIndex,
|
||||||
|
domainFromPath,
|
||||||
|
mergeIndexes,
|
||||||
|
scanSource,
|
||||||
|
} = require("../source-index");
|
||||||
|
|
||||||
|
test("scanSource discovers current atom and component forms", () => {
|
||||||
|
const source = [
|
||||||
|
"MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4), atom_reads(R_PrimCursor)) { mac_yield() };",
|
||||||
|
"MipsAtomComp_(ac_load_pair) { load_word(R_T0, R_T1, 0) };",
|
||||||
|
"internal MipsAtom* normalize_proc(AtomArena_R aa) MipsAtom_Proc_(aa, { mac_yield() })",
|
||||||
|
"FI_ void ac_store_pair(MipsAtomBuilder_R ab) atom_dbg_skip MipsAtomComp_Proc_(ab, { store_word(R_T0, R_T1, 0) })",
|
||||||
|
].join("\n");
|
||||||
|
|
||||||
|
const result = scanSource(source, "C:/projects/Pikuma/ps1/code/duffle/mips.atom.c");
|
||||||
|
|
||||||
|
assert.equal(result.index.atoms.has("cube_g4_face"), true);
|
||||||
|
assert.equal(result.index.atoms.has("normalize"), true);
|
||||||
|
assert.equal(result.index.components.has("ac_load_pair"), true);
|
||||||
|
assert.equal(result.index.components.has("ac_store_pair"), true);
|
||||||
|
assert.equal(result.index.componentAliases.has("mac_load_pair"), true);
|
||||||
|
assert.equal(result.index.componentAliases.has("mac_store_pair"), true);
|
||||||
|
assert.equal(result.index.macros.get("mac_store_pair"), "component");
|
||||||
|
assert.equal(result.index.componentCallees.get("mac_store_pair").includes("store_word"), true);
|
||||||
|
assert.equal(result.index.phases.has("cube_g4"), true);
|
||||||
|
assert.equal(result.index.registers.get("R_PrimCursor"), "gpr");
|
||||||
|
assert.deepEqual(result.errors, []);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("scanSource discovers binds, labels, registers, typedefs, and macro domains", () => {
|
||||||
|
const source = [
|
||||||
|
"typedef Struct_(Binds_CubeTri) { U4 PrimCursor; };",
|
||||||
|
"typedef Enum_(U4, PadStatus) { PadStatus_Ok };",
|
||||||
|
"typedef U4 const MipsCode;",
|
||||||
|
"enum { R_PrimCursor = R_T7 atom_reg, C2_Custom = 12, gte_cr_Custom = 13 };",
|
||||||
|
"#define load_word(rt, base, off) enc_i(rt, base, off)",
|
||||||
|
"atom_bind(Binds_CubeTri)",
|
||||||
|
"atom_label(exit)",
|
||||||
|
"atom_offset(entry, exit)",
|
||||||
|
].join("\n");
|
||||||
|
|
||||||
|
const result = scanSource(source, "C:/projects/Pikuma/ps1/code/duffle/mips.h");
|
||||||
|
|
||||||
|
assert.equal(result.index.bindTypes.has("Binds_CubeTri"), true);
|
||||||
|
assert.equal(result.index.types.has("PadStatus"), true);
|
||||||
|
assert.equal(result.index.types.has("MipsCode"), true);
|
||||||
|
assert.equal(result.index.registers.get("R_PrimCursor"), "gpr");
|
||||||
|
assert.equal(result.index.registers.get("C2_Custom"), "cop2");
|
||||||
|
assert.equal(result.index.registers.get("gte_cr_Custom"), "cop2");
|
||||||
|
assert.equal(result.index.macros.get("load_word"), "cpu");
|
||||||
|
assert.equal(result.index.labels.has("entry"), true);
|
||||||
|
assert.equal(result.index.labels.has("exit"), true);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("domainFromPath uses the declaration file rather than parent directory names", () => {
|
||||||
|
assert.equal(domainFromPath("C:/x/code/hello_gte/hello_gte.atom.c"), null);
|
||||||
|
assert.equal(domainFromPath("C:/x/code/duffle/mips.h"), "cpu");
|
||||||
|
assert.equal(domainFromPath("C:/x/code/duffle/gte.h"), "gte");
|
||||||
|
assert.equal(domainFromPath("C:/x/code/duffle/gp.h"), "gpu");
|
||||||
|
});
|
||||||
|
|
||||||
|
test("component aliases inherit the domain of the instructions they emit", () => {
|
||||||
|
const headers = mergeIndexes(
|
||||||
|
scanSource("#define load_word(a,b,c) 1\n#define store_word(a,b,c) 1\n#define shift_aright_var(a,b,c) 1\n#define jump_reg(rd) 1\n", "C:/x/code/duffle/mips.h").index,
|
||||||
|
scanSource("#define gte_sw(rt, base, off) 1\n", "C:/x/code/duffle/gte.h").index
|
||||||
|
);
|
||||||
|
const math = scanSource(
|
||||||
|
[
|
||||||
|
"MipsAtomComp_(ac_load_v3s4) { load_word(R_T0, R_T1, 0) };",
|
||||||
|
"#define mac_load_p3s4 mac_load_v3s4",
|
||||||
|
].join("\n"),
|
||||||
|
"C:/x/code/duffle/math.atom.c"
|
||||||
|
);
|
||||||
|
const shift = scanSource(
|
||||||
|
"MipsAtomComp_(ac_shift_aright_var_v3_self) { shift_aright_var(R_T0, R_T0, R_T1) };",
|
||||||
|
"C:/x/code/duffle/gte.atom.c"
|
||||||
|
);
|
||||||
|
const gte = scanSource(
|
||||||
|
"MipsAtomComp_(ac_gte_store_f3) { gte_sw(C2_SXY0, R_T0, 0) };",
|
||||||
|
"C:/x/code/duffle/gte.atom.c"
|
||||||
|
);
|
||||||
|
const yieldAtom = scanSource(
|
||||||
|
"MipsAtomComp_(ac_yield) { load_word(R_AtomJmp, R_TapePtr, 0), jump_reg(R_AtomJmp), nop };",
|
||||||
|
"C:/x/code/duffle/lottes_tape.h"
|
||||||
|
);
|
||||||
|
|
||||||
|
const merged = mergeIndexes(headers, math.index, shift.index, gte.index, yieldAtom.index);
|
||||||
|
assert.equal(merged.macros.get("mac_load_v3s4"), "cpu");
|
||||||
|
assert.equal(merged.macros.get("mac_load_p3s4"), "cpu");
|
||||||
|
assert.equal(merged.macros.get("mac_shift_aright_var_v3_self"), "cpu");
|
||||||
|
assert.equal(merged.macros.get("mac_gte_store_f3"), "gte");
|
||||||
|
assert.equal(merged.macros.get("mac_yield"), "control");
|
||||||
|
});
|
||||||
|
|
||||||
|
test("scanSource tags utility header defines as utility, not a hardware domain", () => {
|
||||||
|
const source = [
|
||||||
|
"#define assert(cond) ((void)(cond))",
|
||||||
|
"#define stringify(name) #name",
|
||||||
|
"#define u4_hi(imm) ((imm) >> 16)",
|
||||||
|
].join("\n");
|
||||||
|
const result = scanSource(source, "C:/projects/Pikuma/ps1/code/duffle/dsl.h");
|
||||||
|
|
||||||
|
assert.equal(result.index.macros.get("assert"), "utility");
|
||||||
|
assert.equal(result.index.macros.get("stringify"), "utility");
|
||||||
|
assert.equal(result.index.macros.get("u4_hi"), "utility");
|
||||||
|
});
|
||||||
|
|
||||||
|
test("mergeIndexes prefers a hardware domain over a later utility define", () => {
|
||||||
|
const left = createIndex();
|
||||||
|
left.macros.set("sub_s", "utility");
|
||||||
|
const right = createIndex();
|
||||||
|
right.macros.set("sub_s", "cpu");
|
||||||
|
|
||||||
|
assert.equal(mergeIndexes(left, right).macros.get("sub_s"), "cpu");
|
||||||
|
assert.equal(mergeIndexes(right, left).macros.get("sub_s"), "cpu");
|
||||||
|
});
|
||||||
|
|
||||||
|
test("mergeIndexes preserves domain-specific aliases", () => {
|
||||||
|
const left = createIndex();
|
||||||
|
left.macros.set("load_word", "cpu");
|
||||||
|
const right = createIndex();
|
||||||
|
right.componentAliases.add("mac_gte_store");
|
||||||
|
right.macros.set("mac_gte_store", "gte");
|
||||||
|
|
||||||
|
const merged = mergeIndexes(left, right);
|
||||||
|
assert.equal(merged.macros.get("load_word"), "cpu");
|
||||||
|
assert.equal(merged.macros.get("mac_gte_store"), "gte");
|
||||||
|
});
|
||||||
@@ -1,24 +1,17 @@
|
|||||||
This is free and unencumbered software released into the public domain.
|
Copyright (C) 2026 Edward R. Gonzalez
|
||||||
|
|
||||||
Anyone is free to copy, modify, publish, use, compile, sell, or
|
This software is provided 'as-is', without any express or implied
|
||||||
distribute this software, either in source code form or as a compiled
|
warranty. In no event will the authors be held liable for any damages
|
||||||
binary, for any purpose, commercial or non-commercial, and by any
|
arising from the use of this software.
|
||||||
means.
|
|
||||||
|
|
||||||
In jurisdictions that recognize copyright laws, the author or authors
|
Permission is granted to anyone to use this software for any purpose,
|
||||||
of this software dedicate any and all copyright interest in the
|
including commercial applications, and to alter it and redistribute it
|
||||||
software to the public domain. We make this dedication for the benefit
|
freely, subject to the following restrictions:
|
||||||
of the public at large and to the detriment of our heirs and
|
|
||||||
successors. We intend this dedication to be an overt act of
|
|
||||||
relinquishment in perpetuity of all present and future rights to this
|
|
||||||
software under copyright law.
|
|
||||||
|
|
||||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
1. The origin of this software must not be misrepresented; you must not
|
||||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
claim that you wrote the original software. If you use this software
|
||||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
in a product, an acknowledgment in the product documentation would be
|
||||||
IN NO EVENT SHALL THE AUTHORS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
appreciated but is not required.
|
||||||
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
2. Altered source versions must be plainly marked as such, and must not be
|
||||||
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
misrepresented as being the original software.
|
||||||
OTHER DEALINGS IN THE SOFTWARE.
|
3. This notice may not be removed or altered from any source distribution.
|
||||||
|
|
||||||
For more information, please refer to <https://unlicense.org>
|
|
||||||
|
|||||||
@@ -1,462 +0,0 @@
|
|||||||
/*
|
|
||||||
* atom_dsl.h
|
|
||||||
* ============================================================================
|
|
||||||
*
|
|
||||||
* ATOM DSL — annotation layer for tape atoms (lottes_tape.h).
|
|
||||||
*
|
|
||||||
* This header turns `__attribute__((annotate(...)))` and `_Pragma(...)` into
|
|
||||||
* a small named DSL that the metaprogram can validate against.
|
|
||||||
*
|
|
||||||
* The C compiler treats every macro below as a no-op:
|
|
||||||
* - atom_init / atom_terminate / atom_bind / atom_setup / atom_commit /
|
|
||||||
* atom_annot all expand to `__attribute__((annotate("..."))) MipsAtom_(name)`
|
|
||||||
* — accepted by GCC (with -Wno-attributes), absent at runtime.
|
|
||||||
* - atom_resource / atom_region / atom_group / atom_cadence / atom_async
|
|
||||||
* expand to `_Pragma("...")` — accepted by any C11 preprocessor.
|
|
||||||
*
|
|
||||||
* The metaprogram (tape_atom_annotation_pass.lua) reads the source-as-written
|
|
||||||
* and validates:
|
|
||||||
* - every MipsAtom_ has one atom_*() annotation (no orphans)
|
|
||||||
* - phase is recognized (init/bind/setup/work/commit/terminate)
|
|
||||||
* - reads/writes reference canonical wave-context registers
|
|
||||||
* - rbind atoms reference a real Binds_* struct declaration
|
|
||||||
* - word-counts in tapre metadata agree with the body's actual .word count
|
|
||||||
* - resource/region/group/cadence/async pragmas are spelled correctly and
|
|
||||||
* reference known enum values
|
|
||||||
*
|
|
||||||
* ============================================================================
|
|
||||||
*
|
|
||||||
* PUTTING IT ON AN ATOM — the canonical pattern
|
|
||||||
*
|
|
||||||
* _tape_resources_
|
|
||||||
* atom_resource(cube_tri, "model_ship_cube")
|
|
||||||
* atom_region (cube_tri, PRIM_ARENA)
|
|
||||||
* atom_group (cube_tri, GROUP_RENDER_PRIMS)
|
|
||||||
* atom_cadence (cube_tri, CADENCE_FRAME)
|
|
||||||
*
|
|
||||||
* atom_annot(cube_tri, phase_work,
|
|
||||||
* tape_regs(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
|
||||||
* tape_regs(R_PrimCursor, R_FaceCursor))
|
|
||||||
* internal MipsAtom_(cube_tri) {
|
|
||||||
* atom_label(culling),
|
|
||||||
* // ... atom body ...
|
|
||||||
* atom_label(bounds_chk),
|
|
||||||
* };
|
|
||||||
*
|
|
||||||
* atom_offset(culling, bounds_chk) // ← branch target, validated
|
|
||||||
*
|
|
||||||
* RBIND pattern — `Binds_*` is the contract
|
|
||||||
*
|
|
||||||
* // Wave-context register layout (declarative):
|
|
||||||
* typedef struct Binds_TrackFaceBatch {
|
|
||||||
* U4 R_PrimCursor, R_FaceCursor,
|
|
||||||
* R_VertBase, R_OtBase;
|
|
||||||
* } Binds_TrackFaceBatch;
|
|
||||||
*
|
|
||||||
* atom_resource(rbind_track_face_batch, "track_face_batch_42")
|
|
||||||
* atom_region (rbind_track_face_batch, HEAP_3D)
|
|
||||||
* atom_group (rbind_track_face_batch, GROUP_LOAD_FACES)
|
|
||||||
* atom_cadence (rbind_track_face_batch, CADENCE_ONDEMAND)
|
|
||||||
* atom_async (rbind_track_face_batch, true)
|
|
||||||
*
|
|
||||||
* atom_bind(rbind_track_face_batch, Binds_TrackFaceBatch,
|
|
||||||
* tape_regs(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase))
|
|
||||||
* internal MipsAtom_(rbind_track_face_batch) { ... };
|
|
||||||
*
|
|
||||||
* Annotation rules
|
|
||||||
* ----------------
|
|
||||||
* 1. Each MipsAtom_(name) needs EXACTLY ONE atom_*() macro on the line
|
|
||||||
* immediately above. No annotation = orphan (warning). Two annotations
|
|
||||||
* on the same name = duplicate (error).
|
|
||||||
*
|
|
||||||
* 2. atom_init and atom_terminate take only the name.
|
|
||||||
*
|
|
||||||
* 3. atom_setup and atom_commit take name + reads.
|
|
||||||
*
|
|
||||||
* 4. atom_bind takes name + Binds_* type + writes.
|
|
||||||
*
|
|
||||||
* 5. atom_annot takes name + phase token + reads + writes.
|
|
||||||
* Phase tokens: phase_init / phase_bind / phase_setup / phase_work /
|
|
||||||
* phase_commit / phase_terminate.
|
|
||||||
*
|
|
||||||
* 6. Optional pragmas (atom_resource / atom_region / atom_group /
|
|
||||||
* atom_cadence / atom_async) attach metadata to the atom. They can
|
|
||||||
* appear in any order, with one per atom. They're independent of the
|
|
||||||
* atom_*() macro — multiple pragmatics are fine.
|
|
||||||
*
|
|
||||||
* ============================================================================
|
|
||||||
*
|
|
||||||
* WHY A SEPARATE LAYER (not just put everything in source comments)?
|
|
||||||
*
|
|
||||||
* Source comments are invisible to the compiler. Annotations live in the
|
|
||||||
* source as actual C tokens, so:
|
|
||||||
* - they can never silently get out of sync with the code (the build
|
|
||||||
* fails at preprocessing if the metaprogram disagrees)
|
|
||||||
* - they can be cross-validated against metadata (build fails if a
|
|
||||||
* WORD_COUNT entry drifts away from the .word count in source)
|
|
||||||
* - they make the C compiler a witness ("there's a marker here, and
|
|
||||||
* it's labelled, and it has arguments") without making the C compile
|
|
||||||
* itself do any work
|
|
||||||
*
|
|
||||||
* ============================================================================
|
|
||||||
*/
|
|
||||||
|
|
||||||
#ifdef INTELLISENSE_DIRECTIVES
|
|
||||||
#pragma once
|
|
||||||
// #include <stdint.h>
|
|
||||||
#endif
|
|
||||||
|
|
||||||
/* ============================================================================
|
|
||||||
* PHASE TOKENS — strings, used as the second arg to atom_annot(...)
|
|
||||||
*
|
|
||||||
* Why strings? They preserve the metaprogram's ability to read phase directly
|
|
||||||
* from the source-as-written, even when the macro isn't expanded. The Lua
|
|
||||||
* tool also has a MACRO_EXPANSION table for resolving phase_* source-level
|
|
||||||
* references.
|
|
||||||
*
|
|
||||||
* atom_annot(cube_tri, phase_work, ...) ← legal
|
|
||||||
* atom_annot(cube_tri, "work", ...) ← legal (and equivalent)
|
|
||||||
* atom_annot(cube_tri, phase_setup, ...) ← legal
|
|
||||||
*
|
|
||||||
* ============================================================================*/
|
|
||||||
|
|
||||||
#define phase_init "init"
|
|
||||||
#define phase_bind "bind"
|
|
||||||
#define phase_setup "setup"
|
|
||||||
#define phase_work "work"
|
|
||||||
#define phase_commit "commit"
|
|
||||||
#define phase_terminate "terminate"
|
|
||||||
|
|
||||||
/* ============================================================================
|
|
||||||
* WAVE-CONTEXT REGISTERS — canonical register set for the tape wave model.
|
|
||||||
*
|
|
||||||
* The tape-atom runtime carries four registers across a wave:
|
|
||||||
*
|
|
||||||
* R_PrimCursor output pointer into the prim arena (next OT entry to write)
|
|
||||||
* R_FaceCursor input pointer into the face array (next face to consume)
|
|
||||||
* R_VertBase base pointer into the vertex arena (this wave's vertices)
|
|
||||||
* R_OtBase base pointer into the ordering table (this wave's OT slot)
|
|
||||||
*
|
|
||||||
* Each atom declares its reads/writes against this canonical set. The Lua
|
|
||||||
* tool rejects wave-context positions that reference any other register
|
|
||||||
* (warning today — the C compiler's R_T4..R_T7 / R_RA / etc. aliases are
|
|
||||||
* implementation details and not part of the typed surface).
|
|
||||||
*
|
|
||||||
* If your atom needs to touch GTE / SP / DMA / other side state, declare it
|
|
||||||
* at the source level as you normally would — but DO NOT put those registers
|
|
||||||
* in tape_regs(...). Wave-context is a closed set.
|
|
||||||
*
|
|
||||||
* ============================================================================*/
|
|
||||||
|
|
||||||
/* ============================================================================
|
|
||||||
* REGION TOKENS — memory regions atoms may allocate from or write into.
|
|
||||||
*
|
|
||||||
* Use atom_region(name, REGION) to declare. The Lua tool validates that the
|
|
||||||
* region is in this set, AND that:
|
|
||||||
* - rbind atoms declare the source region (usually HEAP_3D or CDROM_STREAM)
|
|
||||||
* - work atoms declare the destination region (the arena they push to)
|
|
||||||
* - commit atoms must declare a region equal to what setup wrote, so the
|
|
||||||
* C-side mirror is consistent
|
|
||||||
*
|
|
||||||
* Add new regions by extending this list and the metaprogram's KNOWN_REGIONS.
|
|
||||||
* Don't add regions ad-hoc — every new region becomes part of the contract.
|
|
||||||
*
|
|
||||||
* ============================================================================*/
|
|
||||||
#define REGION_PRIM_ARENA prim_arena /* OT/prim packet arena */
|
|
||||||
#define REGION_FACE_ARENA face_arena /* face index array */
|
|
||||||
#define REGION_VERTEX_ARENA vertex_arena /* vertex pool */
|
|
||||||
#define REGION_OT_ARENA ot_arena /* ordering-table array */
|
|
||||||
#define REGION_HEAP_3D heap_3d_models /* loaded model heap */
|
|
||||||
#define REGION_CDROM_STREAM cdrom_stream /* CDROM read buffer */
|
|
||||||
#define REGION_VRAM vram_heap /* VRAM texture/GPU buffer */
|
|
||||||
|
|
||||||
/* ============================================================================
|
|
||||||
* CADENCE TOKENS — how often the atom runs.
|
|
||||||
*
|
|
||||||
* frame runs every vsync (rendering, input poll)
|
|
||||||
* once runs exactly once per process lifetime (init, terminate)
|
|
||||||
* ondemand runs when triggered by event (CDROM load, async DMA complete)
|
|
||||||
*
|
|
||||||
* Used as a hint for the metaprogram to flag:
|
|
||||||
* - frame-cadence atoms that have side effects (they'll be hit many times,
|
|
||||||
* so avoid global state mutation unless it's idempotent)
|
|
||||||
* - once-cadence atoms inside "if (frame_count == 0)" guards (the guard
|
|
||||||
* is then provably one-shot, the metaprogram can lift initialization)
|
|
||||||
* - ondemand atoms that are missed by the wave scheduler (forces async
|
|
||||||
* and discards yield results without further processing)
|
|
||||||
*
|
|
||||||
* ============================================================================*/
|
|
||||||
#define CADENCE_FRAME frame
|
|
||||||
#define CADENCE_ONCE once
|
|
||||||
#define CADENCE_ONDEMAND ondemand
|
|
||||||
|
|
||||||
/* ============================================================================
|
|
||||||
* tape_regs(...) — wave-context register list
|
|
||||||
*
|
|
||||||
* tape_regs(R_PrimCursor, R_FaceCursor) → (R_PrimCursor, R_FaceCursor)
|
|
||||||
*
|
|
||||||
* The macro produces a comma-evaluated expression that the C compiler
|
|
||||||
* silently discards (it's wrapped in parentheses in the call argument
|
|
||||||
* position — the result is never bound). The Lua tool pattern-matches the
|
|
||||||
* "tape_regs(...)" token to extract the list.
|
|
||||||
*
|
|
||||||
* You can have at most one tape_regs(...) in the reads slot and one in the
|
|
||||||
* writes slot of atom_annot. To declare multiple disjoint sets (rare), just
|
|
||||||
* declare the union — the metaprogram doesn't track which reads need which
|
|
||||||
* writes at this granularity.
|
|
||||||
*
|
|
||||||
* ============================================================================*/
|
|
||||||
#define atom_reads(...) (__VA_ARGS__)
|
|
||||||
#define atom_writes(...) (__VA_ARGS__)
|
|
||||||
|
|
||||||
/* ============================================================================
|
|
||||||
* ATOM ANNOTATION MACROS
|
|
||||||
*
|
|
||||||
* Each expands to `__attribute__((annotate("kind"))) MipsAtom_(name)` —
|
|
||||||
* the GCC attribute is accepted under -Wno-attributes (already in your
|
|
||||||
* build flags) and stripped at runtime. The annotation string is just the
|
|
||||||
* macro kind ("atom_annot", "atom_bind", etc.) — the metaprogram reads
|
|
||||||
* the macro call's full args list from the source-as-written.
|
|
||||||
*
|
|
||||||
* ============================================================================*/
|
|
||||||
|
|
||||||
/* ----------------------------------------------------------------------------
|
|
||||||
* atom_init — entry into tape_runtime_main
|
|
||||||
*
|
|
||||||
* atom_init(tape_main)
|
|
||||||
* internal MipsAtom_(tape_main) { ... };
|
|
||||||
*
|
|
||||||
* Implies: no reads, no writes (wave-context not established yet).
|
|
||||||
* ----------------------------------------------------------------------------*/
|
|
||||||
#define atom_init(name) __attribute__((annotate("atom_init")))
|
|
||||||
|
|
||||||
/* ----------------------------------------------------------------------------
|
|
||||||
* atom_terminate — exit from tape_runtime_main
|
|
||||||
*
|
|
||||||
* atom_terminate(tape_exit)
|
|
||||||
* internal MipsAtom_(tape_exit) { ... };
|
|
||||||
*
|
|
||||||
* Implies: no reads, no writes (wave-context destroyed at this point).
|
|
||||||
* ----------------------------------------------------------------------------*/
|
|
||||||
#define atom_terminate(name) __attribute__((annotate("atom_terminate")))
|
|
||||||
|
|
||||||
/* ----------------------------------------------------------------------------
|
|
||||||
* atom_setup — pre-work atom: prepares engine state (e.g., set_gte_world)
|
|
||||||
*
|
|
||||||
* atom_setup(set_gte_world, tape_regs(R_TapePtr))
|
|
||||||
* internal MipsAtom_(set_gte_world) { ... };
|
|
||||||
*
|
|
||||||
* Reads: anything (the engine state you're reading)
|
|
||||||
* Writes: engine state (GTE / DMA / etc. — declared in source, not part of
|
|
||||||
* wave-context, so doesn't go in tape_regs)
|
|
||||||
*
|
|
||||||
* The metaprogram checks that setup is followed (in atomic order) by a work
|
|
||||||
* atom in the same wave — there's no point in setting up state if no one
|
|
||||||
* reads it.
|
|
||||||
* ----------------------------------------------------------------------------*/
|
|
||||||
#define atom_setup(name, reads) __attribute__((annotate("atom_setup")))
|
|
||||||
|
|
||||||
/* ----------------------------------------------------------------------------
|
|
||||||
* atom_commit — post-work atom: flushes wave-context back to C-side state
|
|
||||||
*
|
|
||||||
* atom_commit(sync_prim_cursor, tape_regs(R_PrimCursor))
|
|
||||||
* internal MipsAtom_(sync_prim_cursor) { ... };
|
|
||||||
*
|
|
||||||
* Reads: wave-context registers (the ones you sync back to C)
|
|
||||||
* Writes: C-side mirror (declared in source — not part of wave-context)
|
|
||||||
*
|
|
||||||
* The metaprogram checks that commit is preceded (in atomic order) by a
|
|
||||||
* work atom that wrote the registers this commit is reading.
|
|
||||||
* ----------------------------------------------------------------------------*/
|
|
||||||
#define atom_commit(name, reads) __attribute__((annotate("atom_commit")))
|
|
||||||
|
|
||||||
/* ----------------------------------------------------------------------------
|
|
||||||
* atom_bind — rbind atom: read wave-context registers from tape pointer
|
|
||||||
*
|
|
||||||
* atom_bind(rbind_cube_tri, Binds_CubeTri,
|
|
||||||
* tape_regs(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase))
|
|
||||||
* internal MipsAtom_(rbind_cube_tri) { ... };
|
|
||||||
*
|
|
||||||
* The binds_struct MUST be a typedef'd type (declared via
|
|
||||||
* `typedef struct Binds_X { ... } Binds_X;` somewhere in the source).
|
|
||||||
* The Lua tool cross-references this. Missing struct = error.
|
|
||||||
*
|
|
||||||
* Implicit: reads R_TapePtr, writes the four wave-context registers.
|
|
||||||
* ----------------------------------------------------------------------------*/
|
|
||||||
#define atom_bind(name, binds_struct, writes) __attribute__((annotate("atom_bind")))
|
|
||||||
|
|
||||||
/* ----------------------------------------------------------------------------
|
|
||||||
* atom_annot — generic work atom with explicit phase
|
|
||||||
*
|
|
||||||
* atom_annot(cube_tri, phase_work,
|
|
||||||
* tape_regs(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
|
||||||
* tape_regs(R_PrimCursor, R_FaceCursor))
|
|
||||||
* internal MipsAtom_(cube_tri) { ... };
|
|
||||||
*
|
|
||||||
* Use this for the bulk of your atoms. For init/setup/commit/bind, prefer
|
|
||||||
* the convenience macros above — they pin the phase for you.
|
|
||||||
*
|
|
||||||
* The phase arg is one of: phase_init / phase_bind / phase_setup /
|
|
||||||
* phase_work / phase_commit / phase_terminate. Spelling mistakes are errors.
|
|
||||||
* ----------------------------------------------------------------------------*/
|
|
||||||
#define atom_annot(name, phase, reads, writes) __attribute__((annotate("atom_annot")))
|
|
||||||
|
|
||||||
/* ============================================================================
|
|
||||||
* RESOURCE / GROUP / CADENCE / REGION / ASYNC — optional atom metadata
|
|
||||||
*
|
|
||||||
* These don't annotate the atom semantically (phase/reads/writes do that).
|
|
||||||
* They attach extra context that the metaprogram uses to catch:
|
|
||||||
* - same resource loaded twice in different ways
|
|
||||||
* - atoms that span multiple regions (likely bug — pick one)
|
|
||||||
* - frame-cadence atoms that should be once-cadence (perf / correctness)
|
|
||||||
* - ondemand atoms that aren't async (CDROM races)
|
|
||||||
*
|
|
||||||
* You can use as many as apply to a given atom, in any order, immediately
|
|
||||||
* above the atom_*() macro.
|
|
||||||
*
|
|
||||||
* ============================================================================*/
|
|
||||||
|
|
||||||
/* ----------------------------------------------------------------------------
|
|
||||||
* atom_resource — name the logical resource the atom references
|
|
||||||
*
|
|
||||||
* atom_resource(cube_tri, "model_ship_cube")
|
|
||||||
* atom_resource(load_track_faces, "track_lavender_field_0x42")
|
|
||||||
* atom_resource(play_engine_sfx, "sfx_engine_loop")
|
|
||||||
*
|
|
||||||
* Use any human-readable string. The metaprogram:
|
|
||||||
* - validates resource strings are non-empty and don't contain control chars
|
|
||||||
* - flags duplicates across atoms with the same name (two atoms claiming
|
|
||||||
* ownership of a resource is usually a refactor artifact or bug)
|
|
||||||
* - flags references to resources that no atom actually defines
|
|
||||||
*
|
|
||||||
* The arg is a STRING LITERAL, so it can't accidentally alias a variable.
|
|
||||||
* ----------------------------------------------------------------------------*/
|
|
||||||
#define atom_resource(name, res_id) //_Pragma("atom " #name " resource=" res_id)
|
|
||||||
|
|
||||||
/* ----------------------------------------------------------------------------
|
|
||||||
* atom_region — name the memory region the atom touches
|
|
||||||
*
|
|
||||||
* atom_region(cube_tri, REGION_PRIM_ARENA)
|
|
||||||
* atom_region(load_faces, REGION_HEAP_3D)
|
|
||||||
* atom_region(load_tex, REGION_VRAM)
|
|
||||||
*
|
|
||||||
* Use REGION_* tokens above. The metaprogram enforces the closed set.
|
|
||||||
*
|
|
||||||
* Edge cases the metaprogram catches:
|
|
||||||
* - rbind atom that doesn't declare a SOURCE region (where is it loading from?)
|
|
||||||
* - work atom with no destination region (where is it pushing to?)
|
|
||||||
* - region that disagrees with the Binds_* struct layout (you said it's a
|
|
||||||
* prim_arena rbind but the struct has 4 faces in it — wait, that's wrong)
|
|
||||||
* ----------------------------------------------------------------------------*/
|
|
||||||
#define atom_region(name, region) //_Pragma("atom " #name " region=" #region)
|
|
||||||
|
|
||||||
/* ----------------------------------------------------------------------------
|
|
||||||
* atom_group — bundle atoms into a logical batch (track-load, sound-load, etc.)
|
|
||||||
*
|
|
||||||
* atom_group(load_track_face_42, GROUP_LOAD_FACES)
|
|
||||||
* atom_group(load_track_face_43, GROUP_LOAD_FACES)
|
|
||||||
* atom_group(swap_face_42_43, GROUP_VISIBILITY_SWAP)
|
|
||||||
*
|
|
||||||
* Use any token as the group id. The metaprogram:
|
|
||||||
* - validates all atoms in a group emit their waves in the same tb_group
|
|
||||||
* (no spawning other waves inside a group)
|
|
||||||
* - flags groups with only one member (probably a typo — meant to be a group?)
|
|
||||||
* - validates cross-group edges (no atom reads what another group writes,
|
|
||||||
* unless explicitly grouped together)
|
|
||||||
*
|
|
||||||
* Useful when:
|
|
||||||
* - subdivisible work (track-face batches, polygon subdivision) needs to
|
|
||||||
* confirm that all batches of one logical visible scene are emitted
|
|
||||||
* together
|
|
||||||
* - async loads (CDROM -> VRAM) need to be grouped so all batches complete
|
|
||||||
* before the swap
|
|
||||||
*
|
|
||||||
* Use GROUPS for sound effects to track which sound plays during which atom,
|
|
||||||
* which is needed if the sound tool ever has to validate "this atom is the
|
|
||||||
* trigger for an audio play".
|
|
||||||
* ----------------------------------------------------------------------------*/
|
|
||||||
#define atom_group(name, group_id) //_Pragma("atom " #name " group=" #group_id)
|
|
||||||
|
|
||||||
/* ----------------------------------------------------------------------------
|
|
||||||
* atom_cadence — declare execution frequency
|
|
||||||
*
|
|
||||||
* atom_cadence(render_frame, CADENCE_FRAME) // every vsync
|
|
||||||
* atom_cadence(load_track_faces, CADENCE_ONDEMAND) // on demand
|
|
||||||
* atom_cadence(init_heap, CADENCE_ONCE) // process lifetime
|
|
||||||
*
|
|
||||||
* Default (no atom_cadence call) is CADENCE_FRAME — most atoms run every
|
|
||||||
* frame. Override explicitly when not.
|
|
||||||
*
|
|
||||||
* The metaprogram's checks:
|
|
||||||
* - CADENCE_ONCE atoms inside `if (frame == 0)` or `if (!initialized)` are
|
|
||||||
* tagged, validating that guards are required (or warning if missing)
|
|
||||||
* - CADENCE_FRAME atoms that mutate state outside the wave context get
|
|
||||||
* flagged (likely a bug — state should persist through commits)
|
|
||||||
* - CADENCE_ONDEMAND atoms must have atom_async — otherwise the trigger
|
|
||||||
* mechanism is undefined
|
|
||||||
* ----------------------------------------------------------------------------*/
|
|
||||||
#define atom_cadence(name, cadence) //_Pragma("atom " #name " cadence=" #cadence)
|
|
||||||
|
|
||||||
/* ----------------------------------------------------------------------------
|
|
||||||
* atom_async — declare whether the atom yields / interacts with CDROM DMA
|
|
||||||
*
|
|
||||||
* atom_async(load_track_tex, true) // CDROM read yield
|
|
||||||
* atom_async(load_vram, true) // VRAM upload DMA
|
|
||||||
* atom_async(render_frame, false) // pure compute, no async
|
|
||||||
*
|
|
||||||
* The metaprogram requires this for CADENCE_ONDEMAND atoms. For
|
|
||||||
* CADENCE_FRAME, it's optional but documents intent.
|
|
||||||
*
|
|
||||||
* Note: CDROM ATOMS in Psy-Q are typically implemented as a chain of
|
|
||||||
* "async-init" atom followed by a "wait-for-completion" atom. Both atoms
|
|
||||||
* should be marked async=true, and both should have the same resource/group
|
|
||||||
* tag (so the metaprogram can verify they're paired).
|
|
||||||
* ----------------------------------------------------------------------------*/
|
|
||||||
#define atom_async(name, is_async) //_Pragma("atom " #name " async=" #is_async)
|
|
||||||
|
|
||||||
/* ============================================================================
|
|
||||||
* WORD-COUNT ANNOTATION FOR A #define MAC
|
|
||||||
*
|
|
||||||
* tape_words(mac_yield, 1)
|
|
||||||
* #define mac_yield() \
|
|
||||||
* load_word(R_AtomJmp, R_TapePtr, 0), \
|
|
||||||
* add_ui_self(R_TapePtr, 4), \
|
|
||||||
* jump_reg(R_AtomJmp), \
|
|
||||||
* nop
|
|
||||||
*
|
|
||||||
* The compiler accepts the unknown _Pragma. The Lua tool reads it and
|
|
||||||
* cross-checks against WORD_COUNT(mac_yield, 1) in tape_atom.metadata.h.
|
|
||||||
* If they disagree, build fails.
|
|
||||||
*
|
|
||||||
* Use sparingly — only on multi-word macros (single-word ones don't need
|
|
||||||
* drift tracking; they're checked by the .word-count pass anyway).
|
|
||||||
*
|
|
||||||
* ============================================================================*/
|
|
||||||
#define tape_words(name, n) //_Pragma(#name " tape_atom words=" #n)
|
|
||||||
|
|
||||||
/* ============================================================================
|
|
||||||
* atom_label / atom_offset — branch target machinery
|
|
||||||
*
|
|
||||||
* atom_label(culling) ← nothing in C; anchor only
|
|
||||||
* ... body ...
|
|
||||||
* atom_label(bounds_chk) ← another anchor
|
|
||||||
*
|
|
||||||
* atom_offset(culling, bounds_chk) ← resolved by gen/.offsets.h
|
|
||||||
*
|
|
||||||
* The metaprogram generates gen/atom_offsets.h with one
|
|
||||||
* #define atom_offset__culling__bounds_chk ((target - branch_pos - 1))
|
|
||||||
* per atom_offset(F, T) call. The preprocessor then expands your call to
|
|
||||||
* the right immediate value.
|
|
||||||
*
|
|
||||||
* If gen/atom_offsets.h is stale (or atom_label(name) is undefined),
|
|
||||||
* `atom_offset__F__T` becomes an undefined macro and the C build fails.
|
|
||||||
* This catches:
|
|
||||||
* - typo in atom_label (no anchor → metaprogram doesn't emit the macro)
|
|
||||||
* - .offsets.h not regenerated after body edits
|
|
||||||
* - body edit that broke the offset math (recompile + retest picks it up
|
|
||||||
* in CPU emulator)
|
|
||||||
*
|
|
||||||
* ============================================================================*/
|
|
||||||
|
|
||||||
#define atom_offset(F, T) atom_offset_ ## F ## _ ## T
|
|
||||||
#define atom_label(name) /* anchor — see metaprogram documentation */
|
|
||||||
@@ -0,0 +1,15 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# pragma once
|
||||||
|
#endif
|
||||||
|
|
||||||
|
enum {
|
||||||
|
bios_init_pad_2 = 0x12,
|
||||||
|
bios_start_pad_2 = 0x13,
|
||||||
|
bios_flushcache = 0x44,
|
||||||
|
bios_table_addr = 0xA0,
|
||||||
|
bios_btable_addr = 0xB0,
|
||||||
|
};
|
||||||
|
|
||||||
|
enum {
|
||||||
|
bios_pad_buffer_size = 0x22,
|
||||||
|
};
|
||||||
@@ -0,0 +1,181 @@
|
|||||||
|
/*
|
||||||
|
* dsl.atom.h
|
||||||
|
* ============================================================================
|
||||||
|
*
|
||||||
|
* ATOM DSL: Annotation layer for tape atoms (lottes_tape.h).
|
||||||
|
* The metaprogram (scripts/passes/annotation.lua) reads source-as-written and validates:
|
||||||
|
* - atom_info(...) shape: up to three sub-calls (atom_bind(Binds_X), atom_reads(...), atom_writes(...)) in any order and are optional.
|
||||||
|
* - rbind atoms (atom_info(..., atom_bind(Binds_X), ...)) reference a real Binds_* struct declaration.
|
||||||
|
* - atom word-counts in word_counts.metadata.h match the body's actual .word count.
|
||||||
|
*
|
||||||
|
* Pure macro anntation.
|
||||||
|
* ---------------
|
||||||
|
* Don't want to constraint the macro usage to some attribute placment constraint, etc, don't want ot dela with the compiler.
|
||||||
|
* atom_info, atom_bind, atom_reads, atom_writes, atom_label, atom_dbg_skip each expand to a C comment or to nothing
|
||||||
|
* (C preprocessor strips them to whitespace).
|
||||||
|
*
|
||||||
|
* ============================================================================
|
||||||
|
* Usage:
|
||||||
|
* MipsAtom_(cube_tri) atom_info(
|
||||||
|
* atom_reads (R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||||
|
* , atom_writes(R_PrimCursor, R_FaceCursor)
|
||||||
|
* ){
|
||||||
|
* atom_label(culling),
|
||||||
|
* // ... atom body ...
|
||||||
|
* atom_offset(culling, bounds_chk) // branch target, validated
|
||||||
|
* // ... atom body ...
|
||||||
|
* atom_label(bounds_chk),
|
||||||
|
* };
|
||||||
|
*
|
||||||
|
*
|
||||||
|
* Data Binding pattern -- atom_bind as a sub-call of atom_info
|
||||||
|
*
|
||||||
|
* // Wave-context register layout (declarative):
|
||||||
|
* typedef Struct_(Binds_TrackFaceBatch) {
|
||||||
|
* U4 PrimCursor;
|
||||||
|
* U4 FaceCursor;
|
||||||
|
* U4 VertBase;
|
||||||
|
* U4 OtBase;
|
||||||
|
* };
|
||||||
|
* MipsAtom_(rbind_track_face_batch) atom_info(
|
||||||
|
* atom_bind(Binds_TrackFaceBatch)
|
||||||
|
* , atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||||
|
* ){ ... };
|
||||||
|
*
|
||||||
|
* Annotation rules
|
||||||
|
* ----------------
|
||||||
|
* 1. atom_info(...) is OPTIONAL. Atoms without atom_info are silently skipped by the metaprogram.
|
||||||
|
* 2. If present, atom_info takes up to three sub-calls, all order-independent within the arg list:
|
||||||
|
* - atom_bind(Binds_X)
|
||||||
|
* - atom_reads(...)
|
||||||
|
* - atom_writes(...)
|
||||||
|
* 3. atom_bind(Binds_X): metaprogram cross-references Binds_X against the `typedef struct Binds_X { ... } Binds_X;` declaration.
|
||||||
|
* 4. atom_reads(...) and atom_writes(...): Used to to check if registers are used correctly in macros: R_PrimCursor / R_FaceCursor / R_VertBase / R_OtBase.
|
||||||
|
* 5. atom_label(name: Utilize with atom_offset as a target location.
|
||||||
|
* 6. atom_offset(F, T): Resolved by gen/atom_offsets.h, generated from the atom_label markers. Calculated during the offset pass of the lua metaprogram.
|
||||||
|
*/
|
||||||
|
|
||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
#pragma once
|
||||||
|
#endif
|
||||||
|
|
||||||
|
/* ============================================================================
|
||||||
|
* atom_reads(...) / atom_writes(...)
|
||||||
|
*
|
||||||
|
* Used during the static analysis pass of the metaprogram to do
|
||||||
|
* ============================================================================*/
|
||||||
|
#define atom_reads(...) (__VA_ARGS__)
|
||||||
|
#define atom_writes(...) (__VA_ARGS__)
|
||||||
|
|
||||||
|
/* ----------------------------------------------------------------------------
|
||||||
|
* atom_reg (per-enum opt-in marker for the DWARF register-alias registry)
|
||||||
|
*
|
||||||
|
* Bare `atom_reg` token adjacent to an enum entry that alias as debug-visible for scan_source's register_alias_registry.
|
||||||
|
* Lua scanner reads the bare token.
|
||||||
|
* ----------------------------------------------------------------------------*/
|
||||||
|
#define atom_reg /* atom_reg: opt the preceding enum entry into the DWARF registry */
|
||||||
|
|
||||||
|
// ----------------------------------------------------------------------------
|
||||||
|
// atom_auto_reg(atom, sym) — per-atom auto-allocated GPR binding.
|
||||||
|
// enum {
|
||||||
|
// atom_auto_reg(cube_g4_face, R_Fwdx), // expands to: R_Fwdx = R_Fwdx_Code /* atom_auto_reg: cube_g4_face */,
|
||||||
|
// atom_auto_reg(cube_g4_face, R_Eye_z) atom_type(S4), // atom_type chains after
|
||||||
|
// };
|
||||||
|
// (The macro IS the entire enum entry — no separate LHS=RHS. The `atom` scope is
|
||||||
|
// preserved in a trailing C-comment on the RHS so the Lua scanner can recover
|
||||||
|
// it after preprocessing strips the macro form. R_<Sym>_Code is resolved from gen/auto_reg.h which the .c file #include's before the enum declaration.)
|
||||||
|
#define atom_auto_reg(atom, sym) sym = sym ## _Code /* atom_auto_reg: atom */
|
||||||
|
|
||||||
|
// ----------------------------------------------------------------------------
|
||||||
|
// phase_auto_reg(phase, sym) — per-phase auto-allocated GPR binding.
|
||||||
|
// enum {
|
||||||
|
// phase_auto_reg(cube_g4, R_Temp0), // expands to: R_Temp0 = R_Temp0_Code /* phase_auto_reg: cube_g4 */,
|
||||||
|
// phase_auto_reg(cube_g4, R_Temp1),
|
||||||
|
// };
|
||||||
|
// (Same macro-as-enum-entry form as atom_auto_reg above; the `phase` scope is preserved in a trailing C-comment on the RHS for the Lua scanner to recover.)
|
||||||
|
#define phase_auto_reg(phase, sym) sym = sym ## _Code /* phase_auto_reg: phase */
|
||||||
|
|
||||||
|
/* ============================================================================
|
||||||
|
* atom_info :
|
||||||
|
* MipsAtom_(cube_tri) atom_info(
|
||||||
|
* atom_reads (R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||||
|
* , atom_writes(R_PrimCursor, R_FaceCursor)
|
||||||
|
* ){ ... };
|
||||||
|
*
|
||||||
|
* - atom_bind(Binds_X): metaprogram cross-references Binds_X against the `typedef struct Binds_X { ... } Binds_X;` declaration.
|
||||||
|
* - atom_reads(...): comma-list of registers
|
||||||
|
* - atom_writes(...): comma-list of registers
|
||||||
|
* ============================================================================*/
|
||||||
|
#define atom_info(...) /* atom_info(__VA_ARGS__) */
|
||||||
|
|
||||||
|
/* ----------------------------------------------------------------------------
|
||||||
|
* DEBUG SOURCE-STEP MARKER
|
||||||
|
*
|
||||||
|
* Place `atom_dbg_skip` (BARE) before a MipsAtom_, MipsAtomComp_, or MipsAtomComp_Proc_.
|
||||||
|
* The following declaration kind determines whether the marker selects a whole atom or a component inline view.
|
||||||
|
* The source scanner associates the marker with that declaration; placement diagnostics are handled by the annotation pass.
|
||||||
|
*
|
||||||
|
* Example:
|
||||||
|
* atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(rret_addr), nop };
|
||||||
|
* atom_dbg_skip MipsAtomComp_(ac_yield) { ... };
|
||||||
|
* atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, { ... });
|
||||||
|
* ----------------------------------------------------------------------------*/
|
||||||
|
#define atom_dbg_skip /* atom_dbg_skip: skip the following atom or component source view */
|
||||||
|
|
||||||
|
/* ----------------------------------------------------------------------------
|
||||||
|
* Typed-view annotations (Registry for DWARF RR_<R_X> chain resolution)
|
||||||
|
* atom_type(<T>) -- overloaded:
|
||||||
|
* (a) enum-site default: `R_Foo = R_Tn, atom_reg atom_type(T)`
|
||||||
|
* Sets the per-alias default typed view in the register_alias_registry.
|
||||||
|
* Consumed by the DWARF chain step (e) when no per-atom atom_ctx / atom_phase / atom_type callsite provides a stronger resolution.
|
||||||
|
* (b) callsite override: `atom_reads(R_Foo atom_type(T), ...)` Overrides the per-alias default for THIS atom only.
|
||||||
|
* Last-write-wins per R_Name; conflict -> error.
|
||||||
|
* atom_ctx(<atom_name>) -- atom-info sub-call:
|
||||||
|
* Propagate another atom's atom.rbind.fields (its Binds_* typed fields) into THIS atom's typed-view resolution.
|
||||||
|
* The named atom must be an rbind atom (have `atom_bind(Binds_X)` in its `atom_info`).
|
||||||
|
* Used as the escape hatch when atom_phase is not the natural correlation.
|
||||||
|
* atom_phase(<label>) -- atom-info sub-call:
|
||||||
|
* Free-form C-identifier label for grouping atoms.
|
||||||
|
* Within a phase, the FIRST atom in source-order that owns its own atom.rbind provides
|
||||||
|
* the Binds_* field types used by all other atoms in the same phase.
|
||||||
|
* The preferred correlation mechanism; atom_ctx is the escape hatch for non-natural cases.
|
||||||
|
*
|
||||||
|
* All three expand to C comments
|
||||||
|
* (the bare-token convention matching `atom_reg` and `atom_dbg_skip`).
|
||||||
|
* The Lua scanner reads the bare tokens in source-as-written; the C preprocessor strips them.
|
||||||
|
* ----------------------------------------------------------------------------*/
|
||||||
|
#define atom_type(T) /* atom_type: associate <T> with the preceding enum entry (enum site) or this register (atom-info site) */
|
||||||
|
#define atom_ctx(atom_name) /* atom_ctx: propagate <atom_name>'s Binds_* field types into this atom's typed views */
|
||||||
|
#define atom_phase(label) /* atom_phase: tag this atom with <label> for grouped typed-view resolution */
|
||||||
|
|
||||||
|
/* ----------------------------------------------------------------------------
|
||||||
|
* atom_bind(Binds_X) -- rbind sub-call of atom_info
|
||||||
|
*
|
||||||
|
* MipsAtom_(rbind_cube_tri) atom_info(
|
||||||
|
* atom_bind(Binds_CubeTri)
|
||||||
|
* , atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||||
|
* ){ ... };
|
||||||
|
*
|
||||||
|
* The Binds_X MUST be a typedef'd type (declared via `typedef struct Binds_X { ... } Binds_X;` somewhere in the source).
|
||||||
|
* ----------------------------------------------------------------------------*/
|
||||||
|
#define atom_bind(binds_struct) /* atom_bind(binds_struct) */
|
||||||
|
|
||||||
|
#define Binds_(type) (tmpl(Binds,type)) // TODO(Ed): Do we want to use this?
|
||||||
|
|
||||||
|
/* ============================================================================
|
||||||
|
* atom_label / atom_offset — branch target machinery
|
||||||
|
*
|
||||||
|
* atom_label(culling) ← nothing in C; anchor only
|
||||||
|
* ... body ...
|
||||||
|
* atom_label(bounds_chk) ← another anchor
|
||||||
|
*
|
||||||
|
* atom_offset(culling, bounds_chk) ← resolved by gen/offsets.h
|
||||||
|
*
|
||||||
|
* The metaprogram generates gen/offsets.h with one #define with the offset value per atom_offset(F, T) call.
|
||||||
|
* The preprocessor then expands the call to the right immediate value.
|
||||||
|
*
|
||||||
|
* If gen/offsets.h is stale (or atom_label(name) is undefined), `atom_offset_F_T` becomes an undefined macro and the C build fails.
|
||||||
|
* ============================================================================*/
|
||||||
|
#define atom_offset(F, T) atom_offset_ ## F ## _ ## T
|
||||||
|
// atom_label is a pure annotation for the metaprogram's offset calculations.
|
||||||
|
#define atom_label(name) /* atom_label anchor: name */
|
||||||
+39
-25
@@ -3,7 +3,7 @@
|
|||||||
# include "assert.h"
|
# include "assert.h"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#define offset_of(type, member) cast(U8,__builtin_offsetof(type,member))
|
#define offset_of(type, member) cast(U8,__builtin_offsetof(type,member)) // Compiler builtin version of O_
|
||||||
#define static_assert _Static_assert
|
#define static_assert _Static_assert
|
||||||
#define typeof __typeof__
|
#define typeof __typeof__
|
||||||
#define typeof_ptr(ptr) typeof((ptr)[0])
|
#define typeof_ptr(ptr) typeof((ptr)[0])
|
||||||
@@ -28,8 +28,9 @@
|
|||||||
#define internal static // internal
|
#define internal static // internal
|
||||||
|
|
||||||
#define asm __asm__
|
#define asm __asm__
|
||||||
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
|
||||||
|
|
||||||
|
#define A_(data) (& data)
|
||||||
|
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
||||||
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
||||||
#define C_(type,data) ((type)(data)) // for enforced precedence
|
#define C_(type,data) ((type)(data)) // for enforced precedence
|
||||||
#define expect_(x, y) __builtin_expect(x, y) // so compiler knows the common path
|
#define expect_(x, y) __builtin_expect(x, y) // so compiler knows the common path
|
||||||
@@ -43,7 +44,9 @@
|
|||||||
|
|
||||||
#define R_ restrict
|
#define R_ restrict
|
||||||
#define V_ volatile
|
#define V_ volatile
|
||||||
// Fictional, used for intiution.
|
|
||||||
|
#pragma region Fictional //, used for intiution
|
||||||
|
|
||||||
#define EUB_ restrict // Execute Unit Bound: Data is siloed in the ALU Register File. The Load/Store Unit is bypassed. (Route to Execution Unit. Keep in registers)
|
#define EUB_ restrict // Execute Unit Bound: Data is siloed in the ALU Register File. The Load/Store Unit is bypassed. (Route to Execution Unit. Keep in registers)
|
||||||
#define ISO_ restrict // Isolated Provenance: Alternative to Exu_. Guarantees electrical memory isolation,
|
#define ISO_ restrict // Isolated Provenance: Alternative to Exu_. Guarantees electrical memory isolation,
|
||||||
// unlocking the compiler’s ability to safely pack data across multiple parallel SIMD lanes (vectorization).
|
// unlocking the compiler’s ability to safely pack data across multiple parallel SIMD lanes (vectorization).
|
||||||
@@ -67,7 +70,8 @@
|
|||||||
#define latch_load_anchor(ptr) //__atomic_load_n(ptr, ooo_anchor_)
|
#define latch_load_anchor(ptr) //__atomic_load_n(ptr, ooo_anchor_)
|
||||||
#define latch_store_drain(ptr, val) //__atomic_store_n(ptr, val, ooo_drain_)
|
#define latch_store_drain(ptr, val) //__atomic_store_n(ptr, val, ooo_drain_)
|
||||||
#define pulse_xchg_weld(ptr, val) //__atomic_exchange_n(ptr, val, ooo_weld_)
|
#define pulse_xchg_weld(ptr, val) //__atomic_exchange_n(ptr, val, ooo_weld_)
|
||||||
//end of: Fictional.
|
|
||||||
|
#pragma endreigon Fictional
|
||||||
|
|
||||||
|
|
||||||
// R_ (restrict) establishes an "Eigen" or "Proprius" mapping.
|
// R_ (restrict) establishes an "Eigen" or "Proprius" mapping.
|
||||||
@@ -87,22 +91,23 @@
|
|||||||
#define PtrSet_(type) TypeR_(type); typedef TypeV_(type)
|
#define PtrSet_(type) TypeR_(type); typedef TypeV_(type)
|
||||||
#define TSet_(type) type; typedef PtrSet_(type)
|
#define TSet_(type) type; typedef PtrSet_(type)
|
||||||
|
|
||||||
#define array_len(a) (U4)(sizeof(a) / sizeof(typeof((a)[0])))
|
#define Array_len(a) (U4)(sizeof(a) / sizeof(typeof((a)[0])))
|
||||||
#define array_decl(type, ...) (type[]){__VA_ARGS__}
|
#define Array_decl(type, ...) (type[]){__VA_ARGS__}
|
||||||
#define Array_sym(type,len) A ## len ## _ ## type
|
#define Array_sym(type,len) A ## len ## _ ## type
|
||||||
#define Array_expand(type,len) type Array_sym(type, len)[len]; typedef PtrSet_(Array_sym(type, len))
|
#define Array_expand(type,len) type Array_sym(type, len)[len]; typedef PtrSet_(Array_sym(type, len))
|
||||||
#define Array_(type,len) Array_expand(type,len)
|
#define Array_(type,len) Array_expand(type,len)
|
||||||
#define Bit_(id,b) id = (1 << b), tmpl(id,pos) = b
|
#define Bit_(id,b) id = (1 << b), tmpl(id,pos) = b
|
||||||
|
#define Bitmask_(b) (1u << b)
|
||||||
#define Enum_(underlying_type, symbol) underlying_type TSet_(symbol); enum symbol
|
#define Enum_(underlying_type, symbol) underlying_type TSet_(symbol); enum symbol
|
||||||
#define Proc_(symbol) symbol
|
#define Proc_(symbol) symbol
|
||||||
#define Relative_(symbol) // Does nothing but annotate that a symbol is associated with another.
|
#define Relative_(symbol) // Does nothing but annotate that a symbol is associated with another.
|
||||||
#define Struct_(symbol) struct symbol TSet_(symbol); struct symbol
|
#define Struct_(symbol) struct symbol TSet_(symbol); struct symbol
|
||||||
#define Union_(symbol) union symbol TSet_(symbol); union symbol
|
#define Union_(symbol) union symbol TSet_(symbol); union symbol
|
||||||
|
|
||||||
#define Opt_(proc) Struct_(tmpl(Opt,proc))
|
#define Opt_(proc) Struct_(tmpl(Opt,proc))
|
||||||
#define opt_(symbol, ...) (tmpl(Opt,symbol)){__VA_ARGS__}
|
#define opt_(symbol, ...) (tmpl(Opt,symbol)){__VA_ARGS__}
|
||||||
#define Ret_(proc) Struct_(tmpl(Ret,proc))
|
#define Ret_(proc) Struct_(tmpl(Ret,proc))
|
||||||
#define ret_(proc) tmpl(Ret,proc) proc
|
#define ret_(proc) tmpl(Ret,proc) proc
|
||||||
|
|
||||||
// Using Byte-Width convention for the fundamental types.
|
// Using Byte-Width convention for the fundamental types.
|
||||||
typedef __UINT8_TYPE__ TSet_(U1);
|
typedef __UINT8_TYPE__ TSet_(U1);
|
||||||
@@ -130,20 +135,21 @@ typedef __UINT32_TYPE__ TSet_(B4);
|
|||||||
#define u4_v(value) C_(U4 V_*, value)
|
#define u4_v(value) C_(U4 V_*, value)
|
||||||
enum { false = 0, true = 1, true_overflow, };
|
enum { false = 0, true = 1, true_overflow, };
|
||||||
|
|
||||||
#define u4_lo(value) ((value) & 0xFFFFU)
|
#define u4_lo(value) (u4_(value) & 0xFFFFU)
|
||||||
#define u4_hi(value) ((value) >> 12)
|
#define u4_hi(value) (u4_(value) >> (S_(U2) * 8))
|
||||||
|
|
||||||
typedef void Proc_(VoidFn) (void);
|
typedef void Proc_(VoidFn) (void);
|
||||||
|
|
||||||
#define kilo(n) (C_(U4, n) << 10)
|
#define Kilo_(n) (C_(U4, n) << 10)
|
||||||
#define mega(n) (C_(U4, n) << 20)
|
#define Mega_(n) (C_(U4, n) << 20)
|
||||||
#define giga(n) (C_(U4, n) << 30)
|
#define Giga_(n) (C_(U4, n) << 30)
|
||||||
#define tera(n) (C_(U4, n) << 40)
|
#define Tera_(n) (C_(U4, n) << 40)
|
||||||
|
|
||||||
#define null C_(U4, 0)
|
#define null C_(U4, 0)
|
||||||
#define nullptr C_(void*, 0)
|
#define nullptr C_(void*, 0)
|
||||||
#define O_(type, field) (C_(U4, & C_(type*,0)->field))
|
#define O_(type, field) C_(U4, & C_(type*,0)->field)
|
||||||
|
#define OA_(type, aexpr) C_(U4, & C_(type*,0) aexpr)
|
||||||
#define OT_(field) O_(typeof_ptr(& field), filed))
|
#define OT_(field) O_(typeof_ptr(& field), field))
|
||||||
#define S_(data) C_(U4, sizeof(data))
|
#define S_(data) C_(U4, sizeof(data))
|
||||||
|
|
||||||
#define sop_1(op,a,b) C_(U1, s1_(a) op s1_(b))
|
#define sop_1(op,a,b) C_(U1, s1_(a) op s1_(b))
|
||||||
@@ -164,6 +170,8 @@ def_signed_ops(le, <=)
|
|||||||
#undef def_signed_ops
|
#undef def_signed_ops
|
||||||
#undef def_signed_op
|
#undef def_signed_op
|
||||||
|
|
||||||
|
// Unused, we arent' doing any C-like asm since we have the asm dsl. We'll keep the non-generics if we somehow do.
|
||||||
|
#if 0
|
||||||
#define def_generic_sop(op, a, ...) _Generic((a), U1: op ## _s1, U2: op ## _s2, U4: op ## _s4) (a, __VA_ARGS__)
|
#define def_generic_sop(op, a, ...) _Generic((a), U1: op ## _s1, U2: op ## _s2, U4: op ## _s4) (a, __VA_ARGS__)
|
||||||
#define add_s(a,b) def_generic_sop(add,a,b)
|
#define add_s(a,b) def_generic_sop(add,a,b)
|
||||||
#define sub_s(a,b) def_generic_sop(sub,a,b)
|
#define sub_s(a,b) def_generic_sop(sub,a,b)
|
||||||
@@ -173,11 +181,12 @@ def_signed_ops(le, <=)
|
|||||||
#define ge_s(a,b) def_generic_sop(ge, a,b)
|
#define ge_s(a,b) def_generic_sop(ge, a,b)
|
||||||
#define le_s(a,b) def_generic_sop(le, a,b)
|
#define le_s(a,b) def_generic_sop(le, a,b)
|
||||||
#undef def_generic_sop
|
#undef def_generic_sop
|
||||||
|
#endif
|
||||||
|
|
||||||
#define alignas _Alignas
|
#define alignas _Alignas
|
||||||
#define alignof _Alignof
|
#define alignof _Alignof
|
||||||
#define byte_pad(amount, ...) B1 glue(_PAD_, __VA_ARGS__) [amount]
|
#define byte_pad(amount, ...) B1 glue(_PAD_, __VA_ARGS__) [amount]
|
||||||
#define pcast(type, data) (C_(type*, & (data)) [0])
|
#define C_ptr(type, data) (C_(type*, & (data)) [0])
|
||||||
|
|
||||||
#define dbg_args(...) __VA_ARGS__
|
#define dbg_args(...) __VA_ARGS__
|
||||||
|
|
||||||
@@ -192,6 +201,8 @@ def_signed_ops(le, <=)
|
|||||||
#define defer_info(type,expr, ...) for(type info= {__VA_ARGS__}; info.once!=1;++info.once,(expr)) // Defer with tracked state
|
#define defer_info(type,expr, ...) for(type info= {__VA_ARGS__}; info.once!=1;++info.once,(expr)) // Defer with tracked state
|
||||||
|
|
||||||
#define do_while(cond) for (U8 once=0; once!=1 || (cond); ++once)
|
#define do_while(cond) for (U8 once=0; once!=1 || (cond); ++once)
|
||||||
|
|
||||||
|
#define Jmp_nZero_(cond,label) if (cond) goto label;
|
||||||
#pragma endregion Control Flow & Iteration
|
#pragma endregion Control Flow & Iteration
|
||||||
|
|
||||||
#define span_iter(type, iter, m_begin, op, m_end) ( \
|
#define span_iter(type, iter, m_begin, op, m_end) ( \
|
||||||
@@ -208,13 +219,16 @@ def_signed_ops(le, <=)
|
|||||||
typedef Span_(S4);
|
typedef Span_(S4);
|
||||||
typedef Span_(U4);
|
typedef Span_(U4);
|
||||||
|
|
||||||
#if 0
|
|
||||||
#pragma region Debug
|
#pragma region Debug
|
||||||
#define debug_trap() __builtin_debugtrap()
|
#define debug_trap() __builtin_trap()
|
||||||
#if BUILD_DEBUG
|
#if BUILD_DEBUG
|
||||||
IA_ void assert(U8 cond) { if(cond){return;} else{debug_trap(); ms_exit_process(1);} }
|
#define assert(cond) if(cond == false){debug_trap();}
|
||||||
#else
|
#else
|
||||||
#define assert(cond)
|
# ifndef assert
|
||||||
|
# include <assert.h>
|
||||||
|
# endif
|
||||||
#endif
|
#endif
|
||||||
#pragma endregion Debug
|
#pragma endregion Debug
|
||||||
#endif
|
|
||||||
|
#define GCC_OPTIMIZATION_DISABLE _Pragma("GCC push_options") _Pragma("GCC optimize(\"O0\")")
|
||||||
|
#define GCC_OPTIMIZATION_ENABLE _Pragma("GCC pop_options")
|
||||||
|
|||||||
+14
-22
@@ -50,17 +50,13 @@
|
|||||||
#define asm_words(...) m_expand(glue(GCC_ASM_INL_, GCC_ASM_COUNT_ARGS(__VA_ARGS__))(__VA_ARGS__))
|
#define asm_words(...) m_expand(glue(GCC_ASM_INL_, GCC_ASM_COUNT_ARGS(__VA_ARGS__))(__VA_ARGS__))
|
||||||
// Very nasty macro expansion. See the Cruft pragma region after all the DSL defines
|
// Very nasty macro expansion. See the Cruft pragma region after all the DSL defines
|
||||||
|
|
||||||
/* reg_str(n) — Stringify an integer register id into the GCC asm
|
/* reg_str(n) — Stringify an integer register id into the GCC asm string form (e.g. 12 → "$12").
|
||||||
* string form (e.g. 12 → "$12"). Use this anywhere GCC's parser
|
* Use this anywhere GCC's parser expects a literal string identifying a register: clobber lists,
|
||||||
* expects a literal string identifying a register: clobber lists,
|
* asm templates, etc. The two-level macro is the standard preprocessor idiom for forcing one level of expansion before stringify —
|
||||||
* asm templates, etc. The two-level macro is the standard preprocessor
|
* without it, `#n` would stringify the macro name `R_T4` to `"R_T4"` instead of expanding `R_T4` to its value first.
|
||||||
* idiom for forcing one level of expansion before stringify — without
|
|
||||||
* it, `#n` would stringify the macro name `R_T4` to `"R_T4"` instead
|
|
||||||
* of expanding `R_T4` to its value first.
|
|
||||||
*
|
*
|
||||||
* For declaring a register variable bound to a specific GPR, use the
|
* For declaring a register variable bound to a specific GPR, use the `rgcc(n)` bundle from gcc_asm.h instead —
|
||||||
* `rgcc(n)` bundle from gcc_asm.h instead — it adds the `__asm__()`
|
* it adds the `__asm__()` qualifier around the string.
|
||||||
* qualifier around the string.
|
|
||||||
*
|
*
|
||||||
* register V3_S2* p0 __asm__(reg_str(R_T4)) = ...; // verbose
|
* register V3_S2* p0 __asm__(reg_str(R_T4)) = ...; // verbose
|
||||||
* register V3_S2* p0 rgcc(R_T4) = ...; // bundled
|
* register V3_S2* p0 rgcc(R_T4) = ...; // bundled
|
||||||
@@ -85,7 +81,7 @@
|
|||||||
* - The string "$12" is derived from it via reg_str, so they cannot drift apart.
|
* - The string "$12" is derived from it via reg_str, so they cannot drift apart.
|
||||||
* - Spelling `__asm__(reg_str(R_T4_Code))` at every call site is noise.
|
* - Spelling `__asm__(reg_str(R_T4_Code))` at every call site is noise.
|
||||||
*
|
*
|
||||||
* tmpl defined in dsl.h (the token-paste glue).
|
* tmpl defined in dsl.h (token-paste glue).
|
||||||
* rgcc define here (gcc_asm.h) because the `__asm__` keyword is GCC-specific.
|
* rgcc define here (gcc_asm.h) because the `__asm__` keyword is GCC-specific.
|
||||||
* Anyone porting to a different compiler's asm dialect overrides rgcc,
|
* Anyone porting to a different compiler's asm dialect overrides rgcc,
|
||||||
* and the integer→string derivation in rlit can be retargeted in one place.
|
* and the integer→string derivation in rlit can be retargeted in one place.
|
||||||
@@ -94,12 +90,10 @@
|
|||||||
* ------------------------------------------------------------------------ */
|
* ------------------------------------------------------------------------ */
|
||||||
#define rgcc(n) __asm__(rlit(n))
|
#define rgcc(n) __asm__(rlit(n))
|
||||||
|
|
||||||
/* rgcc_ref(n) — GCC operand-reference form "%N". Not currently used
|
/* rgcc_ref(n) — GCC operand-reference form "%N". Not currently used by the placeholder-pun macros
|
||||||
* by the placeholder-pun macros (the .word bodies are fully baked
|
* (the .word bodies are fully baked at compile time and have no runtime operand references),
|
||||||
* at compile time and have no runtime operand references), but kept
|
* but kept here for completeness in case a future asm template needs to refer to a runtime input by position.
|
||||||
* here for completeness in case a future asm template needs to refer
|
* Mirror of rgcc but produces "%N" instead of "$N". */
|
||||||
* to a runtime input by position. Mirror of rgcc but produces "%N"
|
|
||||||
* instead of "$N". */
|
|
||||||
#define rgcc_ref_(n) "%" #n
|
#define rgcc_ref_(n) "%" #n
|
||||||
#define rgcc_ref(n) rgcc_ref_(n)
|
#define rgcc_ref(n) rgcc_ref_(n)
|
||||||
|
|
||||||
@@ -147,11 +141,9 @@
|
|||||||
9, 8, 7, 6, 5, 4, 3, 2, 1, 0))
|
9, 8, 7, 6, 5, 4, 3, 2, 1, 0))
|
||||||
|
|
||||||
/* --- 2. String Concatenation Helpers --- *
|
/* --- 2. String Concatenation Helpers --- *
|
||||||
* NOTE: we use `%0`, `%1`, ... not `%c0`, `%c1`, ... because GCC's
|
* NOTE: we use `%0`, `%1`, ... not `%c0`, `%c1`, ... because GCC's asm-parser rejects `%cN` in this position with "invalid use of '%c'".
|
||||||
* asm-parser rejects `%cN` in this position with "invalid use of '%c'".
|
* The `%cN` form is for printing *character* constants; for arbitrary integer immediates (the only kind `"i"(...)` produces),
|
||||||
* The `%cN` form is for printing *character* constants; for arbitrary
|
* the plain `%N` form is the right one. Both expand to the bare immediate.
|
||||||
* integer immediates (the only kind `"i"(...)` produces), the plain
|
|
||||||
* `%N` form is the right one. Both expand to the bare immediate.
|
|
||||||
*/
|
*/
|
||||||
#define GCC_ASM_W1 "%0"
|
#define GCC_ASM_W1 "%0"
|
||||||
#define GCC_ASM_W2 GCC_ASM_W1 ", %1"
|
#define GCC_ASM_W2 GCC_ASM_W1 ", %1"
|
||||||
|
|||||||
@@ -1,9 +0,0 @@
|
|||||||
// Auto-generated by tape_atom_offset_gen.meta.lua — DO NOT EDIT
|
|
||||||
// Source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
|
||||||
#pragma once
|
|
||||||
|
|
||||||
#pragma region lottes_tape
|
|
||||||
|
|
||||||
|
|
||||||
#pragma endregion lottes_tape
|
|
||||||
|
|
||||||
@@ -0,0 +1,408 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
#pragma once
|
||||||
|
#endif
|
||||||
|
// Auto-generated by ps1_meta.lua — DO NOT EDIT
|
||||||
|
// Directory: C:\projects\Pikuma\ps1\code\duffle/
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\word_count.metadata.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\memory.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\math.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gcc_asm.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\mips.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gp.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gte.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\pad.atom.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.atom.c
|
||||||
|
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
|
||||||
|
|
||||||
|
#ifndef WORD_COUNT
|
||||||
|
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
||||||
|
#endif
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
/* ---------------------------------------------------------------------------
|
||||||
|
* MACRO ATOM Components (Reusable Assembly Components)
|
||||||
|
* These do NOT yield. They are expanded inline inside Tape Atoms.
|
||||||
|
* ---------------------------------------------------------------------------*/
|
||||||
|
// The 'Yield' sequence for Tape Atoms (mac_yield).
|
||||||
|
#define mac_yield(...) \
|
||||||
|
load_word(R_AtomJmp, R_TapePtr, 0) \
|
||||||
|
LdSlot_ \
|
||||||
|
, add_ui_self( R_TapePtr, S_(MipsCode)) \
|
||||||
|
, jump_reg( R_AtomJmp) \
|
||||||
|
, BdSlot_ nop
|
||||||
|
WORD_COUNT(mac_yield, 4)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_yield_load(...) \
|
||||||
|
load_word(R_AtomJmp, R_TapePtr, 0)
|
||||||
|
WORD_COUNT(mac_yield_load, 1)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_yield_tail(...) \
|
||||||
|
add_ui_self(R_TapePtr, S_(MipsCode)) \
|
||||||
|
, jump_reg( R_AtomJmp) \
|
||||||
|
, BdSlot_ nop
|
||||||
|
WORD_COUNT(mac_yield_tail, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_load_half_v3(tx, ty, tz, base, offset) \
|
||||||
|
load_half(tx, base, offset + OA_(U2,[0])) \
|
||||||
|
, load_half(ty, base, offset + OA_(U2,[1])) \
|
||||||
|
, load_half(tz, base, offset + OA_(U2,[2]))
|
||||||
|
WORD_COUNT(mac_load_half_v3, 3)
|
||||||
|
|
||||||
|
#define mac_load_v3s2(transfer, base, offset) \
|
||||||
|
mac_load_half_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||||
|
WORD_COUNT(mac_load_v3s2, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_load_v2s2(rs_x, rs_y, r_base, offset) \
|
||||||
|
load_half(rs_x, r_base, offset + O_(V3_S2,x)) \
|
||||||
|
, load_half(rs_y, r_base, offset + O_(V3_S2,y))
|
||||||
|
WORD_COUNT(mac_load_v2s2, 2)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_store_v2s2(rt_x, rt_y, base, offset) \
|
||||||
|
store_half(rt_x, base, offset + O_(V2_S2,x)) \
|
||||||
|
, store_half(rt_y, base, offset + O_(V2_S2,y))
|
||||||
|
WORD_COUNT(mac_store_v2s2, 2)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_load_word_v3(tx, ty, tz, base, offset) \
|
||||||
|
load_word(tx, base, offset + OA_(U4,[0])) \
|
||||||
|
, load_word(ty, base, offset + OA_(U4,[1])) \
|
||||||
|
, load_word(tz, base, offset + OA_(U4,[2]))
|
||||||
|
WORD_COUNT(mac_load_word_v3, 3)
|
||||||
|
|
||||||
|
#define mac_load_v3s4(transfer, base, offset) \
|
||||||
|
mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||||
|
WORD_COUNT(mac_load_v3s4, 3)
|
||||||
|
|
||||||
|
#define mac_load_p3s4(transfer, base, offset) \
|
||||||
|
mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||||
|
WORD_COUNT(mac_load_p3s4, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_store_half_v3(tx, ty, tz, base, offset) \
|
||||||
|
store_half(tx, base, offset + OA_(U2,[0])) \
|
||||||
|
, store_half(ty, base, offset + OA_(U2,[1])) \
|
||||||
|
, store_half(tz, base, offset + OA_(U2,[2]))
|
||||||
|
WORD_COUNT(mac_store_half_v3, 3)
|
||||||
|
|
||||||
|
#define mac_store_v3s2(transfer, base, offset) \
|
||||||
|
mac_store_half_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||||
|
WORD_COUNT(mac_store_v3s2, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_store_word_v3(tx, ty, tz, base, offset) \
|
||||||
|
store_word(tx, base, offset + OA_(U4,[0])) \
|
||||||
|
, store_word(ty, base, offset + OA_(U4,[1])) \
|
||||||
|
, store_word(tz, base, offset + OA_(U4,[2]))
|
||||||
|
WORD_COUNT(mac_store_word_v3, 3)
|
||||||
|
|
||||||
|
#define mac_store_v3s4(transfer, base, offset) \
|
||||||
|
mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||||
|
WORD_COUNT(mac_store_v3s4, 3)
|
||||||
|
|
||||||
|
#define mac_store_p3s4(transfer, base, offset) \
|
||||||
|
mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||||
|
WORD_COUNT(mac_store_p3s4, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_add_si_v3s4(rt_x, rt_y, rt_z, base, offset) \
|
||||||
|
add_si(rt_x, base, O_(V3_S4,x)) \
|
||||||
|
, add_si(rt_y, base, O_(V3_S4,y)) \
|
||||||
|
, add_si(rt_z, base, O_(V3_S4,z))
|
||||||
|
WORD_COUNT(mac_add_si_v3s4, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_sub_s_v3(dx, dy, dz, sx, sy, sz, tx, ty, tz) \
|
||||||
|
sub_s(dx, sx, tx) \
|
||||||
|
, sub_s(dy, sy, ty) \
|
||||||
|
, sub_s(dz, sz, tz)
|
||||||
|
WORD_COUNT(mac_sub_s_v3, 3)
|
||||||
|
|
||||||
|
#define mac_sub_v3s4(d, s, t) \
|
||||||
|
mac_sub_s_v3(d.x, d.y, d.z, s.x, s.y, s.z, t.x, t.y, t.z)
|
||||||
|
WORD_COUNT(mac_sub_v3s4, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_sub_s_v3_self(ds_x, ds_y, ds_z, tx, ty, tz) \
|
||||||
|
sub_s(ds_x, ds_x, tx) \
|
||||||
|
, sub_s(ds_y, ds_y, ty) \
|
||||||
|
, sub_s(ds_z, ds_z, tz)
|
||||||
|
WORD_COUNT(mac_sub_s_v3_self, 3)
|
||||||
|
|
||||||
|
#define mac_sub_v3s4_self(ds, t) \
|
||||||
|
mac_sub_s_v3_self(ds.x, ds.y, ds.z, t.x, t.y, t.z)
|
||||||
|
WORD_COUNT(mac_sub_v3s4_self, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_store_rects2(rt_x, rt_y, rt_width, rt_height, base, offset) \
|
||||||
|
store_half(rt_x, base, offset + O_(Rect_S2,x)) \
|
||||||
|
, store_half(rt_y, base, offset + O_(Rect_S2,y)) \
|
||||||
|
, store_half(rt_width, base, offset + O_(Rect_S2,width)) \
|
||||||
|
, store_half(rt_height, base, offset + O_(Rect_S2,height))
|
||||||
|
WORD_COUNT(mac_store_rects2, 4)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_load_word_imm(dst, imm) \
|
||||||
|
load_upper_i(dst, u4_hi(imm)) \
|
||||||
|
, or_i_self( dst, u4_lo(imm))
|
||||||
|
WORD_COUNT(mac_load_word_imm, 2)
|
||||||
|
|
||||||
|
#define mac_shift_aright_v3_self(dt_x, dt_y, dt_z, shift_amount) \
|
||||||
|
shift_aright(dt_x, dt_x, shift_amount) \
|
||||||
|
, shift_aright(dt_y, dt_y, shift_amount) \
|
||||||
|
, shift_aright(dt_z, dt_z, shift_amount)
|
||||||
|
WORD_COUNT(mac_shift_aright_v3_self, 3)
|
||||||
|
|
||||||
|
#define mac_shift_aright_v3s4_self(dt, shift) \
|
||||||
|
mac_shift_aright_v3_self(dt.x, dt.y, dt.z, shift)
|
||||||
|
WORD_COUNT(mac_shift_aright_v3s4_self, 3)
|
||||||
|
|
||||||
|
#define mac_shift_aright_var_v3(rd_v0, rd_v1, rd_v2, rs_v0, rs_v1, rs_v2, r_shift) \
|
||||||
|
shift_aright_var(rd_v0, rs_v0, r_shift) \
|
||||||
|
, shift_aright_var(rd_v1, rs_v1, r_shift) \
|
||||||
|
, shift_aright_var(rd_v2, rs_v2, r_shift)
|
||||||
|
WORD_COUNT(mac_shift_aright_var_v3, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_shift_aright_var_v3_self(rds_v0, rds_v1, rds_v2, r_shift) \
|
||||||
|
shift_aright_var(rds_v0, rds_v0, r_shift) \
|
||||||
|
, shift_aright_var(rds_v1, rds_v1, r_shift) \
|
||||||
|
, shift_aright_var(rds_v2, rds_v2, r_shift)
|
||||||
|
WORD_COUNT(mac_shift_aright_var_v3_self, 3)
|
||||||
|
|
||||||
|
#define mac_shift_aright_var_v3s4_self(ds, shift) \
|
||||||
|
mac_shift_aright_var_v3_self(ds.x, ds.y, ds.z, shift)
|
||||||
|
WORD_COUNT(mac_shift_aright_var_v3s4_self, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_load_tri_indices(r_face_cusor, r_i0, r_i1, r_i2) \
|
||||||
|
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)) \
|
||||||
|
, load_half_u(r_i1, r_face_cusor, 1 * S_(S2)) \
|
||||||
|
, load_half_u(r_i2, r_face_cusor, 2 * S_(S2))
|
||||||
|
WORD_COUNT(mac_load_tri_indices, 3)
|
||||||
|
|
||||||
|
#define mac_gte_mv_to_cr_diag_v3s4(v) \
|
||||||
|
gte_mv_to_ctrl_r(v.y, gte_cr_RT13) \
|
||||||
|
, gte_mv_to_ctrl_r(v.z, gte_cr_RT22) \
|
||||||
|
, gte_mv_to_ctrl_r(v.x, gte_cr_RT11)
|
||||||
|
WORD_COUNT(mac_gte_mv_to_cr_diag_v3s4, 3)
|
||||||
|
|
||||||
|
#define mac_gte_ld_ir123_v3s4(v) \
|
||||||
|
gte_mv_to_data_r(v.x, C2_IR1) \
|
||||||
|
, gte_mv_to_data_r(v.y, C2_IR2) \
|
||||||
|
, gte_mv_to_data_r(v.z, C2_IR3)
|
||||||
|
WORD_COUNT(mac_gte_ld_ir123_v3s4, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_gte_op_cross_v3s4(a, b) \
|
||||||
|
mac_gte_mv_to_cr_diag_v3s4(a) \
|
||||||
|
GteDelay_ /* RT diagonal: D1 = a.x, D2 = a.y, D3 = a.z */ \
|
||||||
|
, mac_gte_ld_ir123_v3s4(b) \
|
||||||
|
GteDelay_ /* IR: second operand (b.xyz) */ \
|
||||||
|
, gte_cmdw_cross /* OP: MAC1/2/3 = a × b (S12.20) */ \
|
||||||
|
, mac_gte_mv_from_mac123_v3s4(a) \
|
||||||
|
GteDelay_ /* Read MAC1/2/3 → a.xyz (overwrites source-A's load targets) */ \
|
||||||
|
, mac_shift_aright_v3s4_self(a, 12) /* Right-shift MAC by 12 (S12.20 → S12.0 OuterProduct12) */
|
||||||
|
WORD_COUNT(mac_gte_op_cross_v3s4, 13)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_gte_store_f3(r_primitive_cursor) \
|
||||||
|
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)) \
|
||||||
|
, gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_F3,p1)) \
|
||||||
|
, gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_F3,p2))
|
||||||
|
WORD_COUNT(mac_gte_store_f3, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_gte_load_tri_verts(vbase, v0, v1, v2) \
|
||||||
|
shift_lleft(R_AT, v0, v3s2_byteoff) \
|
||||||
|
, add_u_self(R_AT, vbase) \
|
||||||
|
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||||
|
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||||
|
, LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY0) \
|
||||||
|
, gte_mv_to_data_r(R_V1, C2_VZ0) \
|
||||||
|
, shift_lleft(R_AT, v1, v3s2_byteoff) \
|
||||||
|
, add_u_self(R_AT, vbase) \
|
||||||
|
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||||
|
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||||
|
, LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY1) \
|
||||||
|
, gte_mv_to_data_r(R_V1, C2_VZ1) \
|
||||||
|
, shift_lleft(R_AT, v2, v3s2_byteoff) \
|
||||||
|
, add_u_self(R_AT, vbase) \
|
||||||
|
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||||
|
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||||
|
, LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY2) \
|
||||||
|
, gte_mv_to_data_r(R_V1, C2_VZ2)
|
||||||
|
WORD_COUNT(mac_gte_load_tri_verts, 18)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_gte_store_g4_p012(r_primitive_cursor) \
|
||||||
|
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)) \
|
||||||
|
, gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)) \
|
||||||
|
, gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2))
|
||||||
|
WORD_COUNT(mac_gte_store_g4_p012, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_gte_store_g4_p3(r_primitive_cursor) \
|
||||||
|
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3))
|
||||||
|
WORD_COUNT(mac_gte_store_g4_p3, 1)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_gte_sqr_v3(r_sx, r_sy, r_sz, r_sq_x, r_sq_y, r_sq_z) \
|
||||||
|
mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop) \
|
||||||
|
, gte_mv_from_data_r(r_sq_x, C2_MAC1) \
|
||||||
|
, gte_mv_from_data_r(r_sq_y, C2_MAC2) \
|
||||||
|
, gte_mv_from_data_r(r_sq_z, C2_MAC3)
|
||||||
|
WORD_COUNT(mac_gte_sqr_v3, 8)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, delay_slot) \
|
||||||
|
gte_mv_to_data_r(r_sx, C2_IR1) \
|
||||||
|
, gte_mv_to_data_r(r_sy, C2_IR2) \
|
||||||
|
, gte_mv_to_data_r(r_sz, C2_IR3) \
|
||||||
|
, delay_slot \
|
||||||
|
, gte_cmdw_sqr
|
||||||
|
WORD_COUNT(mac_gte_sqr_v3s4, 5)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_gte_gpf_scale(r_sx, r_sy, r_sz, r_recip_est, r_shift, r_dx, r_dy, r_dz) \
|
||||||
|
gte_mv_to_data_r(r_recip_est, C2_IR0) \
|
||||||
|
, gte_mv_to_data_r(r_sx, C2_IR1) \
|
||||||
|
, gte_mv_to_data_r(r_sy, C2_IR2) \
|
||||||
|
, gte_mv_to_data_r(r_sz, C2_IR3) \
|
||||||
|
, GteDelay_ nop2 /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */ \
|
||||||
|
, gte_cmdw_gpf \
|
||||||
|
, gte_mv_from_data_r(r_dx, C2_MAC1) \
|
||||||
|
, gte_mv_from_data_r(r_dy, C2_MAC2) \
|
||||||
|
, gte_mv_from_data_r(r_dz, C2_MAC3) \
|
||||||
|
, shift_aright_var(r_dx, r_dx, r_shift) \
|
||||||
|
, shift_aright_var(r_dy, r_dy, r_shift) \
|
||||||
|
, shift_aright_var(r_dz, r_dz, r_shift)
|
||||||
|
WORD_COUNT(mac_gte_gpf_scale, 13)
|
||||||
|
|
||||||
|
#define mac_trans_mt3s3s4(r_mtx, r_off, r_t0, r_t1, r_t2) \
|
||||||
|
load_word( r_t0, r_off, O_(V3_S4,x)) \
|
||||||
|
, load_word( r_t1, r_off, O_(V3_S4,y)) \
|
||||||
|
, load_word( r_t2, r_off, O_(V3_S4,z)) \
|
||||||
|
, store_word(r_t0, r_mtx, O_(MT3_S2S4,t[0])) \
|
||||||
|
, store_word(r_t1, r_mtx, O_(MT3_S2S4,t[1])) \
|
||||||
|
, store_word(r_t2, r_mtx, O_(MT3_S2S4,t[2]))
|
||||||
|
WORD_COUNT(mac_trans_mt3s3s4, 6)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_lzcr_round_even_half_shift(r_shift, r_mag_sq, r_mag_sq_copy) \
|
||||||
|
and_i(r_shift, r_shift, gte_lzcr_even_mask) \
|
||||||
|
, or_u(r_mag_sq_copy, r_mag_sq, 0) \
|
||||||
|
, li_s( r_mag_sq, 31) \
|
||||||
|
, sub_s( r_mag_sq, r_mag_sq, r_shift) \
|
||||||
|
, shift_aright(r_mag_sq, r_mag_sq, 1)
|
||||||
|
WORD_COUNT(mac_lzcr_round_even_half_shift, 5)
|
||||||
|
|
||||||
|
#define mac_gte_general_purpose_interopolation(to_ir0, to_ir1, to_ir2, to_ir3, fr_mac1, fr_mac2, fr_mac3, nop_slot1, nop_slot2) \
|
||||||
|
gte_mv_to_data_r(to_ir0, C2_IR0) \
|
||||||
|
, gte_mv_to_data_r(to_ir1, C2_IR1) /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */ \
|
||||||
|
, gte_mv_to_data_r(to_ir2, C2_IR2) \
|
||||||
|
, gte_mv_to_data_r(to_ir3, C2_IR3) /* IR3 = src.z (reloaded) */ \
|
||||||
|
, GteDelay_ nop_slot1 \
|
||||||
|
, GteDelay_ nop_slot2 \
|
||||||
|
, gte_cmdw_gpf \
|
||||||
|
, gte_mv_from_data_r(fr_mac1, C2_MAC1) \
|
||||||
|
, gte_mv_from_data_r(fr_mac2, C2_MAC2) \
|
||||||
|
, gte_mv_from_data_r(fr_mac3, C2_MAC3)
|
||||||
|
WORD_COUNT(mac_gte_general_purpose_interopolation, 10)
|
||||||
|
|
||||||
|
#define mac_gte_mv_from_data_r_mac123(fr_mac1, fr_mac2, fr_mac3) \
|
||||||
|
gte_mv_from_data_r(fr_mac1, C2_MAC1) \
|
||||||
|
, gte_mv_from_data_r(fr_mac2, C2_MAC2) \
|
||||||
|
, gte_mv_from_data_r(fr_mac3, C2_MAC3)
|
||||||
|
WORD_COUNT(mac_gte_mv_from_data_r_mac123, 3)
|
||||||
|
|
||||||
|
#define mac_gte_mv_from_mac123_v3s4(v) \
|
||||||
|
mac_gte_mv_from_data_r_mac123(v.x, v.y, v.z)
|
||||||
|
WORD_COUNT(mac_gte_mv_from_mac123_v3s4, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \
|
||||||
|
mac_load_word_imm(reg_transfer, cmd) \
|
||||||
|
, store_word( reg_transfer, reg_base, port)
|
||||||
|
WORD_COUNT(mac_gcmd_push, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_store_rgb8(rr, rg, rb, base, offset) \
|
||||||
|
store_byte(rr, base, offset + O_(RGB8,r)) \
|
||||||
|
, store_byte(rg, base, offset + O_(RGB8,g)) \
|
||||||
|
, store_byte(rb, base, offset + O_(RGB8,b))
|
||||||
|
WORD_COUNT(mac_store_rgb8, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_pack_color_word(r_base, off, cmd, r, g, b) \
|
||||||
|
load_upper_i(R_AT, (cmd) << 8 | (b)) \
|
||||||
|
, or_i_self( R_AT, ((g) << 8) | (r)) \
|
||||||
|
, store_word( R_AT, r_base, (off))
|
||||||
|
WORD_COUNT(mac_pack_color_word, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_format_f3_color(r_base, r, g, b) \
|
||||||
|
mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b)
|
||||||
|
WORD_COUNT(mac_format_f3_color, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_format_g4_color(r_prim_cursor, r0, g0, b0, r1, g1, b1, r2, g2, b2, r3, g3, b3) \
|
||||||
|
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0) \
|
||||||
|
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1) \
|
||||||
|
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c2), 0, r2,g2,b2) \
|
||||||
|
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3)
|
||||||
|
WORD_COUNT(mac_format_g4_color, 12)
|
||||||
|
|
||||||
|
#define mac_insert_ot_tag(r_ot_base, r_prim_cursor, poly_size) \
|
||||||
|
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
|
||||||
|
, add_u_self( R_T1, r_ot_base) /* T1 = & OrderingTable[OTZ] */ \
|
||||||
|
, load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \
|
||||||
|
, load_upper_i(R_V0, (poly_size/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) \
|
||||||
|
, mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
|
||||||
|
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
|
||||||
|
, store_word( R_AT, r_prim_cursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \
|
||||||
|
, shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
|
||||||
|
, shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \
|
||||||
|
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
|
||||||
|
WORD_COUNT(mac_insert_ot_tag, 11)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_pad_set_centered_axes(state, scratch) \
|
||||||
|
load_upper_i(scratch, (PadAxis_Centered >> 16) & 0xFFFF) \
|
||||||
|
, or_i_self( scratch, PadAxis_Centered & 0xFFFF) /* mac_load_word_imm(scratch, PadAxis_Centered), */ \
|
||||||
|
, store_word( scratch, state, O_(PadState,axes))
|
||||||
|
WORD_COUNT(mac_pad_set_centered_axes, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_pad_set_id_byte(state, r_id, id_value) \
|
||||||
|
add_ui( r_id, R_0, id_value) \
|
||||||
|
, store_byte(r_id, state, O_(PadState,id))
|
||||||
|
WORD_COUNT(mac_pad_set_id_byte, 2)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_pad_set_status(r_tmp, r_state, pad_status) \
|
||||||
|
add_ui( r_tmp, R_0, pad_status) \
|
||||||
|
, store_word(r_tmp, r_state, O_(PadState,status))
|
||||||
|
WORD_COUNT(mac_pad_set_status, 2)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_pad_store_inverted_buttons(r_buttons, r_pad_state) \
|
||||||
|
nor_u( r_buttons, r_buttons, R_0) \
|
||||||
|
, store_half(r_buttons, r_pad_state, O_(PadState,buttons))
|
||||||
|
WORD_COUNT(mac_pad_store_inverted_buttons, 2)
|
||||||
|
|
||||||
@@ -0,0 +1,73 @@
|
|||||||
|
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
|
||||||
|
// Directory: C:\projects\Pikuma\ps1\code\duffle\
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\word_count.metadata.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\memory.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\math.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gcc_asm.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\mips.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gp.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gte.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\pad.atom.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.atom.c
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#pragma region duffle
|
||||||
|
|
||||||
|
|
||||||
|
// --- atom: example_atom_proc (10 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_example_atom_proc_skip 2
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_example_atom_proc_skip = _atom_offset_example_atom_proc_skip,
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- atom: normalize_v3s4 (62 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_aligned_done_srav_path 3
|
||||||
|
#define _atom_offset_srav_path_aligned_done 4
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_aligned_done_srav_path = _atom_offset_aligned_done_srav_path,
|
||||||
|
atom_offset_srav_path_aligned_done = _atom_offset_srav_path_aligned_done,
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- atom: pad_bios_snapshot (84 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_snap_root_skip_disconnected 10
|
||||||
|
#define _atom_offset_disconnected_snap_end 65
|
||||||
|
#define _atom_offset_case_2_id_dispatch 9
|
||||||
|
#define _atom_offset_pending_snap_end 54
|
||||||
|
#define _atom_offset_id_dispatch_try_analog_stick 12
|
||||||
|
#define _atom_offset_id_dispatch_snap_end 40
|
||||||
|
#define _atom_offset_try_analog_stick_try_analog_pad 13
|
||||||
|
#define _atom_offset_analog_stick_snap_end 25
|
||||||
|
#define _atom_offset_try_analog_pad_try_unsupported 12
|
||||||
|
#define _atom_offset_analog_pad_snap_end 10
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_snap_root_skip_disconnected = _atom_offset_snap_root_skip_disconnected,
|
||||||
|
atom_offset_disconnected_snap_end = _atom_offset_disconnected_snap_end,
|
||||||
|
atom_offset_case_2_id_dispatch = _atom_offset_case_2_id_dispatch,
|
||||||
|
atom_offset_pending_snap_end = _atom_offset_pending_snap_end,
|
||||||
|
atom_offset_id_dispatch_try_analog_stick = _atom_offset_id_dispatch_try_analog_stick,
|
||||||
|
atom_offset_id_dispatch_snap_end = _atom_offset_id_dispatch_snap_end,
|
||||||
|
atom_offset_try_analog_stick_try_analog_pad = _atom_offset_try_analog_stick_try_analog_pad,
|
||||||
|
atom_offset_analog_stick_snap_end = _atom_offset_analog_stick_snap_end,
|
||||||
|
atom_offset_try_analog_pad_try_unsupported = _atom_offset_try_analog_pad_try_unsupported,
|
||||||
|
atom_offset_analog_pad_snap_end = _atom_offset_analog_pad_snap_end,
|
||||||
|
};
|
||||||
|
|
||||||
|
#pragma endregion duffle
|
||||||
|
|
||||||
@@ -0,0 +1,61 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# include "dsl.h"
|
||||||
|
# include "gp.h"
|
||||||
|
# include "lottes_tape.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
ATOM_FILE_DEBUGGER_LINE_MARKER(gp_atom_c);
|
||||||
|
|
||||||
|
#pragma region MACs (Mips Atom Components)
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_gcmd_push(AtomBuilder_R ab, U4 cmd, U4 reg_transfer, U4 reg_base, U2 port)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
mac_load_word_imm(reg_transfer, cmd),
|
||||||
|
store_word( reg_transfer, reg_base, port),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_store_rgb8(AtomBuilder_R ab, U1 rr, U1 rg, U1 rb, U4 base, U4 offset)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
store_byte(rr, base, offset + O_(RGB8,r)),
|
||||||
|
store_byte(rg, base, offset + O_(RGB8,g)),
|
||||||
|
store_byte(rb, base, offset + O_(RGB8,b)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_pack_color_word(AtomBuilder_R ab, U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
load_upper_i(R_AT, (cmd) << 8 | (b)),
|
||||||
|
or_i_self( R_AT, ((g) << 8) | (r)),
|
||||||
|
store_word( R_AT, r_base, (off)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_format_f3_color(AtomBuilder_R ab, U4 r_base, U1 r, U1 g, U1 b)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, { mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_format_g4_color(AtomBuilder_R ab, U4 r_prim_cursor,
|
||||||
|
U1 r0, U1 g0, U1 b0,
|
||||||
|
U1 r1, U1 g1, U1 b1,
|
||||||
|
U1 r2, U1 g2, U1 b2,
|
||||||
|
U1 r3, U1 g3, U1 b3)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0),
|
||||||
|
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1),
|
||||||
|
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c2), 0, r2,g2,b2),
|
||||||
|
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list. */
|
||||||
|
// TODO(Ed): Expose R_T1 as a r_t0, r_V0 as r_t2
|
||||||
|
I_ Slice_MipsCode ac_insert_ot_tag(AtomBuilder_R ab, Reg r_ot_base, Reg r_prim_cursor, U2 poly_size) MipsAtomComp_Proc_(ab, {
|
||||||
|
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
|
||||||
|
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
|
||||||
|
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
|
||||||
|
load_upper_i(R_V0, (poly_size/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits),
|
||||||
|
mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
|
||||||
|
or_u( R_AT, R_AT, R_V0), // Merge length
|
||||||
|
store_word( R_AT, r_prim_cursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr)
|
||||||
|
shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)), // AT = (prim_length << 24) | old_addr
|
||||||
|
shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
|
||||||
|
store_word( R_AT, R_T1, O_(PolyTag,code)), // OrderingTable[OTZ] = PrimCursor
|
||||||
|
})
|
||||||
|
|
||||||
|
#pragma endregion MACs (Mips Atom Components)
|
||||||
+414
-314
@@ -1,7 +1,6 @@
|
|||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
* duffle DSL Suffix Conventions
|
* duffle DSL Suffix Conventions
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
*
|
|
||||||
* Every mnemonic in this header follows the same suffix grammar:
|
* Every mnemonic in this header follows the same suffix grammar:
|
||||||
*
|
*
|
||||||
* Primitive commands: gp0_cmd_poly_f3 = 0x20 (byte opcode)
|
* Primitive commands: gp0_cmd_poly_f3 = 0x20 (byte opcode)
|
||||||
@@ -22,12 +21,11 @@
|
|||||||
* 4. Semantic encoders gp0_word_poly_f3(r,g,b)
|
* 4. Semantic encoders gp0_word_poly_f3(r,g,b)
|
||||||
* 3. Composite encoders enc_color_word(cmd, r, g, b)
|
* 3. Composite encoders enc_color_word(cmd, r, g, b)
|
||||||
* 2. Per-field encoders enc_gp0_color_r(r), enc_gp0_color_g(g), ...
|
* 2. Per-field encoders enc_gp0_color_r(r), enc_gp0_color_g(g), ...
|
||||||
* 1. Bitfield layout consts gp0_color_red_shift = 0, gp0_color_red_mask = 0xFF
|
* 1. Bitfield layout consts gp0_color_red_pos = 0, gp0_color_red_width = 8
|
||||||
* 0. Opcode IDs gp0_cmd_poly_f3 = 0x20
|
* 0. Opcode IDs gp0_cmd_poly_f3 = 0x20
|
||||||
*
|
*
|
||||||
* Vendor mnemonics (gte_mtc2, gte_mfc2, etc.) are NOT in this header.
|
* Vendor mnemonics (gte_mtc2, gte_mfc2, etc.) are NOT in this header.
|
||||||
* They live in the opt-in `gp_vendor_sym.h` for users who prefer the
|
* They live in the opt-in `gp_vendor_sym.h` for users who prefer the PSYQ-style names.
|
||||||
* PSYQ-style names.
|
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
|
|
||||||
#ifdef INTELLISENSE_DIRECTIVES
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
@@ -41,14 +39,28 @@
|
|||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
* Hardware MMIO Addresses
|
* Hardware MMIO Addresses
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
*
|
* PSX GPU has two 32-bit ports in the I/O register region at KSEG2 0x1F800000+.
|
||||||
* PSX GPU has two 32-bit ports in the I/O register region at KSEG2
|
* GP0 (offset 0x10) is the data port (commands + params).
|
||||||
* 0x1F800000+. GP0 (offset 0x10) is the data port (commands + params).
|
|
||||||
* GP1 (offset 0x14) is the control port (status, ctrl writes).
|
* GP1 (offset 0x14) is the control port (status, ctrl writes).
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
|
/* IO base address (KSEG2 0x1F800000+ for the I/O register region).
|
||||||
|
* The 16-bit upper half `IO_BASE_ADDR_HI16` is the form used by tape-side macros that pin a register
|
||||||
|
* to hold the IO base and access ports via offsets:
|
||||||
|
* `lui $reg, 0x1F80` (1 word) then `sw $data, GPIO_PORT*_OFFSET($reg)` (1 word).
|
||||||
|
* Mirrors the `IO_BASE_ADDR equ 0x1F80` + `gpio_port0 equ 0x1810` pattern from graphics_hello/gp.s. */
|
||||||
enum {
|
enum {
|
||||||
HW_GP0_ADDR = 0x1F801810, /* GPU data port (commands + parameters) */
|
IO_BASE_ADDR = 0x1F800000, /* full 32-bit I/O region base */
|
||||||
HW_GP1_ADDR = 0x1F801814, /* GPU control port (status, ctrl writes) */
|
IO_BASE_ADDR_HI16 = 0x1F80, /* fits in a single `lui $reg, 0x1F80` */
|
||||||
|
|
||||||
|
/* Offsets from IO_BASE_ADDR to each port. Used by tape-side macros
|
||||||
|
* that pin a register to IO_BASE_ADDR and access ports via offsets:
|
||||||
|
* sw $data, GPIO_PORT0_OFFSET($io_base) ; write GP0
|
||||||
|
* sw $data, GPIO_PORT1_OFFSET($io_base) ; write GP1 */
|
||||||
|
GPIO_PORT0_OFFSET = 0x1810,
|
||||||
|
GPIO_PORT1_OFFSET = 0x1814,
|
||||||
|
|
||||||
|
HW_GP0_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT0_OFFSET,
|
||||||
|
HW_GP1_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT1_OFFSET,
|
||||||
};
|
};
|
||||||
|
|
||||||
#define HW_GP0 C_(U4 V_*, HW_GP0_ADDR)
|
#define HW_GP0 C_(U4 V_*, HW_GP0_ADDR)
|
||||||
@@ -56,89 +68,87 @@ enum {
|
|||||||
|
|
||||||
#define gp0_send(word) (HW_GP0[0] = (word))
|
#define gp0_send(word) (HW_GP0[0] = (word))
|
||||||
#define gp1_send(word) (HW_GP1[0] = (word))
|
#define gp1_send(word) (HW_GP1[0] = (word))
|
||||||
|
#define DmaSlot_ // Annotate an instruction as filling a CPU <-> Command DMA delay slot/s
|
||||||
|
|
||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
* GP0 command byte constants + Layer 1 (GPU bitfield shifts)
|
* GP0 command byte constants + Layer 1 (GPU bitfield shifts)
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
*
|
* 8-bit GP0 opcodes (the upper byte of a primitive's first word). These are the BYTE only.
|
||||||
* 8-bit GP0 opcodes (the upper byte of a primitive's first word).
|
* NO macro body past this point uses a raw shift or raw mask.
|
||||||
* These are the BYTE only; pre-baked 32-bit words are in §10.4.
|
* Mirrors the OPCODE_POS / RS_POS convention from mips.h.
|
||||||
* The layer-1 bitfield-layout constants live in the same enum block
|
|
||||||
* so the encoder in §10.4 can reference them by name. NO macro body
|
|
||||||
* past this point uses a raw shift or raw mask — every shift/width/mask
|
|
||||||
* is named here, named once. Mirrors the OPCODE_SHIFT / RS_SHIFT /
|
|
||||||
* REG_MASK convention from mips.h lines 276-293.
|
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
enum {
|
enum {
|
||||||
gp0_cmd_Nop = 0x00,
|
gp0_cmd_Nop = 0x00,
|
||||||
|
|
||||||
/* Cache management */
|
/* Cache management */
|
||||||
gp0_cmd_ClearCache = 0x01,
|
gp0_cmd_ClearCache = 0x01,
|
||||||
gp0_cmd_FillVram = 0x02,
|
gp0_cmd_FillVram = 0x02,
|
||||||
gp0_cmd_CopyVram = 0x80,
|
gp0_cmd_CopyVram = 0x80,
|
||||||
gp0_cmd_CopyVramChained = 0x81,
|
gp0_cmd_CopyVramChained = 0x81,
|
||||||
gp0_cmd_ReadVram = 0xC0,
|
gp0_cmd_ReadVram = 0xC0,
|
||||||
|
|
||||||
/* Polygons */
|
/* Polygons */
|
||||||
gp0_cmd_poly_f3 = 0x20, /* Flat Triangle */
|
gp0_cmd_poly_f3 = 0x20, /* Flat Triangle */
|
||||||
gp0_cmd_poly_ft3 = 0x24, /* Flat Textured Triangle */
|
gp0_cmd_poly_ft3 = 0x24, /* Flat Textured Triangle */
|
||||||
gp0_cmd_poly_g3 = 0x30, /* Gouraud Triangle */
|
gp0_cmd_poly_g3 = 0x30, /* Gouraud Triangle */
|
||||||
gp0_cmd_poly_gt3 = 0x34, /* Gouraud Textured Tri */
|
gp0_cmd_poly_gt3 = 0x34, /* Gouraud Textured Tri */
|
||||||
gp0_cmd_poly_f4 = 0x28, /* Flat Quad */
|
gp0_cmd_poly_f4 = 0x28, /* Flat Quad */
|
||||||
gp0_cmd_poly_ft4 = 0x2C, /* Flat Textured Quad */
|
gp0_cmd_poly_ft4 = 0x2C, /* Flat Textured Quad */
|
||||||
gp0_cmd_poly_g4 = 0x38, /* Gouraud Quad */
|
gp0_cmd_poly_g4 = 0x38, /* Gouraud Quad */
|
||||||
gp0_cmd_poly_gt4 = 0x3C, /* Gouraud Textured Quad */
|
gp0_cmd_poly_gt4 = 0x3C, /* Gouraud Textured Quad */
|
||||||
|
|
||||||
/* Lines */
|
/* Lines */
|
||||||
gp0_cmd_line_f2 = 0x40,
|
gp0_cmd_line_f2 = 0x40,
|
||||||
gp0_cmd_line_g2 = 0x50,
|
gp0_cmd_line_g2 = 0x50,
|
||||||
|
|
||||||
/* Sprites + Tiles + Rects */
|
/* Sprites + Tiles + Rects */
|
||||||
gp0_cmd_sprt_1 = 0x64,
|
gp0_cmd_sprt_1 = 0x64,
|
||||||
gp0_cmd_sprt_8 = 0x74,
|
gp0_cmd_sprt_8 = 0x74,
|
||||||
gp0_cmd_sprt_16 = 0x7C,
|
gp0_cmd_sprt_16 = 0x7C,
|
||||||
gp0_cmd_tile_1 = 0x60,
|
gp0_cmd_tile_1 = 0x60,
|
||||||
gp0_cmd_tile_8 = 0x68,
|
gp0_cmd_tile_8 = 0x68,
|
||||||
gp0_cmd_tile_16 = 0x70,
|
gp0_cmd_tile_16 = 0x70,
|
||||||
|
|
||||||
/* bitfield shifts / widths / masks ----
|
/* State setters (not drawing primitives; set render context). */
|
||||||
*
|
gp0_cmd_DrawModeSetting = 0xE1, /* TPage / draw-mode (semi-trans, dither, etc.) */
|
||||||
* Generic GP0/GP1 command byte (upper 8 bits of every word sent
|
gp0_cmd_SetTextureWindow = 0xE2,
|
||||||
* to either port). Used by `enc_gp0_cmd(cmd)` and friends below. */
|
gp0_cmd_SetDrawArea_TopLeft = 0xE3,
|
||||||
gp0_cmd_shift = 24,
|
gp0_cmd_SetDrawArea_BotRight = 0xE4,
|
||||||
gp0_cmd_width = 8,
|
gp0_cmd_SetDrawOffset = 0xE5,
|
||||||
gp0_cmd_mask = 0xFF,
|
gp0_cmd_SetMaskBit = 0xE6,
|
||||||
|
|
||||||
/* Color word layout (lives in Poly_F3.color, Poly_G4.c0..c3, etc.):
|
/* bitfield offset pos / widths ----
|
||||||
* bits 31..24 = command byte
|
* Generic GP0/GP1 command byte (upper 8 bits of every word sent to either port). */
|
||||||
* bits 23..16 = BLUE
|
gp0_cmd_pos = 24,
|
||||||
* bits 15..08 = GREEN
|
gp0_cmd_width = 8,
|
||||||
* bits 07..00 = RED (PSX GPU is BGR, NOT RGB) */
|
|
||||||
gp0_color_cmd_shift = 24, gp0_color_cmd_width = 8, gp0_color_cmd_mask = 0xFF,
|
/* Color word layout (lives in Poly_F3.color, Poly_G4.c0..c3, etc.):
|
||||||
gp0_color_blue_shift = 16, gp0_color_blue_width = 8, gp0_color_blue_mask = 0xFF,
|
* bits 31..24 = command byte
|
||||||
gp0_color_green_shift = 8, gp0_color_green_width = 8, gp0_color_green_mask = 0xFF,
|
* bits 23..16 = BLUE
|
||||||
gp0_color_red_shift = 0, gp0_color_red_width = 8, gp0_color_red_mask = 0xFF,
|
* bits 15..08 = GREEN
|
||||||
|
* bits 07..00 = RED (PSX GPU is BGR, NOT RGB) */
|
||||||
|
gp0_color_cmd_pos = 24, gp0_color_cmd_width = 8,
|
||||||
|
gp0_color_blue_pos = 16, gp0_color_blue_width = 8,
|
||||||
|
gp0_color_green_pos = 8, gp0_color_green_width = 8,
|
||||||
|
gp0_color_red_pos = 0, gp0_color_red_width = 8,
|
||||||
};
|
};
|
||||||
|
|
||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
* Layer 1.5 (per-field encoders) + Layer 2 (composite) + Layer 3 (semantic GP0 word builders)
|
* Layer 1.5 (per-field encoders) + Layer 2 (composite) + Layer 3 (semantic GP0 word builders)
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
*
|
* Layer 1.5 encoders take one field's value, mask it to its own width, and shift it to its own position.
|
||||||
* Layer 1.5 encoders take one field's value, mask it to its own width,
|
* Mirrors `enc_op` / `enc_rs` / `enc_rt` in mips.h and `enc_gte_sf` / `enc_gte_mx` in gte.h.
|
||||||
* and shift it to its own position. Mirrors `enc_op` / `enc_rs` /
|
* Layer-2 composite encoders OR the per-field encoders together; layer-3 semantic macros delegate to the composites.
|
||||||
* `enc_rt` in mips.h lines 295-301 and `enc_gte_sf` / `enc_gte_mx` in
|
|
||||||
* gte.h lines 342-347. Layer-2 composite encoders OR the per-field
|
|
||||||
* encoders together; layer-3 semantic macros delegate to the composites.
|
|
||||||
* No raw shifts or magic numbers in any macro body below this point.
|
* No raw shifts or magic numbers in any macro body below this point.
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
|
|
||||||
/* ---- Layer 1.5: per-field encoders ---- */
|
/* ---- Layer 1.5: per-field encoders ---- */
|
||||||
#define enc_gp0_cmd(cmd) (((cmd) & gp0_cmd_mask) << gp0_cmd_shift)
|
#define enc_gp0_cmd(cmd) ((cmd) << gp0_cmd_pos)
|
||||||
|
|
||||||
#define enc_gp0_color_cmd(cmd) (((cmd) & gp0_color_cmd_mask) << gp0_color_cmd_shift)
|
#define enc_gp0_color_cmd(cmd) ((cmd) << gp0_color_cmd_pos)
|
||||||
#define enc_gp0_color_r(r) (((r) & gp0_color_red_mask) << gp0_color_red_shift)
|
#define enc_gp0_color_r(r) ((r) << gp0_color_red_pos)
|
||||||
#define enc_gp0_color_g(g) (((g) & gp0_color_green_mask) << gp0_color_green_shift)
|
#define enc_gp0_color_g(g) ((g) << gp0_color_green_pos)
|
||||||
#define enc_gp0_color_b(b) (((b) & gp0_color_blue_mask) << gp0_color_blue_shift)
|
#define enc_gp0_color_b(b) ((b) << gp0_color_blue_pos)
|
||||||
|
|
||||||
/* ---- Layer 2: composite encoders ---- */
|
/* ---- Layer 2: composite encoders ---- */
|
||||||
#define enc_color_word(cmd, r, g, b) (enc_gp0_color_cmd(cmd) | enc_gp0_color_r(r) | enc_gp0_color_g(g) | enc_gp0_color_b(b))
|
#define enc_color_word(cmd, r, g, b) (enc_gp0_color_cmd(cmd) | enc_gp0_color_r(r) | enc_gp0_color_g(g) | enc_gp0_color_b(b))
|
||||||
@@ -148,7 +158,7 @@ enum {
|
|||||||
/* ---- Layer 3: semantic GP0 word builders ---- */
|
/* ---- Layer 3: semantic GP0 word builders ---- */
|
||||||
|
|
||||||
/* Pre-baked color+command words for all 8 polygon variants.
|
/* Pre-baked color+command words for all 8 polygon variants.
|
||||||
* Mirrors `load_word` / `add_ui` / `jump_reg` style in mips.h lines 340-388. */
|
* Mirrors `load_word` / `add_ui` / `jump_reg` style in mips.h. */
|
||||||
#define gp0_word_poly_f3(r,g,b) enc_color_word(gp0_cmd_poly_f3, (r),(g),(b))
|
#define gp0_word_poly_f3(r,g,b) enc_color_word(gp0_cmd_poly_f3, (r),(g),(b))
|
||||||
#define gp0_word_poly_ft3(r,g,b) enc_color_word(gp0_cmd_poly_ft3, (r),(g),(b))
|
#define gp0_word_poly_ft3(r,g,b) enc_color_word(gp0_cmd_poly_ft3, (r),(g),(b))
|
||||||
#define gp0_word_poly_g3(r,g,b) enc_color_word(gp0_cmd_poly_g3, (r),(g),(b))
|
#define gp0_word_poly_g3(r,g,b) enc_color_word(gp0_cmd_poly_g3, (r),(g),(b))
|
||||||
@@ -159,89 +169,97 @@ enum {
|
|||||||
#define gp0_word_poly_gt4(r,g,b) enc_color_word(gp0_cmd_poly_gt4, (r),(g),(b))
|
#define gp0_word_poly_gt4(r,g,b) enc_color_word(gp0_cmd_poly_gt4, (r),(g),(b))
|
||||||
|
|
||||||
/* Cache management — bare-cmd words (no color/range payload). */
|
/* Cache management — bare-cmd words (no color/range payload). */
|
||||||
#define gp0_word_clear_cache() enc_gp0_cmd_word(gp0_cmd_ClearCache)
|
#define gp0_word_clear_cache() enc_gp0_cmd_word(gp0_cmd_ClearCache)
|
||||||
#define gp0_word_fill_vram() enc_gp0_cmd_word(gp0_cmd_FillVram)
|
#define gp0_word_fill_vram() enc_gp0_cmd_word(gp0_cmd_FillVram)
|
||||||
#define gp0_word_copy_vram() enc_gp0_cmd_word(gp0_cmd_CopyVram)
|
#define gp0_word_copy_vram() enc_gp0_cmd_word(gp0_cmd_CopyVram)
|
||||||
#define gp0_word_read_vram() enc_gp0_cmd_word(gp0_cmd_ReadVram)
|
#define gp0_word_read_vram() enc_gp0_cmd_word(gp0_cmd_ReadVram)
|
||||||
|
|
||||||
|
/* NOP — bare-cmd word (no effect; used as DR_ENV padding). */
|
||||||
|
#define gp0_word_nop() enc_gp0_cmd_word(gp0_cmd_Nop)
|
||||||
|
|
||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
* GP1 command byte constants + Layer 1 (display-mode + range + draw-area bitfield shifts)
|
* GP1 command byte constants + Layer 1 (display-mode + range + draw-area bitfield shifts)
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
*
|
* GP1 status bits are read from HW_GP1;
|
||||||
* GP1 status bits are read from HW_GP1; ctrl writes use GP1 commands
|
* ctrl writes use GP1 commands packed into 32-bit words
|
||||||
* packed into 32-bit words (cmd byte in the upper 8 bits via
|
* (cmd byte in the upper 8 bits via `enc_gp0_cmd(cmd)`).
|
||||||
* `enc_gp0_cmd(cmd)` — never a raw shift).
|
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
enum {
|
enum {
|
||||||
gp1_cmd_Reset = 0x00,
|
gp1_cmd_Reset = 0x00,
|
||||||
gp1_cmd_ResetCmdBuffer = 0x01,
|
gp1_cmd_ResetCmdBuffer = 0x01,
|
||||||
gp1_cmd_AcknowledgeIRQ = 0x02,
|
gp1_cmd_AcknowledgeIRQ = 0x02,
|
||||||
gp1_cmd_DisplayEnable = 0x03,
|
gp1_cmd_DisplayEnable = 0x03,
|
||||||
gp1_cmd_DMADirection = 0x04,
|
gp1_cmd_DMADirection = 0x04,
|
||||||
gp1_cmd_StartDisplayArea = 0x05,
|
gp1_cmd_StartDisplayArea = 0x05,
|
||||||
gp1_cmd_HorizontalDisplayRange = 0x06,
|
gp1_cmd_HorizontalDisplayRange = 0x06,
|
||||||
gp1_cmd_VerticalDisplayRange = 0x07,
|
gp1_cmd_VerticalDisplayRange = 0x07,
|
||||||
gp1_cmd_DisplayMode = 0x08,
|
gp1_cmd_DisplayMode = 0x08,
|
||||||
gp1_cmd_SetTextureWindow = 0x0E,
|
/* Note: GP1 only has commands 0x00..0x08.
|
||||||
gp1_cmd_SetDrawAreaTopLeft = 0xE0,
|
* The state-setter commands (SetTextureWindow, * SetDrawArea*, SetDrawOffset, SetMaskBit)
|
||||||
gp1_cmd_SetDrawAreaBottomRight = 0xE1,
|
* live in the GP0 enum as * 0xE1..0xE6.
|
||||||
gp1_cmd_SetDrawOffset = 0xE2,
|
* DrawArea word builders are below as GP0s * macros (since they emit GP0 commands). */
|
||||||
gp1_cmd_SetMaskBit = 0xE3,
|
|
||||||
|
|
||||||
/* ---- Display-mode payload flags (per PSX-SPX §"GP1 Display Mode").
|
/* ---- Display-mode payload flags (per PSX-SPX §"GP1 Display Mode").
|
||||||
* Bit positions match the encoder shifts below; values are the
|
* Bit positions match the encoder shifts below; values are the *payload* bits only (cmd byte is OR'd in by enc_gp1_disp_mode_word). */
|
||||||
* *payload* bits only (the cmd byte is OR'd in by enc_gp1_disp_mode_word). */
|
gp1_disp_HRes_256 = 0x0,
|
||||||
gp1_disp_HRes_256 = 0x0,
|
gp1_disp_HRes_320 = 0x1,
|
||||||
gp1_disp_HRes_320 = 0x1,
|
gp1_disp_HRes_512 = 0x2,
|
||||||
gp1_disp_HRes_512 = 0x2,
|
gp1_disp_HRes_640 = 0x3,
|
||||||
gp1_disp_HRes_640 = 0x3,
|
gp1_disp_VRes_240 = 0x0,
|
||||||
gp1_disp_VRes_240 = 0x0,
|
gp1_disp_VRes_480 = 0x1,
|
||||||
gp1_disp_VRes_480 = 0x1,
|
gp1_disp_Color15 = 0x0,
|
||||||
gp1_disp_Color15 = 0x0,
|
gp1_disp_Color24 = 0x1,
|
||||||
gp1_disp_Color24 = 0x1,
|
gp1_disp_VInterlace = 0x1,
|
||||||
gp1_disp_VInterlace = 0x1,
|
|
||||||
|
|
||||||
/* ---- Layer 1: GP1 display-mode + range + draw-area shifts/masks ---- */
|
/* ---- Layer 1: GP1 display-mode + range + draw-area shifts/widths ---- */
|
||||||
gp1_disp_hres_shift = 0, gp1_disp_hres_width = 2, gp1_disp_hres_mask = 0x3,
|
gp1_disp_hres_pos = 0, gp1_disp_hres_width = 2,
|
||||||
gp1_disp_vres_shift = 2, gp1_disp_vres_width = 1, gp1_disp_vres_mask = 0x1,
|
gp1_disp_vres_pos = 2, gp1_disp_vres_width = 1,
|
||||||
gp1_disp_color_shift = 4, gp1_disp_color_width = 1, gp1_disp_color_mask = 0x1,
|
gp1_disp_color_pos = 4, gp1_disp_color_width = 1,
|
||||||
gp1_disp_interlace_shift = 5, gp1_disp_interlace_width = 1, gp1_disp_interlace_mask = 0x1,
|
gp1_disp_interlace_pos = 5, gp1_disp_interlace_width = 1,
|
||||||
|
|
||||||
/* GP1 horizontal display range: bits 0..11 = X2, bits 12..23 = X1 */
|
/* GP1 horizontal display range: bits 0..11 = X2, bits 12..23 = X1 */
|
||||||
gp1_hrange_x1_shift = 12, gp1_hrange_x1_width = 12, gp1_hrange_x1_mask = 0xFFF,
|
gp1_hrange_x1_pos = 12, gp1_hrange_x1_width = 12,
|
||||||
gp1_hrange_x2_shift = 0, gp1_hrange_x2_width = 12, gp1_hrange_x2_mask = 0xFFF,
|
gp1_hrange_x2_pos = 0, gp1_hrange_x2_width = 12,
|
||||||
|
|
||||||
/* GP1 vertical display range: bits 0..9 = Y2, bits 10..19 = Y1 */
|
/* GP1 vertical display range: bits 0..9 = Y2, bits 10..19 = Y1 */
|
||||||
gp1_vrange_y1_shift = 10, gp1_vrange_y1_width = 10, gp1_vrange_y1_mask = 0x3FF,
|
gp1_vrange_y1_pos = 10, gp1_vrange_y1_width = 10,
|
||||||
gp1_vrange_y2_shift = 0, gp1_vrange_y2_width = 10, gp1_vrange_y2_mask = 0x3FF,
|
gp1_vrange_y2_pos = 0, gp1_vrange_y2_width = 10,
|
||||||
|
|
||||||
/* GP1 draw area (top-left or bottom-right): bits 0..9 = X, bits 10..19 = Y
|
/* GP1 draw area (top-left or bottom-right): bits 0..9 = X, bits 10..19 = Y
|
||||||
* (10-bit signed — caller pre-signs and masks with the named mask) */
|
* (10-bit signed — caller pre-signs) */
|
||||||
gp1_draw_x_shift = 0, gp1_draw_x_width = 10, gp1_draw_x_mask = 0x3FF,
|
gp1_draw_x_pos = 0, gp1_draw_x_width = 10,
|
||||||
gp1_draw_y_shift = 10, gp1_draw_y_width = 10, gp1_draw_y_mask = 0x3FF,
|
gp1_draw_y_pos = 10, gp1_draw_y_width = 10,
|
||||||
};
|
};
|
||||||
|
|
||||||
/* ---- Layer 1.5: GP1 per-field encoders ---- */
|
/* ---- Layer 1.5: GP1 per-field encoders ---- */
|
||||||
#define enc_gp1_disp_hres(h) (((h) & gp1_disp_hres_mask) << gp1_disp_hres_shift)
|
#define enc_gp1_disp_hres(h) ((h) << gp1_disp_hres_pos)
|
||||||
#define enc_gp1_disp_vres(v) (((v) & gp1_disp_vres_mask) << gp1_disp_vres_shift)
|
#define enc_gp1_disp_vres(v) ((v) << gp1_disp_vres_pos)
|
||||||
#define enc_gp1_disp_color(c) (((c) & gp1_disp_color_mask) << gp1_disp_color_shift)
|
#define enc_gp1_disp_color(c) ((c) << gp1_disp_color_pos)
|
||||||
#define enc_gp1_disp_interlace(i) (((i) & gp1_disp_interlace_mask) << gp1_disp_interlace_shift)
|
#define enc_gp1_disp_interlace(i) ((i) << gp1_disp_interlace_pos)
|
||||||
|
|
||||||
#define enc_gp1_hrange_x1(x1) (((x1) & gp1_hrange_x1_mask) << gp1_hrange_x1_shift)
|
#define enc_gp1_hrange_x1(x1) ((x1) << gp1_hrange_x1_pos)
|
||||||
#define enc_gp1_hrange_x2(x2) (((x2) & gp1_hrange_x2_mask) << gp1_hrange_x2_shift)
|
#define enc_gp1_hrange_x2(x2) ((x2) << gp1_hrange_x2_pos)
|
||||||
#define enc_gp1_vrange_y1(y1) (((y1) & gp1_vrange_y1_mask) << gp1_vrange_y1_shift)
|
#define enc_gp1_vrange_y1(y1) ((y1) << gp1_vrange_y1_pos)
|
||||||
#define enc_gp1_vrange_y2(y2) (((y2) & gp1_vrange_y2_mask) << gp1_vrange_y2_shift)
|
#define enc_gp1_vrange_y2(y2) ((y2) << gp1_vrange_y2_pos)
|
||||||
#define enc_gp1_draw_x(x) (((x) & gp1_draw_x_mask) << gp1_draw_x_shift)
|
#define enc_gp1_draw_x(x) ((x) << gp1_draw_x_pos)
|
||||||
#define enc_gp1_draw_y(y) (((y) & gp1_draw_y_mask) << gp1_draw_y_shift)
|
#define enc_gp1_draw_y(y) ((y) << gp1_draw_y_pos)
|
||||||
|
|
||||||
/* ---- Layer 2: GP1 composite encoders ---- */
|
/* ---- Layer 2: GP1 composite encoders ---- */
|
||||||
#define enc_gp1_disp_mode_word(h, v, c, i) (enc_gp0_cmd(gp1_cmd_DisplayMode) | enc_gp1_disp_hres(h) | enc_gp1_disp_vres(v) | enc_gp1_disp_color(c) | enc_gp1_disp_interlace(i))
|
#define enc_gp1_disp_mode_word(h, v, c, i) (enc_gp0_cmd(gp1_cmd_DisplayMode) | enc_gp1_disp_hres(h) | enc_gp1_disp_vres(v) | enc_gp1_disp_color(c) | enc_gp1_disp_interlace(i))
|
||||||
#define enc_gp1_hrange_word(x1, x2) (enc_gp0_cmd(gp1_cmd_HorizontalDisplayRange) | enc_gp1_hrange_x1(x1) | enc_gp1_hrange_x2(x2))
|
#define enc_gp1_hrange_word(x1, x2) (enc_gp0_cmd(gp1_cmd_HorizontalDisplayRange) | enc_gp1_hrange_x1(x1) | enc_gp1_hrange_x2(x2))
|
||||||
#define enc_gp1_vrange_word(y1, y2) (enc_gp0_cmd(gp1_cmd_VerticalDisplayRange) | enc_gp1_vrange_y1(y1) | enc_gp1_vrange_y2(y2))
|
#define enc_gp1_vrange_word(y1, y2) (enc_gp0_cmd(gp1_cmd_VerticalDisplayRange) | enc_gp1_vrange_y1(y1) | enc_gp1_vrange_y2(y2))
|
||||||
#define enc_gp1_draw_area_tl_word(x, y) (enc_gp0_cmd(gp1_cmd_SetDrawAreaTopLeft) | enc_gp1_draw_x(x) | enc_gp1_draw_y(y))
|
|
||||||
#define enc_gp1_draw_area_br_word(x, y) (enc_gp0_cmd(gp1_cmd_SetDrawAreaBottomRight) | enc_gp1_draw_x(x) | enc_gp1_draw_y(y))
|
/* ---- Layer 2: GP0 state-setter composite encoders ----
|
||||||
|
* GP0(0xE3) SetDrawArea top-left and GP0(0xE4) SetDrawArea bottom-right both use the same X/Y 10-bit signed payload as GP1 DisplayRange. */
|
||||||
|
#define enc_gp0_draw_area_tl_word(x, y) (enc_gp0_cmd(gp0_cmd_SetDrawArea_TopLeft) | enc_gp1_draw_x(x) | enc_gp1_draw_y(y))
|
||||||
|
#define enc_gp0_draw_area_br_word(x, y) (enc_gp0_cmd(gp0_cmd_SetDrawArea_BotRight) | enc_gp1_draw_x(x) | enc_gp1_draw_y(y))
|
||||||
|
|
||||||
/* ---- Layer 3: GP1 semantic word builders ---- */
|
/* ---- Layer 3: GP1 semantic word builders ---- */
|
||||||
|
#define gp1_word_Reset() enc_gp0_cmd_word(gp1_cmd_Reset)
|
||||||
|
#define gp1_word_ResetCmdBuffer() enc_gp0_cmd_word(gp1_cmd_ResetCmdBuffer)
|
||||||
|
#define gp1_word_AcknowledgeIRQ() enc_gp0_cmd_word(gp1_cmd_AcknowledgeIRQ)
|
||||||
|
#define gp1_word_StartDisplayArea() enc_gp0_cmd_word(gp1_cmd_StartDisplayArea)
|
||||||
|
|
||||||
#define gp1_word_display_enable(on) (enc_gp0_cmd(gp1_cmd_DisplayEnable) | ((on) & 1))
|
#define gp1_word_display_enable(on) (enc_gp0_cmd(gp1_cmd_DisplayEnable) | ((on) & 1))
|
||||||
#define gp1_word_display_disable() gp1_word_display_enable(0)
|
#define gp1_word_display_disable() gp1_word_display_enable(0)
|
||||||
#define gp1_word_display_mode_320x240_15bit_ntsc enc_gp1_disp_mode_word(gp1_disp_HRes_320, gp1_disp_VRes_240, gp1_disp_Color15, 0)
|
#define gp1_word_display_mode_320x240_15bit_ntsc enc_gp1_disp_mode_word(gp1_disp_HRes_320, gp1_disp_VRes_240, gp1_disp_Color15, 0)
|
||||||
@@ -250,10 +268,105 @@ enum {
|
|||||||
#define gp1_word_horizontal_range(x1, x2) enc_gp1_hrange_word((x1), (x2))
|
#define gp1_word_horizontal_range(x1, x2) enc_gp1_hrange_word((x1), (x2))
|
||||||
#define gp1_word_vertical_range(y1, y2) enc_gp1_vrange_word((y1), (y2))
|
#define gp1_word_vertical_range(y1, y2) enc_gp1_vrange_word((y1), (y2))
|
||||||
|
|
||||||
|
/* ---- Layer 3: GP0 state-setter semantic word builders ---- */
|
||||||
/* DrawArea: top-left = (X, Y), bottom-right = (X, Y) — X/Y in 10-bit signed.
|
/* DrawArea: top-left = (X, Y), bottom-right = (X, Y) — X/Y in 10-bit signed.
|
||||||
* Caller is responsible for sign-conversion before passing in. */
|
* Caller is responsible for sign-conversion before passing in. */
|
||||||
#define gp1_word_draw_area_top_left(x, y) enc_gp1_draw_area_tl_word((x), (y))
|
#define gp0_word_draw_area_top_left(x, y) enc_gp0_draw_area_tl_word((x), (y))
|
||||||
#define gp1_word_draw_area_bottom_right(x, y) enc_gp1_draw_area_br_word((x), (y))
|
#define gp0_word_draw_area_bottom_right(x, y) enc_gp0_draw_area_br_word((x), (y))
|
||||||
|
|
||||||
|
/* ============================================================================
|
||||||
|
* Pre-baked GPU state words
|
||||||
|
* ============================================================================
|
||||||
|
* Common command words for boot-time GPU init and standard display configurations.
|
||||||
|
* ============================================================================ */
|
||||||
|
|
||||||
|
/* ---- Display enable (1-bit payload on DisplayEnable cmd) ---- */
|
||||||
|
#define gp1_word_display_enabled enc_gp0_cmd_word(gp1_cmd_DisplayEnable)
|
||||||
|
#define gp1_word_display_disabled (enc_gp0_cmd_word(gp1_cmd_DisplayEnable) | 1)
|
||||||
|
|
||||||
|
#define gp1_word_DisplayOn() gp1_word_display_enable(0)
|
||||||
|
#define gp1_word_DisplayOff() gp1_word_display_enable(1)
|
||||||
|
|
||||||
|
/* ---- DMA direction (2-bit payload on DMADirection cmd 0x04) ---- */
|
||||||
|
enum {
|
||||||
|
gp1_dma_dir_Off = 0,
|
||||||
|
gp1_dma_dir_FIFO = 1,
|
||||||
|
gp1_dma_dir_CPU_to_GPU = 2,
|
||||||
|
gp1_dma_dir_GPUREAD_to_CPU = 3,
|
||||||
|
};
|
||||||
|
#define gp1_word_dma_direction(dir) (enc_gp0_cmd(gp1_cmd_DMADirection) | ((dir) & 0x3))
|
||||||
|
#define gp1_word_dma_to_gpu() gp1_word_dma_direction(gp1_dma_dir_CPU_to_GPU)
|
||||||
|
#define gp1_word_dma_read_cpu() gp1_word_dma_direction(gp1_dma_dir_GPUREAD_to_CPU)
|
||||||
|
|
||||||
|
/* ---- Standard display ranges (NTSC + PAL pre-baked) ---- */
|
||||||
|
/* Horizontal range values are in video clock units (8 units/pixel); vertical range values are scanline numbers. */
|
||||||
|
enum {
|
||||||
|
/* NTSC horizontal range: X1=608, X2=3168 */
|
||||||
|
gp1_hrange_NTSC_x1 = 0x260,
|
||||||
|
gp1_hrange_NTSC_x2 = 0xC60,
|
||||||
|
/* PAL horizontal range (same as NTSC for most CRTs) */
|
||||||
|
gp1_hrange_PAL_x1 = 0x260,
|
||||||
|
gp1_hrange_PAL_x2 = 0xC60,
|
||||||
|
|
||||||
|
/* NTSC vertical range: Y1=24, Y2=264 */
|
||||||
|
gp1_vrange_NTSC_y1 = 24,
|
||||||
|
gp1_vrange_NTSC_y2 = 264,
|
||||||
|
/* PAL vertical range: Y1=24, Y2=504 */
|
||||||
|
gp1_vrange_PAL_y1 = 24,
|
||||||
|
gp1_vrange_PAL_y2 = 504,
|
||||||
|
};
|
||||||
|
|
||||||
|
#define gp1_word_horizontal_range_ntsc enc_gp1_hrange_word(gp1_hrange_NTSC_x1, gp1_hrange_NTSC_x2)
|
||||||
|
#define gp1_word_horizontal_range_pal enc_gp1_hrange_word(gp1_hrange_PAL_x1, gp1_hrange_PAL_x2)
|
||||||
|
#define gp1_word_vertical_range_ntsc enc_gp1_vrange_word(gp1_vrange_NTSC_y1, gp1_vrange_NTSC_y2)
|
||||||
|
#define gp1_word_vertical_range_pal enc_gp1_vrange_word(gp1_vrange_PAL_y1, gp1_vrange_PAL_y2)
|
||||||
|
|
||||||
|
/* ---- Draw-mode setting (TPage / draw-area allowance) ---- */
|
||||||
|
/* The "drawing enabled" word is the standard post-init state. */
|
||||||
|
enum {
|
||||||
|
/* Per psx-spx, the standard 0xE1 layout has dfe at bit 10. But libpsyx's PutDrawEnv
|
||||||
|
* uses bit 19 (in the "unused" 14-23 range) for dfe in the DR_ENV code[0] — and the
|
||||||
|
* PSX hardware honors bit 19 in the DR_ENV context (not bit 10). So we need a
|
||||||
|
* separate bit definition for the DR_ENV-specific DrawMode. */
|
||||||
|
gp0_DrawMode_DrawToDispBit = 10, // standard psx-spx bit 10 (dfe)
|
||||||
|
gp0_DrawMode_DR_ENV_DrawToDispBit = 19, // libpsyx DR_ENV code[0] (dfe in DR_ENV context)
|
||||||
|
gp0_DrawMode_DR_ENV_isbgBit = 19, // libpsyx uses bit 19 for isbg too
|
||||||
|
};
|
||||||
|
#define gp0_word_draw_mode_drawing_allowed (enc_gp0_cmd(gp0_cmd_DrawModeSetting) | (1 << gp0_DrawMode_DrawToDispBit))
|
||||||
|
|
||||||
|
/* DR_ENV-specific DrawMode variants (libpsyx SetDrawEnv layout).
|
||||||
|
* The DR_ENV is a 16-word packet emitted at boot by gp_screen_init's ac_put_draw_env_demo
|
||||||
|
* atom component. Within the DR_ENV, the 0xE1 command is reused in three different bit
|
||||||
|
* configurations:
|
||||||
|
* code[0] = `gp0_word_draw_mode_drawing_allowed` (dfe=1; standard post-init state)
|
||||||
|
* code[6] = `gp0_word_dr_env_bg_color_cmd(isbg, r, g, b)` (initial-bg-color path)
|
||||||
|
* code[7] = `gp0_word_dr_env_draw_mode(isbg)` (isbg-flag path)
|
||||||
|
* Bits 0-23 of the 0xE1 word are the payload; bits 24-31 are the cmd byte (0xE1). */
|
||||||
|
#define gp0_word_dr_env_bg_color_cmd(isbg, r, g, b) (enc_gp0_cmd(gp0_cmd_DrawModeSetting) | (1 << gp0_DrawMode_DrawToDispBit) | ((isbg) ? gp0_dr_env_isbg_bit : 0) | enc_gp0_color_r(r) | enc_gp0_color_g(g) | enc_gp0_color_b(b))
|
||||||
|
#define gp0_word_dr_env_draw_mode(isbg) (enc_gp0_cmd(gp0_cmd_DrawModeSetting) | (1 << gp0_DrawMode_DrawToDispBit) | ((isbg) ? gp0_dr_env_isbg_bit : 0))
|
||||||
|
|
||||||
|
/* State-setter bare-cmd words (no immediate payload; the GPU uses the current state machine already programmed). */
|
||||||
|
#define gp0_word_set_texture_window() enc_gp0_cmd_word(gp0_cmd_SetTextureWindow)
|
||||||
|
#define gp0_word_set_draw_offset() enc_gp0_cmd_word(gp0_cmd_SetDrawOffset)
|
||||||
|
#define gp0_word_set_mask_bit() enc_gp0_cmd_word(gp0_cmd_SetMaskBit)
|
||||||
|
|
||||||
|
/* DR_ENV code[5] Mask (0xE6 cmd + isbg bit). The isbg bit is set so the GPU knows the auto-clear path is active (paired with code[6] + code[7]). */
|
||||||
|
#define gp0_word_dr_env_mask() (gp0_word_set_mask_bit() | gp0_dr_env_isbg_bit)
|
||||||
|
|
||||||
|
/* DR_ENV pre-baked constants (libpsyx PutDrawEnv layout).
|
||||||
|
* DR_ENV is a 16-word packet: tag = (length << 24) | addr, where length = 15 (15 code words follow) and addr = 0 (chain to nothing). */
|
||||||
|
enum {
|
||||||
|
PolyTag_len_bits = 8,
|
||||||
|
PolyTag_addr_bits = 24,
|
||||||
|
|
||||||
|
gp0_dr_env_tag = (15 << 24) | 0x00FFFFFF,
|
||||||
|
gp0_dr_env_isbg_bit = (1 << gp0_DrawMode_DR_ENV_isbgBit),
|
||||||
|
};
|
||||||
|
|
||||||
|
/* ---- DrawArea at origin (0,0) and full screen (320x240) ---- */
|
||||||
|
#define gp0_word_draw_area_top_left_origin enc_gp0_draw_area_tl_word(0, 0)
|
||||||
|
#define gp0_word_draw_area_bottom_right_320x240 enc_gp0_draw_area_br_word(319, 239)
|
||||||
|
#define gp0_word_draw_area_bottom_right_640x480 enc_gp0_draw_area_br_word(639, 479)
|
||||||
|
|
||||||
#pragma endregion GPU Ports & Commands
|
#pragma endregion GPU Ports & Commands
|
||||||
|
|
||||||
@@ -264,9 +377,9 @@ enum {
|
|||||||
* Read from HW_GP1; the lower bits are DMA-block-size (variable-width).
|
* Read from HW_GP1; the lower bits are DMA-block-size (variable-width).
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
enum {
|
enum {
|
||||||
gp1_Status_BitReady = 31,
|
gp1_Status_BitReady = 31,
|
||||||
gp1_Status_BitSendingDMA = 25,
|
gp1_Status_BitSendingDMA = 25,
|
||||||
gp1_Status_DMABlockSizeShift = 0,
|
gp1_Status_DMABlockSizeShift = 0,
|
||||||
};
|
};
|
||||||
|
|
||||||
#define gp1_status_is_ready() ((HW_GP1[0] >> gp1_Status_BitReady) & 1)
|
#define gp1_status_is_ready() ((HW_GP1[0] >> gp1_Status_BitReady) & 1)
|
||||||
@@ -277,17 +390,14 @@ enum {
|
|||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
* Primitive structs (8 polygon variants + tag)
|
* Primitive structs (8 polygon variants + tag)
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
|
* Each struct follows the GPU-documented memory layout for the corresponding primitive command.
|
||||||
|
* The PolyTag is the OT-link header; the rest of the struct is the primitive's body.
|
||||||
*
|
*
|
||||||
* Each struct follows the GPU-documented memory layout for the corresponding
|
* The current working layouts match the existing demo
|
||||||
* primitive command. The PolyTag is the OT-link header; the rest of the
|
* (floor_tri uses Poly_F3; cube_tri uses Poly_G4).
|
||||||
* struct is the primitive's body.
|
* They are NOT necessarily byte-identical to the PSX-SPX reference layout.
|
||||||
*
|
* The demo layout uses color+vertex interleaving that doesn't match the standard PSX SDK file format.
|
||||||
* The current working layouts match the existing demo (floor_tri uses
|
* For PSX-SDK file compatibility, the textured variants (FT*, GT*) would need layout adjustments.
|
||||||
* Poly_F3; cube_tri uses Poly_G4). They are NOT necessarily byte-identical
|
|
||||||
* to the PSX-SPX reference layout — the demo layout uses color+vertex
|
|
||||||
* interleaving that doesn't match the standard PSX SDK file format. For
|
|
||||||
* PSX-SDK file compatibility, the textured variants (FT*, GT*) would need
|
|
||||||
* layout adjustments; out of scope for this track.
|
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
|
|
||||||
/* ---------- RGB8 (3-byte packed color) ---------- */
|
/* ---------- RGB8 (3-byte packed color) ---------- */
|
||||||
@@ -295,13 +405,13 @@ typedef Struct_(RGB8) { B1 r; B1 g; B1 b; };
|
|||||||
#define rgb8(r,g,b) ((RGB8){r,g,b})
|
#define rgb8(r,g,b) ((RGB8){r,g,b})
|
||||||
|
|
||||||
/* ---------- PolyTag (the OT-link header; 1 word) ---------- */
|
/* ---------- PolyTag (the OT-link header; 1 word) ---------- */
|
||||||
enum {
|
// enum {
|
||||||
polytag_len_bits = 8,
|
// PolyTag_len_bits = 8,
|
||||||
polytag_addr_bits = 24,
|
// PolyTag_addr_bits = 24,
|
||||||
};
|
// };
|
||||||
typedef Struct_(PolyTag) {
|
typedef Struct_(PolyTag) {
|
||||||
union {
|
union {
|
||||||
U4 bf_addr_len;
|
U4 code;
|
||||||
struct {
|
struct {
|
||||||
U4 addr: 24;
|
U4 addr: 24;
|
||||||
U4 len: 8;
|
U4 len: 8;
|
||||||
@@ -309,117 +419,108 @@ typedef Struct_(PolyTag) {
|
|||||||
};
|
};
|
||||||
};
|
};
|
||||||
|
|
||||||
/* DSL cast convention: every cast uses `C_()`, every pointer qualifier
|
|
||||||
* is `R_` (restrict) or `V_` (volatile). No raw C-style casts. RHS values
|
|
||||||
* are assumed to be `U4` — caller passes a `U4` directly.
|
|
||||||
*
|
|
||||||
* IMPORTANT: do NOT name an arg the same as a struct member being
|
|
||||||
* accessed in the body — preprocessor substitution would replace the
|
|
||||||
* member name with the caller's value expression, yielding `->expr`
|
|
||||||
* which is a parse error. Use `v` (value) for the arg instead. */
|
|
||||||
#define set_len(tag,v) (C_(PolyTag_R,tag)->len = u4_(v))
|
#define set_len(tag,v) (C_(PolyTag_R,tag)->len = u4_(v))
|
||||||
#define set_addr(tag,v) (C_(PolyTag_R,tag)->addr = u4_(v))
|
#define set_addr(tag,v) (C_(PolyTag_R,tag)->addr = u4_(v))
|
||||||
/* `set_code` is no longer in the new PolyTag design — the code byte lives
|
/* `set_code` is no longer in the new PolyTag design
|
||||||
* in the primitive body (e.g. `((Poly_F3*)(p))->code`), not in the tag.
|
* (e.g. `((Poly_F3*)(p))->code`), not in the tag.
|
||||||
* Use the typed primitive structs (Poly_F3, Poly_G4, etc.) and the
|
* Use the typed primitive structs (Poly_F3, Poly_G4, etc.) and the `set_poly_*` setters, which set both the tag's length and the code. */
|
||||||
* `set_poly_*` setters, which set both the tag's length and the code. */
|
|
||||||
#define get_len(tag) C_(U4,C_(PolyTag_R,tag)->len)
|
#define get_len(tag) C_(U4,C_(PolyTag_R,tag)->len)
|
||||||
#define get_addr(tag) C_(U4,C_(PolyTag_R,tag)->addr)
|
#define get_addr(tag) C_(U4,C_(PolyTag_R,tag)->addr)
|
||||||
|
|
||||||
/* ---------- Poly_F3 (Flat Triangle; 5 words) ---------- */
|
/* ---------- Poly_F3 (Flat Triangle; 5 words) ---------- */
|
||||||
typedef Struct_(Poly_F3) {
|
typedef Struct_(Poly_F3) {
|
||||||
U4 tag;
|
U4 tag;
|
||||||
RGB8 color;
|
RGB8 color;
|
||||||
B1 code;
|
B1 code;
|
||||||
union {
|
union {
|
||||||
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; };
|
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; };
|
||||||
A3_V2_S2 points;
|
A3_V2_S2 points;
|
||||||
};
|
};
|
||||||
};
|
};
|
||||||
|
|
||||||
/* ---------- Poly_F4 (Flat Quad; 6 words) ---------- */
|
/* ---------- Poly_F4 (Flat Quad; 6 words) ---------- */
|
||||||
typedef Struct_(Poly_F4) {
|
typedef Struct_(Poly_F4) {
|
||||||
U4 tag;
|
U4 tag;
|
||||||
RGB8 color;
|
RGB8 color;
|
||||||
B1 code;
|
B1 code;
|
||||||
union {
|
union {
|
||||||
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; V2_S2 p3; };
|
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; V2_S2 p3; };
|
||||||
A4_V2_S2 points;
|
A4_V2_S2 points;
|
||||||
};
|
};
|
||||||
};
|
};
|
||||||
|
|
||||||
/* ---------- Poly_G3 (Gouraud Triangle; 6 words) ---------- */
|
/* ---------- Poly_G3 (Gouraud Triangle; 7 words) ---------- */
|
||||||
typedef Struct_(Poly_G3) {
|
typedef Struct_(Poly_G3) {
|
||||||
U4 tag; RGB8 c0; B1 code;
|
U4 tag; RGB8 c0; B1 code;
|
||||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||||
V2_S2 p2;
|
V2_S2 p2;
|
||||||
};
|
};
|
||||||
|
|
||||||
/* ---------- Poly_G4 (Gouraud Quad; 5 words in the demo's interleaved layout) ---------- */
|
/* ---------- Poly_G4 (Gouraud Quad; 9 words) ---------- */
|
||||||
typedef Struct_(Poly_G4) {
|
typedef Struct_(Poly_G4) {
|
||||||
U4 tag; RGB8 c0; B1 code;
|
U4 tag; RGB8 c0; B1 code;
|
||||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||||
V2_S2 p2; RGB8 c3; B1 pad3;
|
V2_S2 p2; RGB8 c3; B1 pad3;
|
||||||
V2_S2 p3;
|
V2_S2 p3;
|
||||||
};
|
};
|
||||||
|
|
||||||
/* ---------- Poly_FT3 (Flat Textured Triangle; placeholder layout) ---------- */
|
/* ---------- Poly_FT3 (Flat Textured Triangle; placeholder layout) ---------- */
|
||||||
/* TODO(Ed): verify the textured-variant layout against PSX-SPX when needed. */
|
/* TODO(Ed): verify the textured-variant layout against PSX-SPX when needed. */
|
||||||
typedef Struct_(Poly_FT3) {
|
typedef Struct_(Poly_FT3) {
|
||||||
U4 tag;
|
U4 tag;
|
||||||
RGB8 color;
|
RGB8 color;
|
||||||
B1 code;
|
B1 code;
|
||||||
U4 tpage;
|
U4 tpage;
|
||||||
U4 clut;
|
U4 clut;
|
||||||
V2_S2 p0; U1 u0; U1 v0;
|
V2_S2 p0; U1 u0; U1 v0;
|
||||||
V2_S2 p1; U1 u1; U1 v1;
|
V2_S2 p1; U1 u1; U1 v1;
|
||||||
V2_S2 p2; U1 u2; U1 v2;
|
V2_S2 p2; U1 u2; U1 v2;
|
||||||
};
|
};
|
||||||
|
|
||||||
/* ---------- Poly_FT4 (Flat Textured Quad; placeholder layout) ---------- */
|
/* ---------- Poly_FT4 (Flat Textured Quad) ---------- */
|
||||||
typedef Struct_(Poly_FT4) {
|
typedef Struct_(Poly_FT4) {
|
||||||
U4 tag;
|
U4 tag;
|
||||||
RGB8 color;
|
RGB8 color;
|
||||||
B1 code;
|
B1 code;
|
||||||
U4 tpage;
|
U4 tpage;
|
||||||
U4 clut;
|
U4 clut;
|
||||||
V2_S2 p0; U1 u0; U1 v0;
|
V2_S2 p0; U1 u0; U1 v0;
|
||||||
V2_S2 p1; U1 u1; U1 v1;
|
V2_S2 p1; U1 u1; U1 v1;
|
||||||
V2_S2 p2; U1 u2; U1 v2;
|
V2_S2 p2; U1 u2; U1 v2;
|
||||||
V2_S2 p3; U1 u3; U1 v3;
|
V2_S2 p3; U1 u3; U1 v3;
|
||||||
};
|
};
|
||||||
|
|
||||||
/* ---------- Poly_GT3 (Gouraud Textured Triangle; placeholder layout) ---------- */
|
/* ---------- Poly_GT3 (Gouraud Textured Triangle) ---------- */
|
||||||
typedef Struct_(Poly_GT3) {
|
typedef Struct_(Poly_GT3) {
|
||||||
U4 tag; RGB8 c0; B1 code;
|
U4 tag; RGB8 c0; B1 code;
|
||||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||||
V2_S2 p2;
|
V2_S2 p2;
|
||||||
U4 tpage;
|
U4 tpage;
|
||||||
U4 clut;
|
U4 clut;
|
||||||
V2_S2 tp0; U1 u0; U1 v0;
|
V2_S2 tp0; U1 u0; U1 v0;
|
||||||
V2_S2 tp1; U1 u1; U1 v1;
|
V2_S2 tp1; U1 u1; U1 v1;
|
||||||
V2_S2 tp2; U1 u2; U1 v2;
|
V2_S2 tp2; U1 u2; U1 v2;
|
||||||
};
|
};
|
||||||
|
|
||||||
/* ---------- Poly_GT4 (Gouraud Textured Quad; placeholder layout) ---------- */
|
/* ---------- Poly_GT4 (Gouraud Textured Quad) ---------- */
|
||||||
typedef Struct_(Poly_GT4) {
|
typedef Struct_(Poly_GT4) {
|
||||||
U4 tag; RGB8 c0; B1 code;
|
U4 tag; RGB8 c0; B1 code;
|
||||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||||
V2_S2 p2; RGB8 c3; B1 pad3;
|
V2_S2 p2; RGB8 c3; B1 pad3;
|
||||||
V2_S2 p3;
|
V2_S2 p3;
|
||||||
U4 tpage;
|
U4 tpage;
|
||||||
U4 clut;
|
U4 clut;
|
||||||
V2_S2 tp0; U1 u0; U1 v0;
|
V2_S2 tp0; U1 u0; U1 v0;
|
||||||
V2_S2 tp1; U1 u1; U1 v1;
|
V2_S2 tp1; U1 u1; U1 v1;
|
||||||
V2_S2 tp2; U1 u2; U1 v2;
|
V2_S2 tp2; U1 u2; U1 v2;
|
||||||
V2_S2 tp3; U1 u3; U1 v3;
|
V2_S2 tp3; U1 u3; U1 v3;
|
||||||
};
|
};
|
||||||
|
|
||||||
/* ---------- Primitive setters (C-level, no emitted words) ----------
|
/* ---------- Primitive setters (C-level) ----------
|
||||||
* DSL cast convention: every cast via C_(), every pointer via R_/V_. */
|
* DSL cast convention: every cast via C_(), every pointer via R_/V_. */
|
||||||
#define set_poly_f3(p) set_len(p, 4), C_(Poly_F3_R, p)->code = gp0_cmd_poly_f3
|
#define set_poly_f3(p) set_len(p, 4), C_(Poly_F3_R, p)->code = gp0_cmd_poly_f3
|
||||||
#define set_poly_ft3(p) set_len(p, 7), C_(Poly_FT3_R,p)->code = gp0_cmd_poly_ft3
|
#define set_poly_ft3(p) set_len(p, 7), C_(Poly_FT3_R,p)->code = gp0_cmd_poly_ft3
|
||||||
@@ -440,7 +541,6 @@ typedef Struct_(Poly_GT4) {
|
|||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
* Texture Page (TPage) bit layout
|
* Texture Page (TPage) bit layout
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
*
|
|
||||||
* The TPage data word sent via GP0(0x2X) has:
|
* The TPage data word sent via GP0(0x2X) has:
|
||||||
* bits 0..3 = texture page X (4 bits, 64-px units, 0..16)
|
* bits 0..3 = texture page X (4 bits, 64-px units, 0..16)
|
||||||
* bit 4 = texture page Y (1 bit, 64-px units, 0/1)
|
* bit 4 = texture page Y (1 bit, 64-px units, 0/1)
|
||||||
@@ -450,94 +550,81 @@ typedef Struct_(Poly_GT4) {
|
|||||||
* bit 10 = drawing to display area (1 bit)
|
* bit 10 = drawing to display area (1 bit)
|
||||||
* bit 11 = texture disable (1 bit)
|
* bit 11 = texture disable (1 bit)
|
||||||
* bits 12..31 = reserved (zero)
|
* bits 12..31 = reserved (zero)
|
||||||
*
|
|
||||||
* The previous version of this file had `gp0_tpage_semi_trans_shift
|
|
||||||
* = 7`, which is WRONG — semi-transparency lives at bits 5..6 (after Y
|
|
||||||
* at bit 4). Likewise the prior `gp0_tpage_clut_depth_shift = 12` and
|
|
||||||
* `gp0_tpage_y_flip_bit = 15` referenced fields that don't exist on the
|
|
||||||
* TPage word. See design.md §10.9 and PSX-SPX §"Rendering Attributes".
|
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
enum {
|
enum {
|
||||||
/* ---- Layer 1: TPage bitfield shifts / widths / masks ---- */
|
/* ---- Layer 1: TPage bitfield shifts / widths ---- */
|
||||||
gp0_tpage_x_shift = 0, gp0_tpage_x_width = 4, gp0_tpage_x_mask = 0xF,
|
gp0_tpage_x_pos = 0, gp0_tpage_x_width = 4,
|
||||||
gp0_tpage_y_shift = 4, gp0_tpage_y_width = 1, gp0_tpage_y_mask = 0x1,
|
gp0_tpage_y_pos = 4, gp0_tpage_y_width = 1,
|
||||||
gp0_tpage_semi_trans_shift = 5, gp0_tpage_semi_trans_width = 2, gp0_tpage_semi_trans_mask = 0x3,
|
gp0_tpage_semi_trans_pos = 5, gp0_tpage_semi_trans_width = 2,
|
||||||
gp0_tpage_color_depth_shift = 7, gp0_tpage_color_depth_width = 2, gp0_tpage_color_depth_mask = 0x3,
|
gp0_tpage_color_depth_pos = 7, gp0_tpage_color_depth_width = 2,
|
||||||
gp0_tpage_dither_shift = 9, gp0_tpage_dither_width = 1, gp0_tpage_dither_mask = 0x1,
|
gp0_tpage_dither_pos = 9, gp0_tpage_dither_width = 1,
|
||||||
gp0_tpage_draw_to_disp_shift = 10, gp0_tpage_draw_to_disp_width = 1, gp0_tpage_draw_to_disp_mask = 0x1,
|
gp0_tpage_draw_to_disp_pos = 10, gp0_tpage_draw_to_disp_width = 1,
|
||||||
gp0_tpage_tex_disable_shift = 11, gp0_tpage_tex_disable_width = 1, gp0_tpage_tex_disable_mask = 0x1,
|
gp0_tpage_tex_disable_pos = 11, gp0_tpage_tex_disable_width = 1,
|
||||||
|
|
||||||
/* TPage color-depth payload values (NOT bit positions — these go in
|
/* TPage color-depth payload values (NOT bit positions — these go in
|
||||||
* the 2-bit field at gp0_tpage_color_depth_shift). */
|
* the 2-bit field at gp0_tpage_color_depth_pos). */
|
||||||
gp0_tpage_color_4bpp = 0x0,
|
gp0_tpage_color_4bpp = 0x0,
|
||||||
gp0_tpage_color_8bpp = 0x1,
|
gp0_tpage_color_8bpp = 0x1,
|
||||||
gp0_tpage_color_16bpp = 0x2,
|
gp0_tpage_color_16bpp = 0x2,
|
||||||
|
|
||||||
/* TPage semi-transparency mode payload values (NOT bit positions). */
|
/* Default TPage value libpsyx's SetDefDrawEnv writes (matches the `li v1, 10; sh v1, 20(v0)` sequence at C11_only.elf:0x8001273C). */
|
||||||
gp0_tpage_semi_trans_none = 0x0,
|
gp0_tpage_default = 10,
|
||||||
gp0_tpage_semi_trans_alpha = 0x1,
|
|
||||||
gp0_tpage_semi_trans_add = 0x2,
|
/* TPage semi-transparency mode payload values. */
|
||||||
gp0_tpage_semi_trans_sub = 0x3,
|
gp0_tpage_semi_trans_none = 0x0,
|
||||||
|
gp0_tpage_semi_trans_alpha = 0x1,
|
||||||
|
gp0_tpage_semi_trans_add = 0x2,
|
||||||
|
gp0_tpage_semi_trans_sub = 0x3,
|
||||||
};
|
};
|
||||||
|
|
||||||
/* ---- Layer 1.5: TPage per-field encoders. Mirrors enc_gte_sf/mx/v in
|
/* ---- Layer 1.5: TPage per-field encoders. Mirrors enc_gte_sf/mx/v in gte.h. ---- */
|
||||||
* gte.h lines 342-347. ---- */
|
#define enc_gp0_tpage_x(x) ((x) << gp0_tpage_x_pos)
|
||||||
#define enc_gp0_tpage_x(x) (((x) & gp0_tpage_x_mask) << gp0_tpage_x_shift)
|
#define enc_gp0_tpage_y(y) ((y) << gp0_tpage_y_pos)
|
||||||
#define enc_gp0_tpage_y(y) (((y) & gp0_tpage_y_mask) << gp0_tpage_y_shift)
|
#define enc_gp0_tpage_semi_trans(s) ((s) << gp0_tpage_semi_trans_pos)
|
||||||
#define enc_gp0_tpage_semi_trans(s) (((s) & gp0_tpage_semi_trans_mask) << gp0_tpage_semi_trans_shift)
|
#define enc_gp0_tpage_color_depth(c) ((c) << gp0_tpage_color_depth_pos)
|
||||||
#define enc_gp0_tpage_color_depth(c) (((c) & gp0_tpage_color_depth_mask) << gp0_tpage_color_depth_shift)
|
#define enc_gp0_tpage_dither(d) ((d) << gp0_tpage_dither_pos)
|
||||||
#define enc_gp0_tpage_dither(d) (((d) & gp0_tpage_dither_mask) << gp0_tpage_dither_shift)
|
#define enc_gp0_tpage_draw_to_disp(d) ((d) << gp0_tpage_draw_to_disp_pos)
|
||||||
#define enc_gp0_tpage_draw_to_disp(d) (((d) & gp0_tpage_draw_to_disp_mask) << gp0_tpage_draw_to_disp_shift)
|
#define enc_gp0_tpage_tex_disable(t) ((t) << gp0_tpage_tex_disable_pos)
|
||||||
#define enc_gp0_tpage_tex_disable(t) (((t) & gp0_tpage_tex_disable_mask) << gp0_tpage_tex_disable_shift)
|
|
||||||
|
|
||||||
/* ---- Layer 2: TPage composite encoder. Mirrors enc_gte_cmdw in gte.h
|
/* ---- Layer 2: TPage composite encoder. Mirrors enc_gte_cmdw in gte.h ---- */
|
||||||
* line 350. ---- */
|
|
||||||
#define enc_gp0_tpage_word(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
|
#define enc_gp0_tpage_word(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
|
||||||
(enc_gp0_tpage_x(x) \
|
(enc_gp0_tpage_x(x) \
|
||||||
| enc_gp0_tpage_y(y) \
|
| enc_gp0_tpage_y(y) \
|
||||||
| enc_gp0_tpage_semi_trans(semi_trans) \
|
| enc_gp0_tpage_semi_trans(semi_trans) \
|
||||||
| enc_gp0_tpage_color_depth(color_depth) \
|
| enc_gp0_tpage_color_depth(color_depth) \
|
||||||
| enc_gp0_tpage_dither(dither) \
|
| enc_gp0_tpage_dither(dither) \
|
||||||
| enc_gp0_tpage_draw_to_disp(draw_to_disp) \
|
| enc_gp0_tpage_draw_to_disp(draw_to_disp) \
|
||||||
| enc_gp0_tpage_tex_disable(tex_disable))
|
| enc_gp0_tpage_tex_disable(tex_disable))
|
||||||
|
|
||||||
typedef Struct_(TexturePage) { U4 raw; };
|
typedef Struct_(TexturePage) { U4 raw; };
|
||||||
|
|
||||||
/* ---- Layer 3: TPage semantic word builder ---- */
|
/* ---- Layer 3: TPage semantic word builder ---- */
|
||||||
#define gp0_word_tpage(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
|
#define gp0_word_tpage(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
|
||||||
enc_gp0_tpage_word((x), (y), (semi_trans), (color_depth), (dither), (draw_to_disp), (tex_disable))
|
enc_gp0_tpage_word((x), (y), (semi_trans), (color_depth), (dither), (draw_to_disp), (tex_disable))
|
||||||
#pragma endregion TPage
|
#pragma endregion TPage
|
||||||
|
|
||||||
#pragma region CLUT
|
#pragma region CLUT
|
||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
* CLUT (Color Look-Up Table) semantics
|
* CLUT (Color Look-Up Table) semantics
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
*
|
|
||||||
* CLUT is loaded into VRAM by sending a GP0 command whose payload is:
|
* CLUT is loaded into VRAM by sending a GP0 command whose payload is:
|
||||||
* bits 0..5 = Y in 16-px units (palette row)
|
* bits 0..5 = Y in 16-px units (palette row)
|
||||||
* bits 6..14 = X in 16-px units (palette column)
|
* bits 6..14 = X in 16-px units (palette column)
|
||||||
* bits 15..23 = reserved (zero)
|
* bits 15..23 = reserved (zero)
|
||||||
* bits 24..31 = command byte — 0x20 (4bpp load) or 0x25 (8bpp load)
|
* bits 24..31 = command byte — 0x20 (4bpp load) or 0x25 (8bpp load)
|
||||||
*
|
|
||||||
* The previous version used `depth_4bpp ? 0 : 5` as the lower-5 bits
|
|
||||||
* of the cmd byte — that's an opaque ternary that hides which opcode
|
|
||||||
* is being sent. The two cmd-byte values are now named; one macro per
|
|
||||||
* depth. Mirrors the named-opcode rule from mips.h lines 188-271 and
|
|
||||||
* the `gte_cmd_rtpt` / `gte_cmd_nclip` named-opcode pattern from
|
|
||||||
* gte.h lines 209-214.
|
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
enum {
|
enum {
|
||||||
/* ---- Layer 1: CLUT bitfield shifts / widths / masks ---- */
|
/* ---- Layer 1: CLUT bitfield shifts / widths ---- */
|
||||||
gp0_clut_y_shift = 0, gp0_clut_y_width = 6, gp0_clut_y_mask = 0x3F,
|
gp0_clut_y_pos = 0, gp0_clut_y_width = 6,
|
||||||
gp0_clut_x_shift = 6, gp0_clut_x_width = 9, gp0_clut_x_mask = 0x1FF,
|
gp0_clut_x_pos = 6, gp0_clut_x_width = 9,
|
||||||
/* CLUT-load cmd-byte variants — the upper byte of the GP0 word. */
|
/* CLUT-load cmd-byte variants — the upper byte of the GP0 word. */
|
||||||
gp0_clut_cmd_Load4bpp = 0x20,
|
gp0_clut_cmd_Load4bpp = 0x20,
|
||||||
gp0_clut_cmd_Load8bpp = 0x25,
|
gp0_clut_cmd_Load8bpp = 0x25,
|
||||||
};
|
};
|
||||||
|
|
||||||
/* ---- Layer 1.5: CLUT per-field encoders ---- */
|
/* ---- Layer 1.5: CLUT per-field encoders ---- */
|
||||||
#define enc_gp0_clut_x(x) (((x) & gp0_clut_x_mask) << gp0_clut_x_shift)
|
#define enc_gp0_clut_x(x) ((x) << gp0_clut_x_pos)
|
||||||
#define enc_gp0_clut_y(y) (((y) & gp0_clut_y_mask) << gp0_clut_y_shift)
|
#define enc_gp0_clut_y(y) ((y) << gp0_clut_y_pos)
|
||||||
|
|
||||||
/* ---- Layer 2: CLUT composite encoder ---- */
|
/* ---- Layer 2: CLUT composite encoder ---- */
|
||||||
#define enc_gp0_clut_word(cmd, x, y) (enc_gp0_cmd(cmd) | enc_gp0_clut_x(x) | enc_gp0_clut_y(y))
|
#define enc_gp0_clut_word(cmd, x, y) (enc_gp0_cmd(cmd) | enc_gp0_clut_x(x) | enc_gp0_clut_y(y))
|
||||||
@@ -552,7 +639,6 @@ enum {
|
|||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
* TIM file format constants and headers
|
* TIM file format constants and headers
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
*
|
|
||||||
* TIM (Sony .TIM texture image) file structure:
|
* TIM (Sony .TIM texture image) file structure:
|
||||||
* +0x00 U4 file_id (always 0x10 = TIM magic)
|
* +0x00 U4 file_id (always 0x10 = TIM magic)
|
||||||
* +0x04 U4 version (always 0x00 for v1)
|
* +0x04 U4 version (always 0x00 for v1)
|
||||||
@@ -570,42 +656,56 @@ enum {
|
|||||||
* +0x06 U2 px_height
|
* +0x06 U2 px_height
|
||||||
* +0x08 ... pixel data
|
* +0x08 ... pixel data
|
||||||
*
|
*
|
||||||
* Future?: add `tim_load_to_vram(tim_ptr, vram_addr)` that
|
* Future?: add `tim_load_to_vram(tim_ptr, vram_addr)` that emits the necessary GP0 commands.
|
||||||
* emits the necessary GP0 commands. Stoppped for now at the
|
* Stoppped for now at the struct + enum level.
|
||||||
* struct + enum level for this track.
|
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
enum {
|
enum {
|
||||||
tim_file_id_magic = 0x10,
|
tim_file_id_magic = 0x10,
|
||||||
tim_type_4bpp = 0x00,
|
tim_type_4bpp = 0x00,
|
||||||
tim_type_8bpp = 0x01,
|
tim_type_8bpp = 0x01,
|
||||||
tim_type_16bpp = 0x02,
|
tim_type_16bpp = 0x02,
|
||||||
tim_type_32bpp = 0x03,
|
tim_type_32bpp = 0x03,
|
||||||
tim_type_mixed = 0x04,
|
tim_type_mixed = 0x04,
|
||||||
tim_flag_has_clut = 0x08,
|
tim_flag_has_clut = 0x08,
|
||||||
};
|
};
|
||||||
|
|
||||||
typedef Struct_(TIM_Header) {
|
typedef Struct_(TIM_Header) {
|
||||||
U4 file_id; /* always 0x10 = "TIM" magic */
|
U4 file_id; /* always 0x10 = "TIM" magic */
|
||||||
U4 version; /* ignored; always 0 */
|
U4 version; /* ignored; always 0 */
|
||||||
U4 flags; /* bits 0..2 = type, bit 3 = has_clut */
|
U4 flags; /* bits 0..2 = type, bit 3 = has_clut */
|
||||||
};
|
};
|
||||||
typedef Struct_(TIM_SectionHeader) {
|
typedef Struct_(TIM_SectionHeader) {
|
||||||
U4 section_length; /* bytes in this section including this header */
|
U4 section_length; /* bytes in this section including this header */
|
||||||
U2 org_x; /* origin in VRAM */
|
U2 org_x; /* origin in VRAM */
|
||||||
U2 org_y;
|
U2 org_y;
|
||||||
U2 width; /* width in pixels */
|
U2 width; /* width in pixels */
|
||||||
U2 height; /* height in pixels */
|
U2 height; /* height in pixels */
|
||||||
};
|
};
|
||||||
#pragma endregion TIM File Format
|
#pragma endregion TIM File Format
|
||||||
|
|
||||||
#pragma region Tape-Side Macros
|
#pragma region Tape-Side Macros
|
||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
* Tape-side macro components
|
* Tape-side GPU operations (NOT in this header)
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
*
|
*
|
||||||
* TODO: mac_gp0_send — write a 32-bit GPU command word to HW_GP0 from
|
* No `mac_gp0_send` or related macros live in gp.h.
|
||||||
* within an atom body. Requires placeholder-pun on a runtime GPR holding
|
* Rationale: the Lottes tape model uses OT-DMA for primitive submission, so atom bodies write to main RAM (the OT/primitive buffer)
|
||||||
* the port address.
|
* and to GTE state — never directly to the GPU ports at 0x1F801810 / 0x1F801814.
|
||||||
|
* See `mac_format_f3_color`, `mac_insert_ot_tag`, `mac_gte_store_f3` in lottes_tape.h for the patterns atom bodies actually use.
|
||||||
|
*
|
||||||
|
* If a feature need arises requires tape-side GPU port writes
|
||||||
|
* (e.g. DMA-kick to start GPU consumption of the OT, VBlank sync via GP1 status poll),
|
||||||
|
* the right home is `lottes_tape.h` alongside the rest of the `mac_*` family:
|
||||||
|
* 1. The caller pins a register to hold the IO base, e.g. register U4 r_io rgcc(R_T4) = IO_BASE_ADDR;
|
||||||
|
* The compiler emits `lui R_T4, IO_BASE_ADDR_HI16` outside the atom body (in the C prologue before tape_run).
|
||||||
|
* 2. The atom body uses `store_word(R_data, R_T4, GPIO_PORT0_OFFSET)` to write to GP0, and `store_word(R_data, R_T4, GPIO_PORT1_OFFSET)`
|
||||||
|
* to write to GP1. Both are preprocessor-encodable because R_T4 is a fixed register and the GPIO_PORT*_OFFSET constants
|
||||||
|
* fit in the `sw`'s 16-bit signed offset field. No placeholder-pun, no asm constraints, no hidden register choice.
|
||||||
|
* Same pattern as the old graphics_hello/hello_gp_routines.s `reg_io_offset`/`gcmd_push` convention.
|
||||||
|
*
|
||||||
|
* This mirrors the existing tape-side wave-context discipline:
|
||||||
|
* the caller binds the IO-base register via `rgcc()`, the macro assumes the binding is in effect,
|
||||||
|
* and the encoding falls out at preprocessor time.
|
||||||
|
* No additional GPU-domain macro layer required.
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
/* #define mac_gp0_send(r_gp_port, word) ... deferred */
|
|
||||||
#pragma endregion Tape-Side Macros
|
#pragma endregion Tape-Side Macros
|
||||||
|
|||||||
@@ -2,10 +2,8 @@
|
|||||||
* duffle DSL — GPU Vendor Mnemonics (opt-in)
|
* duffle DSL — GPU Vendor Mnemonics (opt-in)
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
*
|
*
|
||||||
* Provides the PSYQ-style CamelCase aliases for the canonical duffle GPU
|
* Provides the PSYQ-style CamelCase aliases for the duffle GPU primitive setters and OT operations.
|
||||||
* primitive setters and OT operations. The duffle snake_case names are
|
* The duffle snake_case names are primary; this header is for users who prefer the PSYQ SDK function names from the legacy C API.
|
||||||
* primary; this header is for users who prefer the PSYQ SDK function
|
|
||||||
* names from the legacy C API.
|
|
||||||
*
|
*
|
||||||
* USAGE: #include "duffle/gp_vendor_sym.h" // after gp.h
|
* USAGE: #include "duffle/gp_vendor_sym.h" // after gp.h
|
||||||
*
|
*
|
||||||
@@ -23,15 +21,11 @@
|
|||||||
* OT operations:
|
* OT operations:
|
||||||
* AddPrim(ot, p) -> orderingtbl_add_primitive(ot, p)
|
* AddPrim(ot, p) -> orderingtbl_add_primitive(ot, p)
|
||||||
*
|
*
|
||||||
* The gp0_cmd_* / gp1_cmd_* byte constants are already short and
|
* The gp0_cmd_* / gp1_cmd_* byte constants are already short and descriptive; no vendor alias is provided for them.
|
||||||
* descriptive; no vendor alias is provided for them.
|
|
||||||
*
|
|
||||||
* The vendor mnemonics are NOT registered with the duffle word-count
|
|
||||||
* metadata (tape_atom.metadata.h). They expand to the duffle canonical
|
|
||||||
* macros which DO have word-count entries (the ones emitted by
|
|
||||||
* mac_format_f3_color / mac_gte_store_f3 / etc.). Verification: V13
|
|
||||||
* (objdump byte-identical) holds.
|
|
||||||
*
|
*
|
||||||
|
* The vendor mnemonics are NOT registered with the duffle word-count metadata (word_counts.metadata.h).
|
||||||
|
* They expand to the duffle macros which DO have word-count entries
|
||||||
|
* (the ones emitted by mac_format_f3_color / mac_gte_store_f3 / etc.). Verification: V13 (objdump byte-identical) holds.
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
|
|
||||||
#ifdef INTELLISENSE_DIRECTIVES
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
|||||||
@@ -0,0 +1,394 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
# include "gte.h"
|
||||||
|
# include "gp.h"
|
||||||
|
# include "lottes_tape.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
ATOM_FILE_DEBUGGER_LINE_MARKER(gte_atom_c);
|
||||||
|
|
||||||
|
#pragma region MACs (Mips Atom Components)
|
||||||
|
|
||||||
|
/* Words: 3; Loads 3 S2 indices from the face array */
|
||||||
|
FI_ Slice_MipsCode ac_load_tri_indices(AtomBuilder_R ab, U4 r_face_cusor, U4 r_i0, U4 r_i1, U4 r_i2)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)),
|
||||||
|
load_half_u(r_i1, r_face_cusor, 1 * S_(S2)),
|
||||||
|
load_half_u(r_i2, r_face_cusor, 2 * S_(S2)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_gte_mv_to_cr_diag_v3s4(AtomBuilder_R ab, Reg_(V3_S4) v) MipsAtomComp_Proc_(ab, {
|
||||||
|
gte_mv_to_ctrl_r(v.y, gte_cr_RT13),
|
||||||
|
gte_mv_to_ctrl_r(v.z, gte_cr_RT22),
|
||||||
|
gte_mv_to_ctrl_r(v.x, gte_cr_RT11),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_gte_ld_ir123_v3s4(AtomBuilder_R ab, Reg_(V3_S4) v) MipsAtomComp_Proc_(ab, {
|
||||||
|
gte_mv_to_data_r(v.x, C2_IR1),
|
||||||
|
gte_mv_to_data_r(v.y, C2_IR2),
|
||||||
|
gte_mv_to_data_r(v.z, C2_IR3),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* ─── GTE OP cross product (a × b → a) ───
|
||||||
|
* Sets up RT diagonal from a.xyz, IR1/2/3 from b.xyz, fires OP,
|
||||||
|
* reads MAC1/2/3, shifts right 12 (S12.20 → S12.0 OuterProduct12), writes back to a.xyz.
|
||||||
|
* Composes the three sub-primitives (RT-load, IR-load, OP, MAC-read, shift)
|
||||||
|
* into one component for use by atoms that need the cross product inline.
|
||||||
|
*
|
||||||
|
* Output gpr (a) aliases source-A gpr; MAC read clobbers source-A's load targets,
|
||||||
|
* but by that point the RT load is complete and source A is dead.
|
||||||
|
* Pipeline: clobbers IR1..3, MAC1..3, RT11..33.
|
||||||
|
*
|
||||||
|
* The CPU→COP2 transfer chains (3 ctc2, 3 mtc2) require a 2-slot retirement gap,
|
||||||
|
* and the MFC2→GPR chain (3 mfc2) requires a 1-slot retirement gap, before the GPR can be read.
|
||||||
|
* The hazard nops are inlined below — same convention as ac_gte_gpf_scale — so any atom body inlining this component inherits them.
|
||||||
|
*
|
||||||
|
* Words: 18 (3 ctc2 + 2 nop + 3 mtc2 + 2 nop + 1 op + 3 mfc2 + 1 nop + 3 sra).
|
||||||
|
*/
|
||||||
|
FI_ Slice_MipsCode ac_gte_op_cross_v3s4(AtomBuilder_R ab, Reg_(V3_S4) a, Reg_(V3_S4) b) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
mac_gte_mv_to_cr_diag_v3s4(a), GteDelay_ /* RT diagonal: D1 = a.x, D2 = a.y, D3 = a.z */
|
||||||
|
mac_gte_ld_ir123_v3s4(b), GteDelay_ /* IR: second operand (b.xyz) */
|
||||||
|
gte_cmdw_cross, /* OP: MAC1/2/3 = a × b (S12.20) */
|
||||||
|
mac_gte_mv_from_mac123_v3s4(a), GteDelay_ /* Read MAC1/2/3 → a.xyz (overwrites source-A's load targets) */
|
||||||
|
mac_shift_aright_v3s4_self(a, 12), /* Right-shift MAC by 12 (S12.20 → S12.0 OuterProduct12) */
|
||||||
|
})
|
||||||
|
|
||||||
|
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
|
||||||
|
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
|
||||||
|
FI_ Slice_MipsCode ac_gte_store_f3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)),
|
||||||
|
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_F3,p1)),
|
||||||
|
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_F3,p2)),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
|
||||||
|
I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, Reg vbase, Reg v0, Reg v1, Reg v2) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
shift_lleft(R_AT, v0, v3s2_byteoff), add_u_self(R_AT, vbase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||||
|
shift_lleft(R_AT, v1, v3s2_byteoff), add_u_self(R_AT, vbase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
||||||
|
shift_lleft(R_AT, v2, v3s2_byteoff), add_u_self(R_AT, vbase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
|
||||||
|
* G4 triangle portion to p0/p1/p2.
|
||||||
|
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
|
||||||
|
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
|
||||||
|
* (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
|
||||||
|
FI_ Slice_MipsCode ac_gte_store_g4_p012(AtomBuilder_R ab, Reg r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)),
|
||||||
|
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)),
|
||||||
|
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2)),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* Words: 1; Stores the V3 screen coord to the G4's p3 slot.
|
||||||
|
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2;
|
||||||
|
* SXY0 still holds v0.screen from the earlier RTPT.
|
||||||
|
*/
|
||||||
|
FI_ Slice_MipsCode ac_gte_store_g4_p3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) })
|
||||||
|
|
||||||
|
/* ─── STAGE 1 of normalize: SQR + mfc2 MAC1/2/3 ───
|
||||||
|
* Emits squared magnitude per component (in MAC1/2/3) into caller-provided scratch regs. */
|
||||||
|
FI_ Slice_MipsCode ac_gte_sqr_v3(AtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_sq_x, U4 r_sq_y, U4 r_sq_z) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop),
|
||||||
|
gte_mv_from_data_r(r_sq_x, C2_MAC1),
|
||||||
|
gte_mv_from_data_r(r_sq_y, C2_MAC2),
|
||||||
|
gte_mv_from_data_r(r_sq_z, C2_MAC3),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* ─── SQR FIRE — mtc2 3 GPRs into IR1/IR2/IR3, then fire SQR. ─── */
|
||||||
|
FI_ Slice_MipsCode ac_gte_sqr_v3s4(AtomBuilder_R ab, Reg r_sx, Reg r_sy, Reg r_sz, MipsCode delay_slot)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
gte_mv_to_data_r(r_sx, C2_IR1),
|
||||||
|
gte_mv_to_data_r(r_sy, C2_IR2),
|
||||||
|
gte_mv_to_data_r(r_sz, C2_IR3),
|
||||||
|
delay_slot, gte_cmdw_sqr,
|
||||||
|
})
|
||||||
|
|
||||||
|
/* ─── STAGE 4 of normalize: mtc2 IR0..3 + GPF + mfc2 MAC + srav finalize ───
|
||||||
|
* Reusable standalone — given an IR0 = 1/|v| estimate (typically from a sqrtbl lookup) and a shift count
|
||||||
|
* (typically (31 - LZCR)/2), multiplies IR0*IR[i] via GPF and shifts right to produce the normalized output.
|
||||||
|
* Used standalone for "scale vector by scalar".
|
||||||
|
* Words: 11. Clobbers: IR0..3, MAC1..3. Uses gte_cmdw_gpf (sf=0, lm=0). */
|
||||||
|
FI_ Slice_MipsCode ac_gte_gpf_scale(AtomBuilder_R ab,
|
||||||
|
U4 r_sx, U4 r_sy, U4 r_sz,
|
||||||
|
U4 r_recip_est, U4 r_shift,
|
||||||
|
U4 r_dx, U4 r_dy, U4 r_dz)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
gte_mv_to_data_r(r_recip_est, C2_IR0),
|
||||||
|
gte_mv_to_data_r(r_sx, C2_IR1),
|
||||||
|
gte_mv_to_data_r(r_sy, C2_IR2),
|
||||||
|
gte_mv_to_data_r(r_sz, C2_IR3),
|
||||||
|
GteDelay_ nop2, /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */
|
||||||
|
gte_cmdw_gpf,
|
||||||
|
gte_mv_from_data_r(r_dx, C2_MAC1),
|
||||||
|
gte_mv_from_data_r(r_dy, C2_MAC2),
|
||||||
|
gte_mv_from_data_r(r_dz, C2_MAC3),
|
||||||
|
shift_aright_var(r_dx, r_dx, r_shift),
|
||||||
|
shift_aright_var(r_dy, r_dy, r_shift),
|
||||||
|
shift_aright_var(r_dz, r_dz, r_shift),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* ─── TRANS MATRIX (libgte TransMatrix port) ───
|
||||||
|
* Atom component — auto-generates mac_trans_matrix Mac composer macro.
|
||||||
|
* m->t = v (struct copy; libgte's TransMatrix at 0x8001a540 is just 3 store_words, no GTE, no add).
|
||||||
|
* Uses 1 GPR (r_t1 = off value) per axis; per-axis load-delay-slot pattern.
|
||||||
|
* Words: 9. Clobbers: r_t1. */
|
||||||
|
FI_ Slice_MipsCode ac_trans_mt3s3s4(AtomBuilder_R ab
|
||||||
|
, U4 r_mtx, U4 r_off
|
||||||
|
, U4 r_t0, U4 r_t1, U4 r_t2
|
||||||
|
) MipsAtomComp_Proc_(ab, {
|
||||||
|
load_word( r_t0, r_off, O_(V3_S4,x)),
|
||||||
|
load_word( r_t1, r_off, O_(V3_S4,y)),
|
||||||
|
load_word( r_t2, r_off, O_(V3_S4,z)),
|
||||||
|
store_word(r_t0, r_mtx, O_(MT3_S2S4,t[0])),
|
||||||
|
store_word(r_t1, r_mtx, O_(MT3_S2S4,t[1])),
|
||||||
|
store_word(r_t2, r_mtx, O_(MT3_S2S4,t[2])),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* ─── LZCR ROUND EVEN + HALF-SHIFT ───
|
||||||
|
* Takes the raw LZCR leading-zero/ones count (from mfc2 C2_LZCR, range 1..32 per PSX-SPX cop2r31) and the |v|² sum (in r_mag_sq from the MAC1+MAC2+MAC3 add).
|
||||||
|
* Produces:
|
||||||
|
* r_shift ← LZCR rounded down to even (clear bit 0)
|
||||||
|
* r_mag_sq_copy ← |v|² sum (moved out of r_mag_sq before it's overwritten)
|
||||||
|
* r_mag_sq ← (31 - even_LZCR) / 2 = the final srav/GPF shift amount
|
||||||
|
*
|
||||||
|
* Rounding to even ensures (31 - LZCR) is always odd, so the >> 1 division is consistent — no 0.5 loss.
|
||||||
|
* The caller branches on LZCR < 24 to decide left-shift vs right-shift of r_mag_sq_copy, then saves the shift count.
|
||||||
|
*
|
||||||
|
* Note: C2_LZCR (cop2r31) is a fixed read-only C2 data register — the caller must read it via mfc2 from C2_LZCR;
|
||||||
|
* there is no register choice at the hardware level. Only the GPR that holds the result is caller-determined. */
|
||||||
|
FI_ Slice_MipsCode ac_lzcr_round_even_half_shift(AtomBuilder_R ab,
|
||||||
|
U4 r_shift,
|
||||||
|
U4 r_mag_sq,
|
||||||
|
U4 r_mag_sq_copy)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
and_i(r_shift, r_shift, gte_lzcr_even_mask),
|
||||||
|
or_u(r_mag_sq_copy, r_mag_sq, 0),
|
||||||
|
li_s( r_mag_sq, 31),
|
||||||
|
sub_s( r_mag_sq, r_mag_sq, r_shift),
|
||||||
|
shift_aright(r_mag_sq, r_mag_sq, 1),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_gte_general_purpose_interopolation(AtomBuilder_R ab
|
||||||
|
, Reg to_ir0, Reg to_ir1, Reg to_ir2, Reg to_ir3
|
||||||
|
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3
|
||||||
|
, MipsCode nop_slot1, MipsCode nop_slot2)
|
||||||
|
MipsAtomComp_Proc_(ab, {
|
||||||
|
gte_mv_to_data_r(to_ir0, C2_IR0),
|
||||||
|
gte_mv_to_data_r(to_ir1, C2_IR1), /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */
|
||||||
|
gte_mv_to_data_r(to_ir2, C2_IR2),
|
||||||
|
gte_mv_to_data_r(to_ir3, C2_IR3), /* IR3 = src.z (reloaded) */
|
||||||
|
GteDelay_ nop_slot1,
|
||||||
|
GteDelay_ nop_slot2,
|
||||||
|
gte_cmdw_gpf,
|
||||||
|
gte_mv_from_data_r(fr_mac1, C2_MAC1),
|
||||||
|
gte_mv_from_data_r(fr_mac2, C2_MAC2),
|
||||||
|
gte_mv_from_data_r(fr_mac3, C2_MAC3),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_gte_mv_from_data_r_mac123(AtomBuilder_R ab
|
||||||
|
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3)
|
||||||
|
MipsAtomComp_Proc_(ab, {
|
||||||
|
gte_mv_from_data_r(fr_mac1, C2_MAC1),
|
||||||
|
gte_mv_from_data_r(fr_mac2, C2_MAC2),
|
||||||
|
gte_mv_from_data_r(fr_mac3, C2_MAC3),
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_gte_mv_from_mac123_v3s4(AtomBuilder_R ab, Reg_(V3_S4) v) MipsAtomComp_ProcMap_(ab, mac_gte_mv_from_data_r_mac123(v.x, v.y, v.z))
|
||||||
|
|
||||||
|
#pragma endregion MACs (Mips Atom Components)
|
||||||
|
|
||||||
|
#pragma region Atom Procs
|
||||||
|
|
||||||
|
/* ─── Local copy of PSYQ's sqrtbl (1/sqrt lookup table for VectorNormal). ───
|
||||||
|
* Source: PSYQ 4.7 libgte sqrtbl at 0x800185B4 in hello_camera.elf.
|
||||||
|
* objdump -s --start-address=0x800185B4 --stop-address=0x800185F4 hello_camera.elf → 192 entries × 16-bit signed, in 1.12 fixed-point (max value 0x1000 = 1.0).
|
||||||
|
*
|
||||||
|
* Data is identical to the libgte original (byte-for-byte verified).
|
||||||
|
*
|
||||||
|
* ─── Per-entry semantics (decoded from libgte msc02 VectorNormal) ───
|
||||||
|
* Each entry is `1/sqrt(x)` in 1.12 fixed point (value / 4096).
|
||||||
|
* The 192 entries span 4 octaves of the input magnitude, with 48 entries per octave:
|
||||||
|
* Octave 0 (entries 0- 47): mantissa in [0x8000, 0x10000) output ~[1.000, 0.707]
|
||||||
|
* Octave 1 (entries 48- 95): mantissa in [0x10000, 0x20000) output ~[0.707, 0.500]
|
||||||
|
* Octave 2 (entries 96-143): mantissa in [0x20000, 0x40000) output ~[0.500, 0.354]
|
||||||
|
* Octave 3 (entries144-191): mantissa in [0x40000, 0x80000) output ~[0.354, 0.251]
|
||||||
|
* Within each octave, 8 sub-entries interpolate over the 8 fractional bits of the mantissa
|
||||||
|
* (the byte `(0x80 | (i mod 8))` for the lower-byte of the aligned value).
|
||||||
|
* Sampling the first value of each octave:
|
||||||
|
* [0] 0x1000 = 1.0000 ; 1 / sqrt(1.0000)
|
||||||
|
* [48] 0x0e4f = 0.8940 ; 1 / sqrt(1.2500)
|
||||||
|
* [96] 0x0d10 = 0.8164 ; 1 / sqrt(1.5000)
|
||||||
|
* [144] 0x0c0a = 0.7520 ; 1 / sqrt(1.7500)
|
||||||
|
* And representative sub-entries within octave 0 (mantissa in [0x8000, 0x8100)):
|
||||||
|
* [0] 0x1000 = 1.0000 ; 1 / sqrt(0x8000)
|
||||||
|
* [1] 0x0fe0 = 0.9922 ; 1 / sqrt(0x8100)
|
||||||
|
* [2] 0x0fc1 = 0.9846 ; 1 / sqrt(0x8200)
|
||||||
|
* [3] 0x0fa3 = 0.9773 ; 1 / sqrt(0x8300)
|
||||||
|
* [4] 0x0f85 = 0.9700 ; 1 / sqrt(0x8400)
|
||||||
|
* [5] 0x0f68 = 0.9629 ; 1 / sqrt(0x8500)
|
||||||
|
* [6] 0x0f4c = 0.9561 ; 1 / sqrt(0x8600)
|
||||||
|
* [7] 0x0f30 = 0.9492 ; 1 / sqrt(0x8700)
|
||||||
|
*
|
||||||
|
* The algorithm's `addi -64 / sll 1 / lh` selects the entry at `(aligned - 64) * 2` for the case where `aligned` has its top bit at bit 24.
|
||||||
|
* After the sllv/srav pair, `aligned` always lands in `[0x80, 0x100)`
|
||||||
|
* (with top bit at bit 24 → after `sub $aligned - 64`, the index sits in `[0x40, 0x80) * 2 = [0x80, 0x100)` bytes = entries [64, 128) within the sqrtbl).
|
||||||
|
* The earlier 64 entries (octave 0) are reached when the magnitude after shifting puts the top bit below bit 24 (the `sllv` branch),
|
||||||
|
* and the load upper_halves of the table bracket the input range.
|
||||||
|
* The later 64 entries (octaves 2-3) are the `srav` branch when the magnitude's top bit is well above bit 24.
|
||||||
|
*
|
||||||
|
* Reproduced verbatim from libgte (verified against libpsn00b/psxgte/vector.s:100-123 — 24 rows × 8 halfwords, last entry 0x0804).
|
||||||
|
* */
|
||||||
|
internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
|
||||||
|
0x1000, 0x0fe0, 0x0fc1, 0x0fa3, 0x0f85, 0x0f68, 0x0f4c, 0x0f30,
|
||||||
|
0x0f15, 0x0efb, 0x0ee1, 0x0ec7, 0x0eae, 0x0e96, 0x0e7e, 0x0e66,
|
||||||
|
0x0e4f, 0x0e38, 0x0e22, 0x0e0c, 0x0df7, 0x0de2, 0x0dcd, 0x0db9,
|
||||||
|
0x0da5, 0x0d91, 0x0d7e, 0x0d6b, 0x0d58, 0x0d45, 0x0d33, 0x0d21,
|
||||||
|
0x0d10, 0x0cff, 0x0cee, 0x0cdd, 0x0ccc, 0x0cbc, 0x0cac, 0x0c9c,
|
||||||
|
0x0c8d, 0x0c7d, 0x0c6e, 0x0c5f, 0x0c51, 0x0c42, 0x0c34, 0x0c26,
|
||||||
|
0x0c18, 0x0c0a, 0x0bfd, 0x0bef, 0x0be2, 0x0bd5, 0x0bc8, 0x0bbb,
|
||||||
|
0x0baf, 0x0ba2, 0x0b96, 0x0b8a, 0x0b7e, 0x0b72, 0x0b67, 0x0b5b,
|
||||||
|
0x0b50, 0x0b45, 0x0b39, 0x0b2e, 0x0b24, 0x0b19, 0x0b0e, 0x0b04,
|
||||||
|
0x0af9, 0x0aef, 0x0ae5, 0x0adb, 0x0ad1, 0x0ac7, 0x0abd, 0x0ab4,
|
||||||
|
0x0aaa, 0x0aa1, 0x0a97, 0x0a8e, 0x0a85, 0x0a7c, 0x0a73, 0x0a6a,
|
||||||
|
0x0a61, 0x0a59, 0x0a50, 0x0a47, 0x0a3f, 0x0a37, 0x0a2e, 0x0a26,
|
||||||
|
0x0a1e, 0x0a16, 0x0a0e, 0x0a06, 0x09fe, 0x09f6, 0x09ef, 0x09e7,
|
||||||
|
0x09e0, 0x09d8, 0x09d1, 0x09c9, 0x09c2, 0x09bb, 0x09b4, 0x09ad,
|
||||||
|
0x09a5, 0x099e, 0x0998, 0x0991, 0x098a, 0x0983, 0x097c, 0x0976,
|
||||||
|
0x096f, 0x0969, 0x0962, 0x095c, 0x0955, 0x094f, 0x0949, 0x0943,
|
||||||
|
0x093c, 0x0936, 0x0930, 0x092a, 0x0924, 0x091e, 0x0918, 0x0912,
|
||||||
|
0x090d, 0x0907, 0x0901, 0x08fb, 0x08f6, 0x08f0, 0x08eb, 0x08e5,
|
||||||
|
0x08e0, 0x08da, 0x08d5, 0x08cf, 0x08ca, 0x08c5, 0x08bf, 0x08ba,
|
||||||
|
0x08b5, 0x08b0, 0x08ab, 0x08a6, 0x08a1, 0x089c, 0x0897, 0x0892,
|
||||||
|
0x088d, 0x0888, 0x0883, 0x087e, 0x087a, 0x0875, 0x0870, 0x086b,
|
||||||
|
0x0867, 0x0862, 0x085e, 0x0859, 0x0855, 0x0850, 0x084c, 0x0847,
|
||||||
|
0x0843, 0x083e, 0x083a, 0x0836, 0x0831, 0x082d, 0x0829, 0x0824,
|
||||||
|
0x0820, 0x081c, 0x0818, 0x0814, 0x0810, 0x080c, 0x0808, 0x0804,
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef Struct_(Binds_normalize_v3s4) {
|
||||||
|
U2 src_offset; /* offset of src V3_S4 within the BIOS scratchpad */
|
||||||
|
U2 dst_offset; /* offset of dst V3_S4 within the BIOS scratchpad */
|
||||||
|
};
|
||||||
|
typedef Struct_(RegUse_normalize_v3s4) {
|
||||||
|
union { Reg_(V3_S4) res, src; };
|
||||||
|
union { Reg r0, src_ptr, mac2; };
|
||||||
|
union { Reg r1, dst_ptr; };
|
||||||
|
union { Reg r2, dst_offset, mac1, v_sqr_aligned; };
|
||||||
|
union { Reg r3, src_offset, btarget, shift_count, sqrtbl_index; };
|
||||||
|
union { Reg r4, mac3, v_sqr_sum, scale_exp, srav_shift; };
|
||||||
|
union { Reg r5, lzcr, inv_len; };
|
||||||
|
};
|
||||||
|
/* ─── Full normalize (all 4 stages inline) ───
|
||||||
|
* Generic 4-stage GTE normalize (SQR → sum+LZCR → align+sqrtbl → GPF+srav). */
|
||||||
|
internal MipsAtom* normalize_v3s4(AtomArena_R aa, RegUse_normalize_v3s4 r)
|
||||||
|
MipsAtom_Proc_(aa, {
|
||||||
|
load_half(r.src_offset, R_TapePtr, O_(Binds_normalize_v3s4, src_offset)),
|
||||||
|
load_half(r.dst_offset, R_TapePtr, O_(Binds_normalize_v3s4, dst_offset)),
|
||||||
|
LdSlot_ add_u(r.src_ptr, R_ScratchBase, r.src_offset),
|
||||||
|
LdSlot_ add_u(r.dst_ptr, R_ScratchBase, r.dst_offset),
|
||||||
|
LdSlot_ add_ui_self(R_TapePtr, S_(Binds_normalize_v3s4)),
|
||||||
|
|
||||||
|
mac_load_v3s4(r.src, r.src_ptr, 0),
|
||||||
|
|
||||||
|
/* Stage 1: mtc2 src → IR1/2/3, SQR fires. */
|
||||||
|
LdSlot_ mac_gte_sqr_v3s4(r.src.x, r.src.y, r.src.z, LdSlot_ nop),
|
||||||
|
|
||||||
|
/* Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS. src_ptr is dead; reuse as mac2. */
|
||||||
|
mac_gte_mv_from_data_r_mac123(r.mac1, r.mac2, r.mac3), LdSlot_ nop,
|
||||||
|
add_u_self( r.v_sqr_sum, r.mac1),
|
||||||
|
add_u_self( r.v_sqr_sum, r.mac2),
|
||||||
|
gte_mv_to_data_r( r.v_sqr_sum, C2_LZCS), GteDelay_ nop2,
|
||||||
|
gte_mv_from_data_r(r.lzcr, C2_LZCR), GteDelay_ nop,
|
||||||
|
|
||||||
|
/* Stage 3: even(LZCR), half-shift, align |v|² to bit 24. */
|
||||||
|
mac_lzcr_round_even_half_shift(r.lzcr, r.v_sqr_sum, r.v_sqr_aligned),
|
||||||
|
add_si( r.btarget, r.lzcr, -24),
|
||||||
|
branch_lt_zero(r.btarget, atom_offset(aligned_done, srav_path)), BdSlot_ nop, /* bltz → srav_path (LZCR < 24 path) */
|
||||||
|
jump_rel(atom_offset(srav_path, aligned_done)), /* b → aligned_done (LZCR >= 24 path) */
|
||||||
|
BdSlot_ shift_lleft_var(r.v_sqr_aligned, r.v_sqr_aligned, r.btarget),
|
||||||
|
atom_label(srav_path)
|
||||||
|
li_s( r.shift_count, 24),
|
||||||
|
sub_s(r.shift_count, r.shift_count, r.lzcr),
|
||||||
|
shift_aright_var(r.v_sqr_aligned, r.v_sqr_aligned, r.shift_count),
|
||||||
|
atom_label(aligned_done)
|
||||||
|
add_si( r.v_sqr_aligned, r.v_sqr_aligned, -64),
|
||||||
|
shift_lleft(r.v_sqr_aligned, r.v_sqr_aligned, 1),
|
||||||
|
mac_load_word_imm(r.sqrtbl_index, & gte_normalize_sqr_tbl), add_u_self(r.sqrtbl_index, r.v_sqr_aligned),
|
||||||
|
load_half(r.inv_len, r.sqrtbl_index, 0),
|
||||||
|
LdSlot_ nop,
|
||||||
|
|
||||||
|
mac_gte_general_purpose_interopolation(r.inv_len,
|
||||||
|
r.src.x, r.src.y, r.src.z,
|
||||||
|
r.res.x, r.res.y, r.res.z,
|
||||||
|
GteDelay_ load_word(R_AtomJmp, R_TapePtr, 0), LdSlot_ // ac_yield: word 1
|
||||||
|
GteDelay_ add_ui_self( R_TapePtr, S_(MipsCode)) // ac_yield: word 2
|
||||||
|
),
|
||||||
|
mac_shift_aright_var_v3s4_self(r.res, r.srav_shift),
|
||||||
|
mac_store_v3s4(r.res, r.dst_ptr, 0),
|
||||||
|
|
||||||
|
jump_reg(R_AtomJmp), BdSlot_ nop // ac_yield: word 3-4
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
/* ─── GTE OP cross product (a × b → out) ───
|
||||||
|
* Generalized V3_S4 cross product via GTE OP (OuterProduct12 libpsyx convention).
|
||||||
|
* The >> 12 shift converts S12.20 → S12.0 OuterProduct12. */
|
||||||
|
typedef Struct_(Binds_gte_cross_v3s4) { V3_S4* src_a; V3_S4* src_b; V3_S4* out; };
|
||||||
|
typedef Struct_(RegUse_gte_cross_v3s4) {
|
||||||
|
Reg_(V3_S4) a;
|
||||||
|
Reg_(V3_S4) b;
|
||||||
|
union { Reg out, t0; } x;
|
||||||
|
union { Reg src_a, t1, rt11; } y;
|
||||||
|
union { Reg src_b, t2, rt22; } z;
|
||||||
|
};
|
||||||
|
internal MipsAtom* gte_cross_v3s4(AtomArena_R aa, RegUse_gte_cross_v3s4 r)
|
||||||
|
atom_info(atom_bind(Binds_gte_cross_v3s4)) MipsAtom_Proc_(aa, {
|
||||||
|
load_word(r.y.src_a, R_TapePtr, O_(Binds_gte_cross_v3s4,src_a)),
|
||||||
|
load_word(r.z.src_b, R_TapePtr, O_(Binds_gte_cross_v3s4,src_b)),
|
||||||
|
load_word(r.x.out, R_TapePtr, O_(Binds_gte_cross_v3s4,out)),
|
||||||
|
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_gte_cross_v3s4)),
|
||||||
|
|
||||||
|
mac_load_v3s4(r.a, r.y.src_a, 0), LdSlot_
|
||||||
|
mac_load_v3s4(r.b, r.z.src_b, 0), LdSlot_
|
||||||
|
mac_gte_op_cross_v3s4(r.a, r.b), /* RT diagonal + IR + OP + MAC read + shift */
|
||||||
|
mac_store_v3s4(r.a, r.x.out, 0),
|
||||||
|
|
||||||
|
mac_yield()
|
||||||
|
})
|
||||||
|
#pragma endregion Atom Procs
|
||||||
|
|
||||||
|
#pragma region Baked Atoms
|
||||||
|
|
||||||
|
typedef Struct_(Binds_SetGteMT3S2S4) {
|
||||||
|
MT3_S2S4* transform;
|
||||||
|
};
|
||||||
|
internal MipsAtom_(set_gte_mt3s2s4) atom_info(
|
||||||
|
atom_bind(Binds_SetGteMT3S2S4)
|
||||||
|
, atom_reads(R_TapePtr)
|
||||||
|
){
|
||||||
|
/* Pop matrix address from tape into R_T3 ($11) */
|
||||||
|
load_word(R_T3, R_TapePtr, O_(Binds_SetGteMT3S2S4,transform)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_SetGteMT3S2S4)),
|
||||||
|
/* Load 3x3 Rotation + 3x1 Translation from R_T3 into GTE CONTROL Regs (ctc2) */
|
||||||
|
load_word(R_T0, R_T3, 0),
|
||||||
|
load_word(R_T1, R_T3, 4),
|
||||||
|
gte_mv_to_ctrl_r(R_T0, gte_cr_RT11),
|
||||||
|
gte_mv_to_ctrl_r(R_T1, gte_cr_RT12),
|
||||||
|
load_word(R_T0, R_T3, 8),
|
||||||
|
load_word(R_T1, R_T3, 12),
|
||||||
|
load_word(R_T2, R_T3, 16),
|
||||||
|
gte_mv_to_ctrl_r(R_T0, gte_cr_RT13),
|
||||||
|
gte_mv_to_ctrl_r(R_T1, gte_cr_RT21),
|
||||||
|
gte_mv_to_ctrl_r(R_T2, gte_cr_RT22),
|
||||||
|
load_word(R_T0, R_T3, 20),
|
||||||
|
load_word(R_T1, R_T3, 24),
|
||||||
|
load_word(R_T2, R_T3, 28),
|
||||||
|
gte_mv_to_ctrl_r(R_T0, gte_cr_TRX),
|
||||||
|
gte_mv_to_ctrl_r(R_T1, gte_cr_TRY),
|
||||||
|
gte_mv_to_ctrl_r(R_T2, gte_cr_TRZ),
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
#pragma endregion Baked Atoms
|
||||||
+288
-250
@@ -1,4 +1,4 @@
|
|||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
* duffle DSL Suffix Conventions
|
* duffle DSL Suffix Conventions
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
*
|
*
|
||||||
@@ -16,10 +16,6 @@
|
|||||||
* gte_mv_to_data_r (gte + mv + to + data + register)
|
* gte_mv_to_data_r (gte + mv + to + data + register)
|
||||||
* gte_lw_v0_xy(base) (gte + lw + v0 + xy)
|
* gte_lw_v0_xy(base) (gte + lw + v0 + xy)
|
||||||
* load_upper_i (load-upper + immediate, unique verb)
|
* load_upper_i (load-upper + immediate, unique verb)
|
||||||
*
|
|
||||||
* Vendor mnemonics (gte_mtc2, gte_mfc2, gte_lwc2, gte_swc2, etc.) are
|
|
||||||
* NOT in this header. They live in the opt-in `gte_vendor_sym.h` for
|
|
||||||
* users who prefer the textbook MIPS assembly mnemonics.
|
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
|
|
||||||
#ifdef INTELLISENSE_DIRECTIVES
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
@@ -34,76 +30,27 @@
|
|||||||
* gte.h — Geometry Transformation Engine (COP2) for the PS1
|
* gte.h — Geometry Transformation Engine (COP2) for the PS1
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
*
|
*
|
||||||
* Hand-rolled DSL for emitting GTE/MIPS instruction words as raw `.word`
|
* Hand-rolled DSL for emitting GTE/MIPS instruction words from C.
|
||||||
* constants from C. No GCC inline-assembly string syntax in the code body.
|
* No GCC inline-assembly string syntax in the code body.
|
||||||
*
|
|
||||||
* PHILOSOPHY
|
|
||||||
* ----------
|
|
||||||
* 1. A 32-bit instruction word is composed from per-field encoders. Each
|
|
||||||
* encoder knows only its own bit range; the composite ORs them together.
|
|
||||||
* No magic numbers inside any encoder body. Every shift and mask is a
|
|
||||||
* named constant from the bitfield-layout enum below.
|
|
||||||
*
|
|
||||||
* 2. Pure (compile-time) instructions. Every GTE *command* (RTPS, RTPT,
|
|
||||||
* NCLIP, MVMVA, …) and every COP2 *transfer* (ctc2/cfc2) with a constant
|
|
||||||
* rs/rt/rd — are emitted as a single integer constant via
|
|
||||||
* `asm_inline(...)` from gcc_asm.h. The C compiler constant-folds
|
|
||||||
* these into `.word` directives in .rodata.
|
|
||||||
*
|
|
||||||
* 3. Runtime-base-register instructions (lwc2, swc2, lw, sw, …) cannot be
|
|
||||||
* a pure compile-time word because the `rs` field is chosen by the
|
|
||||||
* compiler at codegen. For these we use a "placeholder-pun" pattern:
|
|
||||||
* a fixed register number (R_T4 = $12) is baked into the rs field of
|
|
||||||
* the `.word` constant, and the macro declares a `"r"(arg)` input
|
|
||||||
* constraint plus a clobber on the same register. The compiler is
|
|
||||||
* therefore *forced* to bind `arg` to that exact register, and the
|
|
||||||
* constant is correct.
|
|
||||||
*
|
|
||||||
* USAGE
|
|
||||||
* -----
|
|
||||||
* // Pure command sequence — all bits compile-time:
|
|
||||||
* asm volatile(
|
|
||||||
* asm_inline( gte_cmd_rtpt , gte_cmd_nclip , gte_cmd_avsz3 )
|
|
||||||
* asm_clobber( clbr_volatile_gprs )
|
|
||||||
* );
|
|
||||||
*
|
|
||||||
* // Runtime-base-register load — caller picks the base GPR:
|
|
||||||
* register V3_S2* p_in_12 __asm__("$12") = verts[0].ptr;
|
|
||||||
* gte_load_v0(p_in_12, R_T4); // R_T4 = 12 = $t4 = $12
|
|
||||||
*
|
|
||||||
* // Three independent bases for an RTPT pipeline:
|
|
||||||
* register V3_S2* p0 gcc_reg(R_T4) = verts[0].ptr;
|
|
||||||
* register V3_S2* p1 gcc_reg(R_T5) = verts[1].ptr;
|
|
||||||
* register V3_S2* p2 gcc_reg(R_T6) = verts[2].ptr;
|
|
||||||
* gte_load_v0(p0, R_T4);
|
|
||||||
* gte_load_v1(p1, R_T5);
|
|
||||||
* gte_load_v2(p2, R_T6);
|
|
||||||
* gte_rtpt();
|
|
||||||
*
|
*
|
||||||
* STYLE NOTES
|
* STYLE NOTES
|
||||||
* -----------
|
* -----------
|
||||||
* - Per-field encoders are named `enc_gte_<field>(value)` and each one
|
* - Per-field encoders are named `enc_gte_<field>(value)` and each one self-masks its argument before shifting.
|
||||||
* self-masks its argument before shifting. Mirrors the `enc_op / enc_rs
|
* Mirrors the `enc_op / enc_rs / enc_rt / ...` family in mips.h.
|
||||||
* / enc_rt / ...` family in mips.h.
|
* - The composite `enc_gte_cmdw(sf, mx, v, cv, lm, cmd)` is a flat OR of the per-field encoders, plus the COP2/CO base.
|
||||||
* - The composite `enc_gte_cmdw(sf, mx, v, cv, lm, cmd)` is a flat OR of
|
* - Pre-baked shortcuts (`gte_cmd_rtpt`, `gte_cmd_rtps`, …) are defined for the common cases so call sites read like assembly source.
|
||||||
* the per-field encoders, plus the COP2/CO base.
|
* - All register/field values are enums (not `#define`s) so they show up in debugger symbol tables and IDE autocomplete.
|
||||||
* - Pre-baked shortcuts (`gte_cmd_rtpt`, `gte_cmd_rtps`, …) are defined
|
|
||||||
* for the common cases so call sites read like assembly source.
|
|
||||||
* - All register/field values are enums (not `#define`s) so they show up
|
|
||||||
* in debugger symbol tables and IDE autocomplete.
|
|
||||||
*
|
*
|
||||||
* SEE ALSO
|
* SEE ALSO
|
||||||
* --------
|
* --------
|
||||||
* - gcc_asm.h: the `.word` emitter (`asm_inline`, `asm_clobber`, clobbers)
|
* - mips.h: The MIPS encoder layer this builds on.
|
||||||
* - mips.h: the MIPS encoder layer this builds on
|
|
||||||
*/
|
*/
|
||||||
|
|
||||||
/* C2 data registers */
|
/* C2 data registers */
|
||||||
|
|
||||||
/* --- GTE Data Registers (Coprocessor 2) ---
|
/* --- GTE Data Registers (Coprocessor 2) ---
|
||||||
* Preprocessor-visible integer ids for the COP2 data register file.
|
* Preprocessor-visible integer ids for the COP2 data register file.
|
||||||
* Each enum value is bound to a parallel `_Code` `#define` so the
|
* Each enum value is bound to a parallel `_Code` `#define` so the preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
||||||
* preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
|
||||||
* Same pattern as the GPR `_Code` set in mips.h. */
|
* Same pattern as the GPR `_Code` set in mips.h. */
|
||||||
#define C2_VXY0_Code 0
|
#define C2_VXY0_Code 0
|
||||||
#define C2_VZ0_Code 1
|
#define C2_VZ0_Code 1
|
||||||
@@ -151,20 +98,20 @@ enum {
|
|||||||
|
|
||||||
/* Semantic Aliases for GTE Data Registers */
|
/* Semantic Aliases for GTE Data Registers */
|
||||||
enum {
|
enum {
|
||||||
gte_in_v0_xy = C2_VXY0, /* Input Vector 0 (X, Y) */
|
C2_InV0_XY = C2_VXY0, /* Input Vector 0 (X, Y) */
|
||||||
gte_in_v0_z = C2_VZ0, /* Input Vector 0 (Z) */
|
C2_InV0_Z = C2_VZ0, /* Input Vector 0 (Z) */
|
||||||
gte_in_v1_xy = C2_VXY1, /* Input Vector 1 (X, Y) */
|
C2_InV1_XY = C2_VXY1, /* Input Vector 1 (X, Y) */
|
||||||
gte_in_v1_z = C2_VZ1, /* Input Vector 1 (Z) */
|
C2_InV1_Z = C2_VZ1, /* Input Vector 1 (Z) */
|
||||||
gte_in_v2_xy = C2_VXY2, /* Input Vector 2 (X, Y) */
|
C2_InV2_XY = C2_VXY2, /* Input Vector 2 (X, Y) */
|
||||||
gte_in_v2_z = C2_VZ2, /* Input Vector 2 (Z) */
|
C2_InV2_Z = C2_VZ2, /* Input Vector 2 (Z) */
|
||||||
gte_in_rgb = C2_RGB, /* Input Color (R, G, B, MipsCode) */
|
C2_In_RGB = C2_RGB, /* Input Color (R, G, B, MipsCode) */
|
||||||
gte_out_scr_xy0 = C2_SXY0, /* Output Screen Coord 0 (X, Y) */
|
C2_OutSrc_XY0 = C2_SXY0, /* Output Screen Coord 0 (X, Y) */
|
||||||
gte_out_scr_xy1 = C2_SXY1, /* Output Screen Coord 1 (X, Y) */
|
C2_OutSrc_XY1 = C2_SXY1, /* Output Screen Coord 1 (X, Y) */
|
||||||
gte_out_scr_xy2 = C2_SXY2, /* Output Screen Coord 2 (X, Y) */
|
C2_OutSrc_XY2 = C2_SXY2, /* Output Screen Coord 2 (X, Y) */
|
||||||
gte_out_depth = C2_OTZ, /* Output Ordering Table Z (Depth) */
|
C2_OutDepth = C2_OTZ, /* Output Ordering Table Z (Depth) */
|
||||||
gte_math_accum0 = C2_MAC0, /* Math Accumulator 0 */
|
C2_MathAccu0 = C2_MAC0, /* Math Accumulator 0 */
|
||||||
gte_math_accum1 = C2_MAC1, /* Math Accumulator 1 */
|
C2_MathAccu1 = C2_MAC1, /* Math Accumulator 1 */
|
||||||
gte_math_accum2 = C2_MAC2, /* Math Accumulator 2 */
|
C2_MathAccu2 = C2_MAC2, /* Math Accumulator 2 */
|
||||||
};
|
};
|
||||||
|
|
||||||
/* --- GTE Command Semantics (The Bitfield Meanings) ---
|
/* --- GTE Command Semantics (The Bitfield Meanings) ---
|
||||||
@@ -211,6 +158,8 @@ enum {
|
|||||||
gte_cmd_nclip = 0x06, /* Normal Clipping (Backface culling) */
|
gte_cmd_nclip = 0x06, /* Normal Clipping (Backface culling) */
|
||||||
gte_cmd_op = 0x0C, /* Outer Product */
|
gte_cmd_op = 0x0C, /* Outer Product */
|
||||||
gte_cmd_mvmva = 0x12, /* Matrix Vector Multiply & Add (Custom math) */
|
gte_cmd_mvmva = 0x12, /* Matrix Vector Multiply & Add (Custom math) */
|
||||||
|
gte_cmd_sqr = 0x28, /* Square vector — MAC[i] = IR[i]²; IR[i] ← MAC[i] saturated */
|
||||||
|
gte_cmd_gpf = 0x3D, /* General-purpose Interpolation — MAC[i] = IR0 * IR[i] */
|
||||||
|
|
||||||
/* --- GTE Command Bit-Field Layout ---
|
/* --- GTE Command Bit-Field Layout ---
|
||||||
* A GTE command word (sent to COP2 with RS=1) is laid out as:
|
* A GTE command word (sent to COP2 with RS=1) is laid out as:
|
||||||
@@ -221,25 +170,46 @@ enum {
|
|||||||
* +------------+--+-----+------+------+------+------+---+--------+----------+
|
* +------------+--+-----+------+------+------+------+---+--------+----------+
|
||||||
* \_____ GTE_PAYLOAD _____/ \__ GTE_CMD __/
|
* \_____ GTE_PAYLOAD _____/ \__ GTE_CMD __/
|
||||||
*
|
*
|
||||||
* Shifts/masks below are the *bit positions* and *bit widths* of each
|
* Offset position & masks below are the *bit positions* and *bit widths* of each configurable field, used by the ENC_GTE_CMD encoder.
|
||||||
* configurable field, used by the ENC_GTE_CMD encoder. Mirrors the
|
* Mirrors the OPCODE_POS / RS_POS convention used in mips.h.
|
||||||
* OPCODE_SHIFT / RS_SHIFT convention used in mips.h.
|
|
||||||
*/
|
*/
|
||||||
|
|
||||||
gte_shift_sf = 19, gte_width_sf = 1, gte_mask_sf = 0x1,
|
gte_pos_sf = 19, gte_width_sf = 1,
|
||||||
gte_shift_mx = 17, gte_width_mx = 2, gte_mask_mx = 0x3,
|
gte_pos_mx = 17, gte_width_mx = 2,
|
||||||
gte_shift_v = 15, gte_width_v = 2, gte_mask_v = 0x3,
|
gte_pos_v = 15, gte_width_v = 2,
|
||||||
gte_shift_cv = 13, gte_width_cv = 2, gte_mask_cv = 0x3,
|
gte_pos_cv = 13, gte_width_cv = 2,
|
||||||
gte_shift_lm = 10, gte_width_lm = 1, gte_mask_lm = 0x1,
|
gte_pos_lm = 10, gte_width_lm = 1,
|
||||||
gte_shift_cmd = 0, gte_width_cmd = 6, gte_mask_cmd = 0x3F,
|
gte_pos_cmd = 0, gte_width_cmd = 6,
|
||||||
|
|
||||||
|
/* Fake command number (bits 24-20) — IGNORED by the GTE hardware per PSX-SPX `geometrytransformationenginegte.md` line 48.
|
||||||
|
* libgte's compiler emits non-zero values in this field as a disassembly signature. */
|
||||||
|
gte_pos_fake_cmd = 20,
|
||||||
|
gte_width_fake_cmd = 5,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/* --- GTE Control Register Aliases (Pitfall 1) ---
|
||||||
|
* Three pairs of aliases map to the C2 control-register slot:
|
||||||
|
* C2[24] = gte_cr_RBK (background R) | gte_cr_OFX (screen offset X)
|
||||||
|
* C2[25] = gte_cr_GBK (background G) | gte_cr_OFY (screen offset Y)
|
||||||
|
* C2[26] = gte_cr_BBK (background B) | gte_cr_H (projection plane distance H)
|
||||||
|
* Cross-alias writes inside one atom body, or across the wave-context boundary, silently clobber each other.
|
||||||
|
* The metaprogram's check_gte_cr_alias_writes (CHECK_RULES row) warns about each pair per source.
|
||||||
|
* See psx-spx docs/gte_reference.md §"Control-register alias table" for the silicon rationale and the libgte outer-product convention.
|
||||||
|
*/
|
||||||
|
|
||||||
|
/* --- RT-matrix packed-slot convention (Pitfall 4) ---
|
||||||
|
* The silicon packs two 16-bit RT elements per 32-bit C2 slot:
|
||||||
|
* C2[2] = (RT22 << 16) | RT13 (gte_cr_RT13 writes the low half, gte_cr_RT22 writes the high half)
|
||||||
|
* C2[4] = (RT33 << 16) | RT22 (gte_cr_RT22 writes the low half — clobbers prior RT22 value if RT13 was also written)
|
||||||
|
* OP and MVMVA read D1/D2/D3 from these packed slots.
|
||||||
|
* The libgte outer-product convention (see ac_apply_matrix_lv at gte.atom.c:108-122) writes C2[2] then C2[4] in sequence;
|
||||||
|
* the SECOND write's low half is RT22, not RT13.
|
||||||
|
*/
|
||||||
|
|
||||||
/* --- GTE Control Register Indices (for ctc2/cfc2) ---
|
/* --- GTE Control Register Indices (for ctc2/cfc2) ---
|
||||||
* Preprocessor-visible integer ids for the COP2 control register file.
|
* Preprocessor-visible integer ids for the COP2 control register file.
|
||||||
* Each enum value is bound to a parallel `_Code` `#define` so the
|
* Each enum value is bound to a parallel `_Code` `#define` so the preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
||||||
* preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
* Same pattern as the GPR `_Code` set in mips.h. Note: indices 21-23 are reserved/unused on real hardware, so there's a gap. */
|
||||||
* Same pattern as the GPR `_Code` set in mips.h. Note: indices 21-23
|
|
||||||
* are reserved/unused on real hardware, so there's a gap. */
|
|
||||||
#define gte_cr_RT11_Code 0
|
#define gte_cr_RT11_Code 0
|
||||||
#define gte_cr_RT12_Code 1 /* packed with RT13 in bits 16..31 */
|
#define gte_cr_RT12_Code 1 /* packed with RT13 in bits 16..31 */
|
||||||
#define gte_cr_RT13_Code 2 /* packed with RT22 in bits 16..31 */
|
#define gte_cr_RT13_Code 2 /* packed with RT22 in bits 16..31 */
|
||||||
@@ -267,8 +237,9 @@ enum {
|
|||||||
#define gte_cr_RFC_Code 27
|
#define gte_cr_RFC_Code 27
|
||||||
#define gte_cr_GFC_Code 28
|
#define gte_cr_GFC_Code 28
|
||||||
#define gte_cr_BFC_Code 29
|
#define gte_cr_BFC_Code 29
|
||||||
#define gte_cr_OFX_Code 30
|
#define gte_cr_OFX_Code 24
|
||||||
#define gte_cr_OFY_Code 31
|
#define gte_cr_OFY_Code 25
|
||||||
|
#define gte_cr_H_Code 26
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
gte_cr_RT11 = gte_cr_RT11_Code, gte_cr_RT12 = gte_cr_RT12_Code, gte_cr_RT13 = gte_cr_RT13_Code,
|
gte_cr_RT11 = gte_cr_RT11_Code, gte_cr_RT12 = gte_cr_RT12_Code, gte_cr_RT13 = gte_cr_RT13_Code,
|
||||||
@@ -284,34 +255,66 @@ enum {
|
|||||||
};
|
};
|
||||||
|
|
||||||
enum { _C2_OPS_ = 0
|
enum { _C2_OPS_ = 0
|
||||||
|
|
||||||
, op_lwc2 = 0x32 /* Load Word to Coprocessor 2 (GTE) */
|
, op_lwc2 = 0x32 /* Load Word to Coprocessor 2 (GTE) */
|
||||||
, op_swc2 = 0x3A /* Store Word from Coprocessor 2 (GTE) */
|
, op_swc2 = 0x3A /* Store Word from Coprocessor 2 (GTE) */
|
||||||
};
|
};
|
||||||
|
|
||||||
/* COP2 (GTE) Transfer Format: ctc2 rt, rd or cfc2 rt, rd
|
/* COP2 transfer sub-opcodes (5-bit field in the `rs` slot of enc_gte_tx).
|
||||||
|
*
|
||||||
|
* Spans the 2x2 {From, To} × {Data, Control} register classes that the GTE exposes:
|
||||||
|
* bit 1 (0x02): register class — 0 = data, 1 = control
|
||||||
|
* bit 2 (0x04): direction — 0 = read, 1 = write
|
||||||
|
*
|
||||||
|
* The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit numbers as general MIPS `cop_mf` / `cop_mt` defined in mips.h
|
||||||
|
* (which target the data register file on any coprocessor).
|
||||||
|
* They are re-aliased here so the four-way table reads like the spec mnemonics (MFC2 / CFC2 / MTC2 / CTC2)
|
||||||
|
* and so the encoding is next to its only consumer (this header).
|
||||||
|
*
|
||||||
|
* Vendor mnemonic aliases (gte_mfc2 / gte_mtc2 / gte_cfc2 / gte_ctc2) live in gte_vendor_sym.h. */
|
||||||
|
enum { _C2_TX_SUBS_ = 0
|
||||||
|
, sub_mfc2 = 0x00 /* MFC2: Move From Coprocessor 2 data reg */
|
||||||
|
, sub_cfc2 = 0x02 /* CFC2: Copy From Coprocessor 2 ctrl reg */
|
||||||
|
, sub_mtc2 = 0x04 /* MTC2: Move To Coprocessor 2 data reg */
|
||||||
|
, sub_ctc2 = 0x06 /* CTC2: Copy To Coprocessor 2 ctrl reg */
|
||||||
|
};
|
||||||
|
|
||||||
|
/* COP2 (GTE) Transfer Format: mfc2 / cfc2 / mtc2 / ctc2 rt, rd
|
||||||
* Layout: [op_cop2:6][sub:5][rt:5][rd:5][0:11]
|
* Layout: [op_cop2:6][sub:5][rt:5][rd:5][0:11]
|
||||||
* - sub: cop_mf (0x00) for cfc2, cop_mt (0x04) for ctc2
|
* - sub: one of sub_mfc2 / sub_cfc2 / sub_mtc2 / sub_ctc2
|
||||||
* - rt: GPR source/dest
|
* - rt: GPR source/dest
|
||||||
* - rd: COP2 control register index (0..31) */
|
* - rd: COP2 register index (0..31):
|
||||||
|
* data class → C2_VXY0_Code..C2_LZCR_Code (gte_in_v0_xy..gte_math_accum2 aliases)
|
||||||
|
* ctrl class → gte_cr_RT11_Code..gte_cr_OFY_Code */
|
||||||
#define enc_gte_tx(sub, rt, rd) (enc_op(op_cop2) | enc_rs(sub) | enc_rt(rt) | enc_rd(rd))
|
#define enc_gte_tx(sub, rt, rd) (enc_op(op_cop2) | enc_rs(sub) | enc_rt(rt) | enc_rd(rd))
|
||||||
|
|
||||||
|
|
||||||
// #define gte_mv_to_data_r(rt, rd) enc_gte_tx(cop_mt, (rt), (rd)) /* Move GPR (rt) to GTE Control Register (rd) */
|
// #define gte_mv_to_data_r(rt, rd) enc_gte_tx(cop_mt, (rt), (rd)) /* Move GPR (rt) to GTE Control Register (rd) */
|
||||||
// #define gte_mv_from_data_r(rt, rd) enc_gte_tx(cop_mf, (rt), (rd)) /* Move GTE Control Register (rd) to GPR (rt) */
|
// #define gte_mv_from_data_r(rt, rd) enc_gte_tx(cop_mf, (rt), (rd)) /* Move GTE Control Register (rd) to GPR (rt) */
|
||||||
|
|
||||||
/* GTE Data vs Control Register Transfers */
|
/* GTE Data vs Control Register Transfers
|
||||||
#define gte_mv_from_data_r(rt, rd) enc_gte_tx(0x00, (rt), (rd)) /* Move from GTE Data Reg (e.g. MAC0, OTZ) */
|
* Each macro emits a single instruction for one of MFC2/CFC2/MTC2/CTC2.
|
||||||
#define gte_mv_from_ctrl_r(rt, rd) enc_gte_tx(0x02, (rt), (rd)) /* Move from GTE Control Reg */
|
*
|
||||||
#define gte_mv_to_data_r(rt, rd) enc_gte_tx(0x04, (rt), (rd)) /* Move to GTE Data Reg (e.g. VXY0) */
|
* `rd` is the C2 register index in the file the sub-opcode names:
|
||||||
#define gte_mv_to_ctrl_r(rt, rd) enc_gte_tx(0x06, (rt), (rd)) /* Move to GTE Control Reg (e.g. Matrices) */
|
* gte_mv_from_data_r / gte_mv_to_data_r → C2 data register file
|
||||||
|
* gte_mv_from_ctrl_r / gte_mv_to_ctrl_r → C2 ctrl register file
|
||||||
|
*
|
||||||
|
* Common pairs:
|
||||||
|
* gte_mv_from_data_r(R_T0, C2_MAC0) — read MAC0 into a GPR
|
||||||
|
* gte_mv_to_data_r (R_V0, C2_VXY0) — write GPR into VXY0
|
||||||
|
* gte_mv_to_ctrl_r (R_T0, gte_cr_RT11) — write GPR into rotation matrix
|
||||||
|
* gte_mv_from_ctrl_r(R_T0, gte_cr_OFX) — read screen-X offset */
|
||||||
|
#define gte_mv_from_data_r(rt, rd) enc_gte_tx(sub_mfc2, (rt), (rd)) /* Move From data reg */
|
||||||
|
#define gte_mv_from_ctrl_r(rt, rd) enc_gte_tx(sub_cfc2, (rt), (rd)) /* Copy From ctrl reg */
|
||||||
|
#define gte_mv_to_data_r(rt, rd) enc_gte_tx(sub_mtc2, (rt), (rd)) /* Move To data reg */
|
||||||
|
#define gte_mv_to_ctrl_r(rt, rd) enc_gte_tx(sub_ctc2, (rt), (rd)) /* Copy To ctrl reg */
|
||||||
|
#define GteDelay_ // Annotate an instruction as filling a CPU <-> GTE DMA delay slot/s
|
||||||
|
|
||||||
/* COP2 Data Load (lwc2): `lwc2 rt, off(rs)`
|
/* COP2 Data Load (lwc2): `lwc2 rt, off(rs)`
|
||||||
* Layout: [op_lwc2:6][rs:5][rt:5][imm:16]
|
* Layout: [op_lwc2:6][rs:5][rt:5][imm:16]
|
||||||
* - rs: GPR base address
|
* - rs: GPR base address
|
||||||
* - rt: COP2 data register index (0..31)
|
* - rt: COP2 data register index (0..31)
|
||||||
* - imm: signed 16-bit offset
|
* - imm: signed 16-bit offset
|
||||||
* NOTE: When `rs` is a runtime register, the encoding cannot be pre-baked
|
* NOTE: When `rs` is a runtime register, the encoding cannot be pre-baked into a .word — use the string-style `gte_load_v0` macro below instead. */
|
||||||
* into a .word — use the string-style `gte_load_v0` macro below instead. */
|
|
||||||
#define enc_gte_lw(rt, base, off) enc_i(op_lwc2, (base), (rt), (off))
|
#define enc_gte_lw(rt, base, off) enc_i(op_lwc2, (base), (rt), (off))
|
||||||
/* Store Word */
|
/* Store Word */
|
||||||
#define enc_gte_sw(rt, base, off) enc_i(op_swc2, (base), (rt), (off))
|
#define enc_gte_sw(rt, base, off) enc_i(op_swc2, (base), (rt), (off))
|
||||||
@@ -320,31 +323,30 @@ enum { _C2_OPS_ = 0
|
|||||||
* `swc2` is redundant when we're already inside the `gte_` namespace.
|
* `swc2` is redundant when we're already inside the `gte_` namespace.
|
||||||
* gte_lw rt, base, off → lwc2 rt, off(base)
|
* gte_lw rt, base, off → lwc2 rt, off(base)
|
||||||
* gte_sw rt, base, off → swc2 rt, off(base)
|
* gte_sw rt, base, off → swc2 rt, off(base)
|
||||||
* For the typical user-facing vector-level load (xy + z as two
|
* For the typical user-facing vector-level load (xy + z as two instructions), use the higher-level `gte_load_vN` macros below. */
|
||||||
* instructions), use the higher-level `gte_load_vN` macros below. */
|
|
||||||
#define gte_lw(rt, base, off) enc_gte_lw(rt, base, off)
|
#define gte_lw(rt, base, off) enc_gte_lw(rt, base, off)
|
||||||
#define gte_sw(rt, base, off) enc_gte_sw(rt, base, off)
|
#define gte_sw(rt, base, off) enc_gte_sw(rt, base, off)
|
||||||
|
|
||||||
/* GTE Command Format
|
/* GTE Command Format
|
||||||
* Opcode is always MIPS_OP_COP2, RS is always 1 (CO).
|
* Opcode is always MIPS_OP_COP2, RS is always 1 (CO).
|
||||||
* The lower 25 bits are the GTE-specific command payload.
|
* Lower 25 bits are GTE-specific command payload.
|
||||||
*
|
*
|
||||||
* The granular `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs`
|
* The `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs` pattern in mips.h:
|
||||||
* pattern in mips.h: each one self-masks and shifts its own field, so a
|
* Each one self-masks and shifts its own field, so a caller can build up a GTE command piece by piece
|
||||||
* caller can build up a GTE command piece by piece (handy for state-driven
|
* (handy for state-driven MVMVA emitters that vary one field at a time).
|
||||||
* MVMVA emitters that vary one field at a time).
|
|
||||||
*
|
*
|
||||||
* `ENC_GTE_CMD` is the all-in-one convenience for emitting a full command
|
* `ENC_GTE_CMD` is an all-in-one convenience for emitting a full command word.
|
||||||
* word in one go. It just ORs the per-field encoders together. */
|
* It just ORs the per-field encoders together. */
|
||||||
#define gte_cmd_base (enc_op(op_cop2) | (1 << 25))
|
#define gte_cmd_base (enc_op(op_cop2) | (1 << 25))
|
||||||
|
|
||||||
/* Per-field encoders. Each one does (value & mask) << shift on its own. */
|
/* Per-field encoders. Each one does (value & mask) << shift on its own. */
|
||||||
#define enc_gte_sf(sf) (((sf) & gte_mask_sf ) << gte_shift_sf )
|
#define enc_gte_sf(sf) ((sf) << gte_pos_sf )
|
||||||
#define enc_gte_mx(mx) (((mx) & gte_mask_mx ) << gte_shift_mx )
|
#define enc_gte_mx(mx) ((mx) << gte_pos_mx )
|
||||||
#define enc_gte_v(v) (((v) & gte_mask_v ) << gte_shift_v )
|
#define enc_gte_v(v) ((v) << gte_pos_v )
|
||||||
#define enc_gte_cv(cv) (((cv) & gte_mask_cv ) << gte_shift_cv )
|
#define enc_gte_cv(cv) ((cv) << gte_pos_cv )
|
||||||
#define enc_gte_lm(lm) (((lm) & gte_mask_lm ) << gte_shift_lm )
|
#define enc_gte_lm(lm) ((lm) << gte_pos_lm )
|
||||||
#define enc_gte_cmd(cmd) (((cmd) & gte_mask_cmd) << gte_shift_cmd)
|
#define enc_gte_cmd(cmd) ((cmd) << gte_pos_cmd )
|
||||||
|
#define enc_gte_fake_cmd(x) ((x) << gte_pos_fake_cmd)
|
||||||
|
|
||||||
/* Composite: all six GTE fields + the COP2/CO base. */
|
/* Composite: all six GTE fields + the COP2/CO base. */
|
||||||
#define enc_gte_cmdw(sf, mx, v, cv, lm, cmd) ( \
|
#define enc_gte_cmdw(sf, mx, v, cv, lm, cmd) ( \
|
||||||
@@ -359,41 +361,35 @@ enum { _C2_OPS_ = 0
|
|||||||
|
|
||||||
/* GTE command words for the common cases.
|
/* GTE command words for the common cases.
|
||||||
*
|
*
|
||||||
* These are pure compile-time integer constants — the C compiler
|
* These are pure compile-time integer constants — the C compiler constant-folds them into `.word` directives in .rodata.
|
||||||
* constant-folds them into `.word` directives in .rodata. Use them
|
* Use them inside `asm_inline(...)` blocks (see `gte_rtpt` below for the idiom).
|
||||||
* inside `asm_inline(...)` blocks (see `gte_rtpt` below for the
|
|
||||||
* canonical idiom).
|
|
||||||
*
|
*
|
||||||
* Decomposition (per the `enc_gte_<field>` definitions above):
|
* Decomposition (per the `enc_gte_<field>` definitions above):
|
||||||
* gte_cmdw_<name> = gte_cmd_base | enc_gte_cmd(<cmd>)
|
* gte_cmdw_<name> = gte_cmd_base | enc_gte_cmd(<cmd>)
|
||||||
|
|
||||||
* The SF/MX/V/CV/LM fields are all zero in the common cases (standard
|
* The SF / MX / V / CV / LM fields are all zero in the common cases
|
||||||
* rotation-matrix, no scaling factor, V0 vector, translation vector,
|
* (standard rotation-matrix, no scaling factor, V0 vector, translation vector, no clamp),
|
||||||
* no clamp), so the only varying bits are the `cmd` field.
|
* so the only varying bits are the `cmd` field.
|
||||||
*
|
*
|
||||||
* Naming follows the file's convention: `gte_cmd_*` is the raw
|
* Naming convention:
|
||||||
* 6-bit `cmd` field id, `gte_cmdw_*` is the fully-encoded 32-bit
|
* - `gte_cmd_*` : Raw 6-bit `cmd` field id
|
||||||
* instruction word ready to drop into a `.word` directive.
|
* - `gte_cmdw_* : 32-bit instruction word ready to drop into a `.word` directive.
|
||||||
*
|
*
|
||||||
* --------------------------------------------------------------------------
|
* --------------------------------------------------------------------------
|
||||||
* PsyQ-compatibility note (RTPS/RTPT):
|
* PsyQ-compatibility note (RTPS/RTPT):
|
||||||
* The original Sony PsyQ `inline_n.h` ships RTPT as `cop2 0x0280030` and
|
* The original Sony PsyQ `inline_n.h` ships RTPT as `cop2 0x0280030` and RTPS as `cop2 0x0180001`.
|
||||||
* RTPS as `cop2 0x0180001`. Both have `0x20` set in the upper-reserved
|
* Both have `0x20` set in the upper-reserved region (bit 21) AND `sf=1` (bit 19) — i.e. the "no division" flag.
|
||||||
* region (bit 21) AND `sf=1` (bit 19) — i.e. the "no division" flag.
|
* Per psx-spec these bits are reserved/must-be-zero,
|
||||||
* Per psx-spec these bits are reserved/must-be-zero, but the real GTE
|
* but the real GTE hardware and PCSX-Redux's GTE model both IGNORE them on these two commands
|
||||||
* hardware and PCSX-Redux's GTE model both IGNORE them on these two
|
* (the perspective divide happens regardless of `sf`).
|
||||||
* commands (the perspective divide happens regardless of `sf`).
|
|
||||||
*
|
*
|
||||||
* If we emit a strictly-spec-compliant word (`sf=0`, reserved bits
|
* If we emit a strictly-spec-compliant word (`sf=0`, reserved bits clear),
|
||||||
* clear), PCSX-Redux's GTE checks those bits more strictly than the
|
* PCSX-Redux's GTE checks those bits more strictly than the silicon does and RTPT silently no-ops.
|
||||||
* silicon does and RTPT silently no-ops — the floor's screen
|
* The floor's screen coordinates come out as raw projection-of-rotation (Z never divided),
|
||||||
* coordinates come out as raw projection-of-rotation (Z never
|
* `nclip` ends up wrong, and the triangle is culled.
|
||||||
* divided), `nclip` ends up wrong, and the triangle is culled.
|
|
||||||
*
|
*
|
||||||
* So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to
|
* So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern.
|
||||||
* match the working bit pattern everyone has shipped for 25 years.
|
* NCLIP / OP / MVMVA stay spec-clean — their reserved bits really are zero in the original PsyQ source.
|
||||||
* NCLIP/OP/MVMVA stay spec-clean — their reserved bits really are
|
|
||||||
* zero in the original PsyQ source.
|
|
||||||
* --------------------------------------------------------------------------
|
* --------------------------------------------------------------------------
|
||||||
*/
|
*/
|
||||||
#define gte_cmdw_psyq_compat (1u << 21 | enc_gte_sf(gte_sf_integer))
|
#define gte_cmdw_psyq_compat (1u << 21 | enc_gte_sf(gte_sf_integer))
|
||||||
@@ -402,9 +398,91 @@ enum { _C2_OPS_ = 0
|
|||||||
#define gte_cmdw_rtpt (gte_cmd_base | enc_gte_cmd(gte_cmd_rtpt ) | gte_cmdw_psyq_compat)
|
#define gte_cmdw_rtpt (gte_cmd_base | enc_gte_cmd(gte_cmd_rtpt ) | gte_cmdw_psyq_compat)
|
||||||
#define gte_cmdw_nclip (gte_cmd_base | enc_gte_cmd(gte_cmd_nclip))
|
#define gte_cmdw_nclip (gte_cmd_base | enc_gte_cmd(gte_cmd_nclip))
|
||||||
#define gte_cmdw_op (gte_cmd_base | enc_gte_cmd(gte_cmd_op ))
|
#define gte_cmdw_op (gte_cmd_base | enc_gte_cmd(gte_cmd_op ))
|
||||||
|
#define gte_cmdw_outer_product gte_cmdw_op /* "outer product" -- PSY-Q terminology */
|
||||||
|
#define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology. */
|
||||||
|
#define gte_cmdw_cross gte_cmdw_op /* "cross product" -- geometric-algebra terminology.
|
||||||
|
* RGA(Lengyel): The GTE OP is a 3D signed-16-bit D x IR cross, not a generic RGA exterior product.
|
||||||
|
* The wedge alias is a 3D complement interpretation of the same 3 scalars (MAC1..MAC3). */
|
||||||
#define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva))
|
#define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* MVMVA with sf=0 (no shift, full-integer), cv=3 (no translation), v=3 (IR vector input).
|
||||||
|
* Reads input from IR1/2/3 (loaded via mtc2 rt, C2_IRx). MAC1/2/3 = RT row · IR (full product, no >>12).
|
||||||
|
* Per PSX-SPX: SAR (sf*12) with sf=0 = SAR 0 = no shift. */
|
||||||
|
#define gte_cmdw_mvmva_sf0_ir (gte_cmd_base | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* MVMVA with sf=1 (>>12 shift, 4.12 fixed-point), cv=3 (no translation), v=3 (IR): for ApplyMatrixLV.
|
||||||
|
* Reads input from IR1/2/3 (loaded via mtc2 rt, C2_IRx). MAC1/2/3 = (RT row · IR) >> 12.
|
||||||
|
* Per PSX-SPX: SAR (sf*12) with sf=1 = SAR 12 = arithmetic right-shift by 12.
|
||||||
|
* This matches the libgte C-side ApplyMatrixLV output (R*pos >> 12). */
|
||||||
|
#define gte_cmdw_mvmva_ir (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* MVMVA: sf=0, mx=3 (Light matrix), v=3 (IR), cv=3 (no TR).
|
||||||
|
* For pass1 of the C11 two-pass decomposition. Reads L matrix.
|
||||||
|
* Since L matrix is typically zero, pass1 contributes 0 to the combine. */
|
||||||
|
#define gte_cmdw_mvmva_sf0_mx3_v3_cv3 (gte_cmd_base | enc_gte_sf(0) | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* MVMVA: sf=1 (>>12), mx=3 (Light matrix), v=2 (V0), cv=0 (with TR).
|
||||||
|
* Matches the C11 ApplyMatrixLV pass 2 command word (0x49E012) exactly.
|
||||||
|
* The combine is (pass1 << 3) + pass2. */
|
||||||
|
#define gte_cmdw_mvmva_pass2_c11 (gte_cmd_base | enc_gte_sf(1) | enc_gte_v(2) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* MVMVA: sf=0, mx=3, v=2, cv=0. Matches the C11 pass 1 command. */
|
||||||
|
#define gte_cmdw_mvmva_pass1_c11 (gte_cmd_base | enc_gte_v(2) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
#define gte_cmdw_mvmva_no_tr gte_cmdw_mvmva_ir
|
||||||
|
|
||||||
|
/* MVMVA pass 2 — C11 ApplyMatrixLV command.
|
||||||
|
* Decoded: op_cop2 | CO | fake_cmd=4 | sf=1 (>>12) | mx=0 (RT matrix) | v=3 (IR) | cv=3 (no translation) | lm=0 | cmd=MVMVA.
|
||||||
|
* Reads (RT row · IR) >> 12 into MAC1/2/3. Per-field composition (no opaque literal)
|
||||||
|
* keeps the bit layout visible at the call site + matches the libgte C-side byte-exact. */
|
||||||
|
#define gte_cmdw_mvmva_c11_pass2 (gte_cmd_base | enc_gte_fake_cmd(4) | enc_gte_sf(1) | enc_gte_v(3) | enc_gte_mx(0) | enc_gte_cv(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* MVMVA: sf=1 (>>12), mx=0 (RT matrix), v=0 (V0), cv=3 (no TR). */
|
||||||
|
#define gte_cmdw_mvmva_sf1_mx0_v0_cv3 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_v(0) | enc_gte_mx(0) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* RTPS with sf=1 (12-bit shift, no translation): matches the output of libgte's ApplyMatrixLV when the GTE pipeline expects R*pos >> 12.
|
||||||
|
* The shift produces values like (-270, 710, 1713) which match the C11 reference path. */
|
||||||
|
#define gte_cmdw_rtps_sf1 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_cmd(gte_cmd_rtps))
|
||||||
|
|
||||||
|
/* SQR / GPF cosmetic-bits compat helpers.
|
||||||
|
* Each command's `_compat` macro ORs in the `fake_cmd` field value libgte happens to emit.
|
||||||
|
* The hardware ignores these bits (per PSX-SPX line 48). */
|
||||||
|
#define gte_cmdw_sqr_fake_sig enc_gte_fake_cmd(0x0A)
|
||||||
|
#define gte_cmdw_gpf_fake_sig enc_gte_fake_cmd(0x19)
|
||||||
|
|
||||||
|
/* SQR — Square Vector.
|
||||||
|
* PSX-SPX `geometrytransformationenginegte.md` §"SQR":
|
||||||
|
* [MAC1,MAC2,MAC3] = [IR1*IR1, IR2*IR2, IR3*IR3] SHR (sf*12)
|
||||||
|
* [IR1,IR2,IR3] = [MAC1,MAC2,MAC3] (saturated to 0x7FFF when lm=1)
|
||||||
|
* Sourced verbatim from libgte msc02 VectorNormal disassembly at 0x800160b0:
|
||||||
|
* 0x4AA00428 = gte_cmd_base | gte_cmdw_sqr_compat | enc_gte_lm(1) | enc_gte_cmd(0x28)
|
||||||
|
* bit 19 sf=0
|
||||||
|
* bit 10 lm=1
|
||||||
|
* bits 5-0 cmd=0x28=SQR
|
||||||
|
* bits 24-20 = 0x0A (libgte "nonsense SDK command number" signature) */
|
||||||
|
#define gte_cmdw_sqr (gte_cmd_base | enc_gte_cmd(gte_cmd_sqr) | enc_gte_lm(1) | gte_cmdw_sqr_fake_sig)
|
||||||
|
|
||||||
|
/* GPF — General-purpose Interpolation.
|
||||||
|
* PSX-SPX `geometrytransformationenginegte.md` §"GPF":
|
||||||
|
* [MAC1,MAC2,MAC3] = (([IR1,IR2,IR3] * IR0) + [MAC1,MAC2,MAC3]) SAR (sf * 12)
|
||||||
|
* [IR1,IR2,IR3] = [MAC1,MAC2,MAC3]
|
||||||
|
* Sourced verbatim from libgte msc02 VectorNormal disassembly at 0x8001613c:
|
||||||
|
* 0x4B90003D = gte_cmd_base | gte_cmdw_gpf_compat | enc_gte_cmd(0x3D)
|
||||||
|
* bit 19 sf = 0
|
||||||
|
* bit 10 lm = 0
|
||||||
|
* bits 5-0 cmd = 0x3D = GPF
|
||||||
|
* bits 24-20 = 0x19 (libgte "nonsense SDK command number" signature) */
|
||||||
|
#define gte_cmdw_gpf (gte_cmd_base | enc_gte_cmd(gte_cmd_gpf) | gte_cmdw_gpf_fake_sig)
|
||||||
|
|
||||||
|
/* Mask to round LZCR (leading-zero/ones count, range 1..32 per PSX-SPX cop2r31) down to even.
|
||||||
|
* The normalize_v3s4 half-shift logic computes (31 - LZCR) >> 1; clearing bit 0 ensures the subtraction result is always odd, so the >> 1 division is consistent (no 0.5 loss). */
|
||||||
|
enum {
|
||||||
|
gte_lzcr_even_mask = 0xFFFE, /* all bits except bit 0 */
|
||||||
|
};
|
||||||
|
|
||||||
|
#define gte_cmdw_rotate_translate_perspective_single gte_cmdw_rtps
|
||||||
#define gte_cmdw_rotate_translate_perspective_triple gte_cmdw_rtpt
|
#define gte_cmdw_rotate_translate_perspective_triple gte_cmdw_rtpt
|
||||||
|
/* RGA(Lengyel): RTPS/RTPT consume the matrix expansion of a rigid transformation (rotation matrix + translation vector) loaded into the RT/TR control registers.
|
||||||
|
* For unitized points the same result equals the motor antiproduct; the GTE executes the LA form, not a symbolic antiproduct. */
|
||||||
|
|
||||||
/* PsyQ compatibility bits for AVSZ3 (Bits 20, 22, 24 must be set) */
|
/* PsyQ compatibility bits for AVSZ3 (Bits 20, 22, 24 must be set) */
|
||||||
#define gte_cmdw_psyq_avsz3_compat (0x15 << 20)
|
#define gte_cmdw_psyq_avsz3_compat (0x15 << 20)
|
||||||
@@ -419,22 +497,20 @@ enum { _C2_OPS_ = 0
|
|||||||
#define gte_cmd_avsz4 0x2E
|
#define gte_cmd_avsz4 0x2E
|
||||||
#define gte_cmdw_avsz4 (gte_cmd_base | enc_gte_cmd(gte_cmd_avsz4) | gte_cmdw_psyq_avsz3_compat)
|
#define gte_cmdw_avsz4 (gte_cmd_base | enc_gte_cmd(gte_cmd_avsz4) | gte_cmdw_psyq_avsz3_compat)
|
||||||
|
|
||||||
|
#define gte_cmdw_avg_sort_z4 gte_cmdw_avsz4
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* @brief Loads a single SVECTOR to GTE vector register V0
|
* @brief Loads a single SVECTOR to GTE vector register V0
|
||||||
*
|
|
||||||
* @details Loads values from an SVECTOR struct to GTE data registers C2_VXY0
|
* @details Loads values from an SVECTOR struct to GTE data registers C2_VXY0
|
||||||
* (XY at offset 0) and C2_VZ0 (Z at offset 4) using `lwc2`.
|
* (XY at offset 0) and C2_VZ0 (Z at offset 4) using `lwc2`.
|
||||||
*
|
*
|
||||||
* Uses string-style GCC inline asm with `%0` substitution because the
|
* Uses string-style GCC inline asm with `%0` substitution because the base register `r0` is a runtime GPR chosen by the compiler.
|
||||||
* base register `r0` is a runtime GPR chosen by the compiler — it cannot
|
* It cannot be encoded into a static `.word` constant.
|
||||||
* be encoded into a static `.word` constant.
|
|
||||||
*
|
*
|
||||||
* Usage:
|
* Usage: asm_gte_load_v0(svector_ptr);
|
||||||
* asm_gte_load_v0(svector_ptr);
|
|
||||||
*/
|
*/
|
||||||
|
|
||||||
/* lwc2 encoding helpers parameterized on the base GPR.
|
/* lwc2 encoding helpers parameterized on the base GPR.
|
||||||
*
|
|
||||||
* gte_lw_v0_xy(base) → lwc2 $0, 0(base) ; C2_VXY0
|
* gte_lw_v0_xy(base) → lwc2 $0, 0(base) ; C2_VXY0
|
||||||
* gte_lw_v0_z(base) → lwc2 $1, 4(base) ; C2_VZ0
|
* gte_lw_v0_z(base) → lwc2 $1, 4(base) ; C2_VZ0
|
||||||
* gte_lw_v1_xy(base) → lwc2 $2, 0(base) ; C2_VXY1
|
* gte_lw_v1_xy(base) → lwc2 $2, 0(base) ; C2_VXY1
|
||||||
@@ -443,8 +519,7 @@ enum { _C2_OPS_ = 0
|
|||||||
* gte_lw_v2_z(base) → lwc2 $5, 4(base) ; C2_VZ2
|
* gte_lw_v2_z(base) → lwc2 $5, 4(base) ; C2_VZ2
|
||||||
*
|
*
|
||||||
* `base` is the GPR number to bake into the .word constant's `rs` field.
|
* `base` is the GPR number to bake into the .word constant's `rs` field.
|
||||||
* These are pure compile-time integers; the C compiler constant-folds
|
* These are pure compile-time integers; the C compiler constant-folds them into .word directives. */
|
||||||
* them into .word directives. */
|
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
GTE_Z_Offset = 4
|
GTE_Z_Offset = 4
|
||||||
@@ -458,28 +533,21 @@ enum {
|
|||||||
#define gte_lw_v2_z(base) enc_gte_lw(gte_in_v2_z, (base), GTE_Z_Offset)
|
#define gte_lw_v2_z(base) enc_gte_lw(gte_in_v2_z, (base), GTE_Z_Offset)
|
||||||
|
|
||||||
/* gte_load_vN(r_ptr, base) — placeholder-punned lwc2 loaders
|
/* gte_load_vN(r_ptr, base) — placeholder-punned lwc2 loaders
|
||||||
*
|
* Emits `.word` constants encoding `lwc2 $N, off(<base>)` for the chosen GTE vector register, where `<base>` is the GPR number you pass in
|
||||||
* Emits `.word` constants encoding `lwc2 $N, off(<base>)` for the chosen
|
|
||||||
* GTE vector register, where `<base>` is the GPR number you pass in
|
|
||||||
* (typically one of R_T4..R_T9 for the standard "3-pointer" pattern).
|
* (typically one of R_T4..R_T9 for the standard "3-pointer" pattern).
|
||||||
*
|
*
|
||||||
* The caller MUST bind `r_ptr` to that same GPR via a register variable:
|
* The caller MUST bind `r_ptr` to that same GPR via a register variable:
|
||||||
*
|
|
||||||
* register V3_S2* p_in_12 __asm__("$12") = my_ptr;
|
* register V3_S2* p_in_12 __asm__("$12") = my_ptr;
|
||||||
* gte_load_v0(p_in_12, R_T4); // R_T4 = 12, base is $12
|
* gte_load_v0(p_in_12, R_T4); // R_T4 = 12, base is $12
|
||||||
*
|
*
|
||||||
* Then `"r"(r_ptr)` inside the asm binds to $12 (the only register
|
* Then `"r"(r_ptr)` inside the asm binds to $12 (the only register `p_in_12` can live in),
|
||||||
* `p_in_12` can live in), which is exactly the register the .word
|
* which is exactly the register the .word constants expect.
|
||||||
* constants expect. A `"$12"` clobber would conflict with the
|
* A `"$12"` clobber would conflict with the register-variable binding ("asm specifier for variable conflicts with asm clobber list"), so we omit it.
|
||||||
* register-variable binding ("asm specifier for variable conflicts
|
* The other ABI-clobbers ($2/$8/$9/$31) stay because the GTE instructions don't touch caller-saved GPRs but the kernel does treat them as volatile.
|
||||||
* with asm clobber list"), so we omit it. The other ABI-clobbers
|
|
||||||
* ($2/$8/$9/$31) stay because the GTE instructions don't touch
|
|
||||||
* caller-saved GPRs but the kernel does treat them as volatile.
|
|
||||||
*
|
*
|
||||||
* WHICH REGISTER TO PICK
|
* WHICH REGISTER TO PICK
|
||||||
* ----------------------
|
* ----------------------
|
||||||
* Any caller-saved GPR is safe. Recommended default for an RTPT-style
|
* Any caller-saved GPR is safe. Recommended default for an RTPT-style 3-pointer pipeline:
|
||||||
* 3-pointer pipeline:
|
|
||||||
* gte_load_v0(p0, R_T4); // $12
|
* gte_load_v0(p0, R_T4); // $12
|
||||||
* gte_load_v1(p1, R_T5); // $13
|
* gte_load_v1(p1, R_T5); // $13
|
||||||
* gte_load_v2(p2, R_T6); // $14
|
* gte_load_v2(p2, R_T6); // $14
|
||||||
@@ -492,32 +560,29 @@ enum {
|
|||||||
* clobbers section : "$2", "$8", ..., "memory" (from asm_clobber)
|
* clobbers section : "$2", "$8", ..., "memory" (from asm_clobber)
|
||||||
* 3 colons total, GCC-legal. No string-syntax mnemonics in the .word body.
|
* 3 colons total, GCC-legal. No string-syntax mnemonics in the .word body.
|
||||||
*
|
*
|
||||||
* The `asm_clobber(...)` helper from gcc_asm.h prepends the colon that
|
* The `asm_clobber(...)` helper from gcc_asm.h prepends the colon that starts the clobbers section. */
|
||||||
* starts the clobbers section. */
|
#define gte_load_v0(r_ptr, base) asm volatile( \
|
||||||
#define gte_load_v0(r_ptr, base) asm volatile( \
|
|
||||||
asm_words( gte_lw_v0_xy(base), gte_lw_v0_z(base) ) \
|
asm_words( gte_lw_v0_xy(base), gte_lw_v0_z(base) ) \
|
||||||
asm_rpins, r_use(r_ptr) \
|
asm_rpins, r_use(r_ptr) \
|
||||||
asm_clobber: rlit(R_V0), rlit(R_T0), rlit(R_T1), rlit(R_RA), clb_mem_drain \
|
asm_clobber: rlit(R_V0), rlit(R_T0), rlit(R_T1), rlit(R_RA), clb_mem_drain \
|
||||||
)
|
)
|
||||||
|
|
||||||
#define gte_load_v1(r_ptr, base) asm volatile( \
|
#define gte_load_v1(r_ptr, base) asm volatile( \
|
||||||
asm_words( gte_lw_v1_xy(base), gte_lw_v1_z(base) ) \
|
asm_words( gte_lw_v1_xy(base), gte_lw_v1_z(base) ) \
|
||||||
asm_rpins, r_use(r_ptr) \
|
asm_rpins, r_use(r_ptr) \
|
||||||
asm_clobber: rlit(R_V0), rlit(R_T0), rlit(R_T1), rlit(R_RA), clb_mem_drain \
|
asm_clobber: rlit(R_V0), rlit(R_T0), rlit(R_T1), rlit(R_RA), clb_mem_drain \
|
||||||
)
|
)
|
||||||
|
|
||||||
#define gte_load_v2(r_ptr, base) asm volatile( \
|
#define gte_load_v2(r_ptr, base) asm volatile( \
|
||||||
asm_words( gte_lw_v2_xy(base), gte_lw_v2_z(base) ) \
|
asm_words( gte_lw_v2_xy(base), gte_lw_v2_z(base) ) \
|
||||||
asm_rpins, r_use(r_ptr) \
|
asm_rpins, r_use(r_ptr) \
|
||||||
asm_clobber: rlit(R_V0), rlit(R_T0), rlit(R_T1), rlit(R_RA), clb_mem_drain \
|
asm_clobber: rlit(R_V0), rlit(R_T0), rlit(R_T1), rlit(R_RA), clb_mem_drain \
|
||||||
)
|
)
|
||||||
|
|
||||||
/* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — the canonical prelude to gte_cmd_rtpt.
|
/* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — prelude to gte_cmd_rtpt.
|
||||||
*
|
|
||||||
* Loads all three GTE input vectors (6 words) from three separate pointers,
|
|
||||||
* one per GTE vector register, each loaded from its own base GPR. Caller
|
|
||||||
* must bind each `pN` to `bN` via a register variable.
|
|
||||||
*
|
*
|
||||||
|
* Loads all three GTE input vectors (6 words) from three separate pointers, one per GTE vector register, each loaded from its own base GPR.
|
||||||
|
* Caller must bind each `pN` to `bN` via a register variable.
|
||||||
* register V3_S2* p0 rgcc(R_T4) = verts[0].ptr; // → __asm__("$12")
|
* register V3_S2* p0 rgcc(R_T4) = verts[0].ptr; // → __asm__("$12")
|
||||||
* register V3_S2* p1 rgcc(R_T5) = verts[1].ptr; // → __asm__("$13")
|
* register V3_S2* p1 rgcc(R_T5) = verts[1].ptr; // → __asm__("$13")
|
||||||
* register V3_S2* p2 rgcc(R_T6) = verts[2].ptr; // → __asm__("$14")
|
* register V3_S2* p2 rgcc(R_T6) = verts[2].ptr; // → __asm__("$14")
|
||||||
@@ -536,34 +601,25 @@ enum {
|
|||||||
|
|
||||||
/**
|
/**
|
||||||
* @brief Rotate, Translate and Perspective Triple (23 cycles)
|
* @brief Rotate, Translate and Perspective Triple (23 cycles)
|
||||||
*
|
* @details Performs rotation, translation and perspective calculation of three vertices at once.
|
||||||
* @details Performs rotation, translation and perspective calculation of three
|
* The equation performed is the same as gte_rtps() only repeated three times for each vertex.
|
||||||
* vertices at once. The equation performed is the same as gte_rtps() only
|
* The result of the first vertex is stored in GTE data register C2_SXY0, the second vector in C2_SXY1 then C2_SXY2.
|
||||||
* repeated three times for each vertex. The result of the first vertex is
|
|
||||||
* stored in GTE data register C2_SXY0, the second vector in C2_SXY1 then
|
|
||||||
* C2_SXY2.
|
|
||||||
*
|
*
|
||||||
* Encoder-style emission (no inline-asm strings in the code body):
|
* Encoder-style emission (no inline-asm strings in the code body):
|
||||||
* 1. Two `nop` words fill the COP2 pipeline latency — the GTE
|
* 1. Two `nop` words fill the COP2 pipeline latency — the GTE takes ~8 cycles per perspective divide,
|
||||||
* takes ~8 cycles per perspective divide, and the nops let any
|
* and the nops let any preceding lwc2/swc2 retire before RTPT starts reading its inputs from V0/V1/V2.
|
||||||
* preceding lwc2/swc2 retire before RTPT starts reading its
|
* 2. The RTPT command word itself is `gte_cmdw_rtpt` (see the pre-baked encoders above) —
|
||||||
* inputs from V0/V1/V2.
|
* `0x0280030` decoded as `op_cop2` | CO(1) | cmd=RTPT, with all SF/MX/V/CV/LM fields zero
|
||||||
* 2. The RTPT command word itself is `gte_cmdw_rtpt` (see the
|
* (standard rotation, no scaling, V0 vector, translation vector, no clamp).
|
||||||
* pre-baked encoders above) — `0x0280030` decoded as
|
|
||||||
* `op_cop2` | CO(1) | cmd=RTPT, with all SF/MX/V/CV/LM fields
|
|
||||||
* zero (standard rotation, no scaling, V0 vector, translation
|
|
||||||
* vector, no clamp).
|
|
||||||
*
|
*
|
||||||
* Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel
|
* Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel ABI)
|
||||||
* ABI) plus the standard "memory" barrier. Does not clobber any COP2
|
* plus the standard "memory" barrier. Does not clobber any COP2 data/control register —
|
||||||
* data/control register — those have to be saved by the caller if
|
* those have to be saved by the caller if they need to survive across the call (RTPT writes SXY0..2, SZ0..3, OTZ, MAC0..3, IR0..3, etc.).
|
||||||
* they need to survive across the call (RTPT writes SXY0..2, SZ0..3,
|
|
||||||
* OTZ, MAC0..3, IR0..3, etc.).
|
|
||||||
*/
|
*/
|
||||||
#define gte_rtpt() \
|
#define gte_rtpt() \
|
||||||
asm volatile( \
|
asm volatile( \
|
||||||
asm_words( nop, nop, gte_cmdw_rtpt ) \
|
asm_words( nop, nop, gte_cmdw_rtpt ) \
|
||||||
asm_clobber: clbr_volatile_gprs \
|
asm_clobber: clbr_volatile_gprs \
|
||||||
)
|
)
|
||||||
|
|
||||||
#define gte_rtpt_asm_str() \
|
#define gte_rtpt_asm_str() \
|
||||||
@@ -574,32 +630,24 @@ enum {
|
|||||||
|
|
||||||
/**
|
/**
|
||||||
* @brief Normal clipping (8 cycles)
|
* @brief Normal clipping (8 cycles)
|
||||||
*
|
* @details Computes the sign of three screen coordinates (C2_SXY0-2) used for backface culling.
|
||||||
* @details Computes the sign of three screen coordinates (C2_SXY0-2) used for
|
* If the value of C2_MAC0 is negative, the coordinates are inverted and thus the triangle is back facing.
|
||||||
* backface culling. If the value of C2_MAC0 is negative, the coordinates are
|
|
||||||
* inverted and thus the triangle is back facing.
|
|
||||||
*
|
*
|
||||||
* The following equation is performed when executing this GTE command:
|
* The following equation is performed when executing this GTE command:
|
||||||
*
|
|
||||||
* MAC0 = SX0*SY1 + SX1*SY2 + SX2*SY0 - SX0*SY2 - SX1*SY0 - SX2*SY1
|
* MAC0 = SX0*SY1 + SX1*SY2 + SX2*SY0 - SX0*SY2 - SX1*SY0 - SX2*SY1
|
||||||
*
|
|
||||||
* Encoder-style emission (no inline-asm strings in the code body):
|
* Encoder-style emission (no inline-asm strings in the code body):
|
||||||
* 1. Two `nop` words fill the COP2 pipeline latency - the GTE
|
* 1. Two `nop` words fill the COP2 pipeline latency
|
||||||
* pipeline takes a few cycles per op, and the nops let any
|
* - the GTE pipeline takes a few cycles per op, and the nops let any preceding
|
||||||
* preceding lwc2/swc2/RTPT retire before NCLIP starts reading
|
* lwc2/swc2/RTPT retire before NCLIP starts reading its inputs from SXY0/SXY1/SXY2.
|
||||||
* its inputs from SXY0/SXY1/SXY2.
|
* 2. The NCLIP command word itself is `gte_cmdw_nclip` (see the pre-baked encoders above)
|
||||||
* 2. The NCLIP command word itself is `gte_cmdw_nclip` (see the
|
* - `0x01400006` decoded as `op_cop2` | CO(1) | cmd=NCLIP, with all SF/MX/V/CV/LM fields zero.
|
||||||
* pre-baked encoders above) - `0x01400006` decoded as
|
* NCLIP is spec-clean in the original PsyQ source (unlike RTPS/RTPT which carry the `gte_cmdw_psyq_compat` quirk),
|
||||||
* `op_cop2` | CO(1) | cmd=NCLIP, with all SF/MX/V/CV/LM fields
|
* so `gte_cmdw_nclip` does NOT OR in any reserved bits.
|
||||||
* zero. NCLIP is spec-clean in the original PsyQ source
|
|
||||||
* (unlike RTPS/RTPT which carry the `gte_cmdw_psyq_compat`
|
|
||||||
* quirk), so `gte_cmdw_nclip` does NOT OR in any reserved bits.
|
|
||||||
*
|
*
|
||||||
* Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel
|
* Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel ABI) plus the standard "memory" barrier.
|
||||||
* ABI) plus the standard "memory" barrier. Does not clobber any COP2
|
* Does not clobber any COP2 data/control register.
|
||||||
* data/control register - those have to be saved by the caller if
|
* Those have to be saved by the caller if they need to survive across the call (NCLIP writes MAC0 only;
|
||||||
* they need to survive across the call (NCLIP writes MAC0 only; it
|
* it is purely a sign-of-double-product computation on SXY0..2).
|
||||||
* is purely a sign-of-double-product computation on SXY0..2).
|
|
||||||
*/
|
*/
|
||||||
#define gte_nclip() \
|
#define gte_nclip() \
|
||||||
asm volatile( \
|
asm volatile( \
|
||||||
@@ -625,17 +673,12 @@ enum {
|
|||||||
"cop2 0x0158002D;")
|
"cop2 0x0158002D;")
|
||||||
|
|
||||||
/* asm_gte_matrix_set_rotation(r0)
|
/* asm_gte_matrix_set_rotation(r0)
|
||||||
|
* Loads the 3x3 rotation matrix at `r0` into the GTE's rotation-matrix control registers (RT11..RT22, indices 0..4) via ctc2.
|
||||||
*
|
*
|
||||||
* Loads the 3x3 rotation matrix at `r0` into the GTE's rotation-matrix
|
* Memory layout at r0: five contiguous 32-bit words (offsets 0..16), each holding two packed 16-bit matrix elements.
|
||||||
* control registers (RT11..RT22, indices 0..4) via ctc2.
|
* The first 1.5 rows of a standard PSX SDK MATRIX struct (where each row is laid out as [RT_xx, RT_xy] | [RT_xz, pad] | ...).
|
||||||
*
|
|
||||||
* Memory layout at r0: five contiguous 32-bit words (offsets 0..16),
|
|
||||||
* each holding two packed 16-bit matrix elements. The first 1.5 rows
|
|
||||||
* of a standard PSX SDK MATRIX struct (where each row is laid out as
|
|
||||||
* [RT_xx, RT_xy] | [RT_xz, pad] | ...).
|
|
||||||
*
|
*
|
||||||
* Generated MIPS (mirrors the source macro):
|
* Generated MIPS (mirrors the source macro):
|
||||||
*
|
|
||||||
* lw $12, 0( %0 ) ; word 0
|
* lw $12, 0( %0 ) ; word 0
|
||||||
* lw $13, 4( %0 ) ; word 1
|
* lw $13, 4( %0 ) ; word 1
|
||||||
* ctc2 $12, $0 ; → C2_RT11
|
* ctc2 $12, $0 ; → C2_RT11
|
||||||
@@ -647,41 +690,36 @@ enum {
|
|||||||
* ctc2 $13, $3 ; → C2_RT21
|
* ctc2 $13, $3 ; → C2_RT21
|
||||||
* ctc2 $14, $4 ; → C2_RT22
|
* ctc2 $14, $4 ; → C2_RT22
|
||||||
*
|
*
|
||||||
* Same contract as gte_load_v0: caller MUST bind `r0` to $12 via a
|
* Same contract as gte_load_v0: caller MUST bind `r0` to $12 via a register variable (`rgcc(R_T4)`) for the `lw $12, off(...)`
|
||||||
* register variable (`rgcc(R_T4)`) for the `lw $12, off(...)`
|
* instructions to read from the right base. The `"r"(r0)` constraint alone doesn't force a specific GPR — it just lets GCC pick one.
|
||||||
* instructions to read from the right base. The `"r"(r0)` constraint
|
* The .word constants here bake R_T4/R_T5/R_T6 into the `rs` field of each lw, so the lw instructions will
|
||||||
* alone doesn't force a specific GPR — it just lets GCC pick one.
|
* only do the right thing if $12 / $13 / $14 hold the matrix base at runtime.
|
||||||
* The .word constants here bake R_T4/R_T5/R_T6 into the `rs` field
|
|
||||||
* of each lw, so the lw instructions will only do the right thing
|
|
||||||
* if $12/$13/$14 hold the matrix base at runtime.
|
|
||||||
*
|
*
|
||||||
* M3_S2* m = ...;
|
* M3_S2* m = ...;
|
||||||
* register M3_S2* m_in_12 rgcc(R_T4) = m;
|
* register M3_S2* m_in_12 rgcc(R_T4) = m;
|
||||||
* asm_gte_matrix_set_rotation(m_in_12);
|
* asm_gte_matrix_set_rotation(m_in_12);
|
||||||
*
|
*
|
||||||
* We clobber $12/$13/$14 (the ones we use as scratch inside the
|
* We clobber $12/$13/$14 (the ones we use as scratch inside the inline asm)
|
||||||
* inline asm) plus the system clobbers; we don't clobber `r0` because
|
* plus the system clobbers; we don't clobber `r0` because the `rgcc` binding already says "this variable lives in $12".
|
||||||
* the `rgcc` binding already says "this variable lives in $12".
|
|
||||||
*
|
*
|
||||||
* WARNING: Incomplete by design. The source macro only writes RT11..RT22
|
* WARNING: Incomplete by design. The source macro only writes RT11..RT22 (5 of 9 rotation elements);
|
||||||
* (5 of 9 rotation elements); RT23 and the entire RT3x row are left
|
* RT23 and the entire RT3x row are left untouched.
|
||||||
* untouched. Real libpsn00b SetRotMatrix writes all 9. Use only when the
|
* Real libpsn00b SetRotMatrix writes all 9. Use only when the GTE's remaining rotation entries are already correct,
|
||||||
* GTE's remaining rotation entries are already correct, or you will
|
* or you will get stale-RT2x/RT3x artifacts in RTPS/RTPT/MVMVA output.
|
||||||
* get stale-RT2x/RT3x artifacts in RTPS/RTPT/MVMVA output.
|
|
||||||
*/
|
*/
|
||||||
#define asm_gte_matrix_set_rotation(r0) \
|
#define asm_gte_matrix_set_rotation(r0) \
|
||||||
asm volatile( \
|
asm volatile( \
|
||||||
asm_words( \
|
asm_words( \
|
||||||
load_word(R_T5, R_T4, 0) \
|
load_word(R_T5, R_T4, 0) \
|
||||||
, load_word(R_T6, R_T4, 4) \
|
, load_word(R_T6, R_T4, 4) \
|
||||||
, gte_mt( R_T5, 0) \
|
, gte_mv_to_data_r( R_T5, 0) \
|
||||||
, gte_mt( R_T6, 1) \
|
, gte_mv_to_data_r( R_T6, 1) \
|
||||||
, load_word(R_T5, R_T4, 8) \
|
, load_word(R_T5, R_T4, 8) \
|
||||||
, load_word(R_T6, R_T4, 12) \
|
, load_word(R_T6, R_T4, 12) \
|
||||||
, load_word(R_T4, R_T4, 16) \
|
, load_word(R_T4, R_T4, 16) \
|
||||||
, gte_mt( R_T5, 2) \
|
, gte_mv_to_data_r( R_T5, 2) \
|
||||||
, gte_mt( R_T6, 3) \
|
, gte_mv_to_data_r( R_T6, 3) \
|
||||||
, gte_mt( R_T4, 4) \
|
, gte_mv_to_data_r( R_T4, 4) \
|
||||||
) \
|
) \
|
||||||
, r_use(r0) \
|
, r_use(r0) \
|
||||||
asm_clobber: clbr_volatile_gprs, rlit(R_T4), rlit(R_T5), rlit(R_T6) \
|
asm_clobber: clbr_volatile_gprs, rlit(R_T4), rlit(R_T5), rlit(R_T6) \
|
||||||
|
|||||||
@@ -2,10 +2,8 @@
|
|||||||
* duffle DSL — GTE Vendor Mnemonics (opt-in)
|
* duffle DSL — GTE Vendor Mnemonics (opt-in)
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
*
|
*
|
||||||
* Provides the textbook MIPS assembly mnemonics for the GTE/COP2
|
* Provides the textbook MIPS assembly mnemonics for the GTE/COP2 instructions as thin aliases to the duffle macros in gte.h.
|
||||||
* instructions as thin aliases to the canonical duffle macros in gte.h.
|
* The duffle names are primary; this header is for users who prefer the textbook mnemonics.
|
||||||
* The duffle names are primary; this header is for users who prefer
|
|
||||||
* the textbook mnemonics.
|
|
||||||
*
|
*
|
||||||
* USAGE: #include "duffle/gte_vendor_sym.h" // after gte.h
|
* USAGE: #include "duffle/gte_vendor_sym.h" // after gte.h
|
||||||
*
|
*
|
||||||
@@ -21,12 +19,6 @@
|
|||||||
* gte_swc2(rt, base, off) -> gte_sw(rt, base, off)
|
* gte_swc2(rt, base, off) -> gte_sw(rt, base, off)
|
||||||
* (the lower-level vector variants gte_lw_v0_xy etc. don't have
|
* (the lower-level vector variants gte_lw_v0_xy etc. don't have
|
||||||
* vendor mnemonics; they're already gte_-prefixed and short)
|
* vendor mnemonics; they're already gte_-prefixed and short)
|
||||||
*
|
|
||||||
* The vendor mnemonics are NOT registered with the duffle word-count
|
|
||||||
* metadata (tape_atom.metadata.h). They expand to the duffle canonical
|
|
||||||
* macros which DO have word-count entries. Verification: V3 (objdump
|
|
||||||
* byte-identical) holds.
|
|
||||||
*
|
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
|
|
||||||
#ifdef INTELLISENSE_DIRECTIVES
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
|||||||
+406
-238
@@ -1,82 +1,267 @@
|
|||||||
#ifdef INTELLISENSE_DIRECTIVES
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
# pragma once
|
# pragma once
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
|
||||||
# include "dsl.h"
|
# include "dsl.h"
|
||||||
# include "gcc_asm.h"
|
# include "gcc_asm.h"
|
||||||
# include "mips.h"
|
# include "mips.h"
|
||||||
# include "gte.h"
|
# include "gte.h"
|
||||||
# include "memory.h"
|
# include "memory.h"
|
||||||
# include "atom_dsl.h"
|
# include "dsl.atom.h"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
typedef U4 const MipsCode;
|
|
||||||
#define MipsAtom_(sym) MipsCode tmpl(code,sym) [] align_(4) =
|
|
||||||
|
|
||||||
#pragma region Tape Drive
|
#pragma region Tape Drive
|
||||||
/* ---------------------------------------------------------------------------
|
/* -----------------------------------------------------------------------------------------------------------
|
||||||
* TAPE DRIVE ABI & REGISTER ALIASES
|
* TAPE DRIVE ABI
|
||||||
* ---------------------------------------------------------------------------
|
* -----------------------------------------------------------------------------------------------------------
|
||||||
* We map the MIPS temporary registers to a persistent global workspace.
|
* Note(Ed): One of the main purposes of this codebase is to help me learn this,
|
||||||
* The C compiler is completely unaware of these bindings.
|
* as such the information below may not* be entirely realized or finalized conceptually.
|
||||||
* ---------------------------------------------------------------------------*/
|
* -----------------------------------------------------------------------------------------------------------
|
||||||
|
* This ABI and its associated legos were directly inspired by researching the work of
|
||||||
|
* Timothy Lottes and Onat Türkçüoğlu; along with many others. It's the simplest bootstrap of a
|
||||||
|
* directly executed chain of assemby arrays (Atoms) that terminate with a yield sequence to the next atom.
|
||||||
|
* These eventually lead to a terminal atom for the tape which is defined below as "tape_exit".
|
||||||
|
*
|
||||||
|
* It behaves as one of the simplest runtime harnesses ontop of a host-enviornment's execution engine
|
||||||
|
* to author and compose programs with. From here various conventions can be further applied.
|
||||||
|
* To make things easier to understand it may be better to focus on what this ABI does not have.
|
||||||
|
* It does not have have any branching within the tape but relative branches within atoms or between atoms.
|
||||||
|
* Branching nearly is always downstream. Automatic stack usage is non-existent.
|
||||||
|
* Push/Pop, FIFO, or Arena/Bump data structures are used by atoms explicitly.
|
||||||
|
* In it's current form with the C11 macro DSL, the user also has fullfill manual register allocation per atom.
|
||||||
|
*
|
||||||
|
* One of the remarkable things about utilizing this ABI is its essentially interopable with CPUs, GPUs, FPGA,
|
||||||
|
* or, basically anything from the 5th generation consoles and onward.
|
||||||
|
* The ABI directly reflects how all computational hardware must be architected in order to execute
|
||||||
|
* digital logic effectively on current era tech.
|
||||||
|
* On the PS1 we don't have access to a few features like multi-threading, speculative execution, or L3 cache;
|
||||||
|
* but, we can set the foundation for legoing whats required for eventually expanding this ABI's paradigm
|
||||||
|
* and core atoms to take those newer hardware features into account. For example, you can easily expand
|
||||||
|
* this to support wave-based execution model on a PS2 or PS3. Not having a stack or
|
||||||
|
* automatic register allocation means the user cannot ignore excessive argument shuffle across workload or
|
||||||
|
* waves and thier phases. Crossing ABI boundaries to other runtimes that do has obviouss penalties.
|
||||||
|
*
|
||||||
|
* Learning data-oriented code becomes a natural progression. Your not fighting a stack-based procedural
|
||||||
|
* paradigm that wants to argument shuffle. There is no ambiguity due to the lack of constraints, for example,
|
||||||
|
* on how the user may "call" a procedure in traditional random dispatch runtimes. The user does have to
|
||||||
|
* hammer down "rules" or patterns for massaging the compiler to dissolve those call frames; just to get
|
||||||
|
* the asesmbly into its desired form. The form is obvious, and once the user gets to author these compoonents
|
||||||
|
* it becomes a game of tetris.
|
||||||
|
*
|
||||||
|
* Another feature is this ABI is very compatible with bootstrapping and developing simple toolchains built off
|
||||||
|
* of bit-packed annotated command streams the user can directly author, maintatain, and immediately execute.
|
||||||
|
* That being like a color forth, or maybe something more familar like an immediate mode library
|
||||||
|
* for various systems such as GUIs. This can make the tetris less of a chore with some helpful policy
|
||||||
|
* generation for allocation of registers, helping to choose resuable components, designing DSL on the fly, etc.
|
||||||
|
* -----------------------------------------------------------------------------------------------------------
|
||||||
|
* TODO(Ed): We need pretty ascii diagrams and proper guides, articles, etc.
|
||||||
|
* -----------------------------------------------------------------------------------------------------------
|
||||||
|
* For now this ideation has just started functioning. I'm abusing C11 & a lua metaprogram to help establish
|
||||||
|
* a hybrid toolchain to ideate on a traditional text-based authoring UX for this paradigm.
|
||||||
|
* If pcsx-redux provides viable hot-reload and persistent data storage beyond save-states
|
||||||
|
* (just copying ram to filesystem), I can author a color forth to mess around with.
|
||||||
|
* With either an editor in-emulator or on the actual machine itself. Assembly is tedius,
|
||||||
|
* but I think this codebase most likely has a pretty ergonomic flavor worst case...
|
||||||
|
* */
|
||||||
|
/* Register Allocation Info */
|
||||||
enum {
|
enum {
|
||||||
R_AtomJmp = R_T9,
|
R_ScratchBase = R_SP atom_reg, /* Scratchpad base address (host frame top) */
|
||||||
R_TapePtr = R_T8, /* The Instruction Stream Pointer */
|
R_AtomJmp = R_FP atom_reg, /* Next atom target (yield handshake scratch) */
|
||||||
R_InCursor = R_T4, /* Input data cursor */
|
R_TapePtr = R_RA atom_reg, /* The Instruction Stream Pointer */
|
||||||
|
/* Stringification codes for the GCC inline assembler clobber lists. */
|
||||||
|
#define R_ScratchBase_Code R_SP_Code
|
||||||
|
#define R_AtomJmp_Code R_FP_Code
|
||||||
|
#define R_TapePtr_Code R_RA_Code
|
||||||
|
|
||||||
R_PrimCursor = R_T7, /* VRAM output cursor (primitive buffer) */
|
// R_InCursor = R_T4,
|
||||||
R_FaceCursor = R_T4, /* Input data cursor (indices/faces) */
|
// #define R_InCursor_Code R_T4_Code
|
||||||
R_VertBase = R_T5, /* Base address of the vertex array */
|
|
||||||
R_OtBase = R_T6, /* Base address of the Ordering Table */
|
|
||||||
|
|
||||||
/* Stringification codes for the GCC inline assembler clobber lists */
|
// Reserved Registers (Callee-saved across the host ABI transition):
|
||||||
#define R_TapePtr_Code R_T8_Code
|
// - R_SP: Holds the scratchpad base while tape code executes.
|
||||||
#define R_InCursor_Code R_T4_Code
|
// - R_FP: Holds the next atom target.
|
||||||
|
// - R_RA: Holds the tape cursor.
|
||||||
|
// All atom-body allocations must stay out of these.
|
||||||
|
// Atom bodies may freely use R2-R25.
|
||||||
|
|
||||||
#define R_PrimCursor_Code R_T7_Code
|
// All allocatable registers for atom bodies (R2-R25, 24 registers):
|
||||||
#define R_FaceCursor_Code R_T4_Code
|
|
||||||
#define R_VertBase_Code R_T5_Code
|
R_PsuedoVolatile = R_AT, // Assembler temporary; never allocate.
|
||||||
#define R_OtBase_Code R_T6_Code
|
|
||||||
|
// Atom Allocation Pool
|
||||||
|
R_Atom0 = R_T0,
|
||||||
|
R_Atom1 = R_T1,
|
||||||
|
R_Atom2 = R_T2,
|
||||||
|
R_Atom3 = R_T3,
|
||||||
|
R_Atom4 = R_T4,
|
||||||
|
R_Atom5 = R_T5,
|
||||||
|
R_Atom6 = R_T6,
|
||||||
|
R_Atom7 = R_T7,
|
||||||
|
R_Atom8 = R_T8,
|
||||||
|
R_Atom9 = R_T9,
|
||||||
|
R_Atom10 = R_V0, // Tend to be used with gte moves
|
||||||
|
R_Atom11 = R_V1, // Tend to be used with gte moves
|
||||||
|
R_Atom12 = R_A0,
|
||||||
|
R_Atom13 = R_A1,
|
||||||
|
R_Atom14 = R_A2,
|
||||||
|
R_Atom15 = R_A3,
|
||||||
|
R_Atom16 = R_S0,
|
||||||
|
R_Atom17 = R_S1,
|
||||||
|
R_Atom18 = R_S2,
|
||||||
|
R_Atom19 = R_S3,
|
||||||
|
R_Atom20 = R_S4,
|
||||||
|
R_Atom21 = R_S5,
|
||||||
|
R_Atom22 = R_S6,
|
||||||
|
R_Atom23 = R_S7,
|
||||||
};
|
};
|
||||||
|
|
||||||
/* The 'Exit' Atom */
|
typedef U2 Reg; // Register parameter used with atom or atom component procedures
|
||||||
MipsAtom_(tape_exit) { jump_reg(rret_addr), nop };
|
#define Reg_(type) tmpl(Reg,type) // Just a way to template register allocations of C-struct types.
|
||||||
|
|
||||||
/* Generalized Tape Engine Runner */
|
typedef U4 const MipsCode; // Underlying type to mips asm words.
|
||||||
FI_ void tape_run(Slice_U4 tape) { register U4* tp rgcc(R_TapePtr) = tape.ptr; asm volatile(
|
typedef Slice_(MipsCode);
|
||||||
asm_words(
|
|
||||||
add_ui( R_SP, R_SP, -MipsStackAlignment) /* Allocate stack space */
|
|
||||||
, store_word(R_RA, R_SP, 0) /* Safely backup $ra to the stack */
|
|
||||||
, load_word( R_AtomJmp, R_TapePtr, 0) /* Bootstrap the first jump */
|
|
||||||
, add_ui_self( R_TapePtr, S_(MipsCode)) /* Advance tape */
|
|
||||||
, call_reg( R_AtomJmp) /* jalr $t9 */
|
|
||||||
, nop /* Branch delay slot */
|
|
||||||
, load_word(R_RA, R_SP, 0) /* Restore $ra from stack */
|
|
||||||
, add_ui_self( R_SP, MipsStackAlignment) /* Deallocate stack space */
|
|
||||||
)
|
|
||||||
asm_rpins, r_use(tp)
|
|
||||||
asm_clobber:
|
|
||||||
rlit(R_AT)
|
|
||||||
, rlit(R_V0), rlit(R_V1)
|
|
||||||
, rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3)
|
|
||||||
/* Tell GCC the tape engine owns and destroys the workspace registers */
|
|
||||||
, rlit(R_PrimCursor), rlit(R_FaceCursor), rlit(R_VertBase), rlit(R_OtBase)
|
|
||||||
, rlit(R_T9)
|
|
||||||
, clb_mem_drain
|
|
||||||
); }
|
|
||||||
|
|
||||||
|
typedef U4 const MipsAtom; // Underlying type to a mips atom defnition
|
||||||
|
typedef Slice_(MipsAtom);
|
||||||
|
|
||||||
|
// Sometimes a user will define a bundle of atoms that represent a procedure of work as:
|
||||||
|
// MipsAtom* <identifier>[...];
|
||||||
|
// Unfortuantely if using slice_from_array it will make the slice's pointer: MipsAtom** so this enforce its defined as MipsAtom*
|
||||||
|
// TODO(Ed): Alternatively we can make the MipsAtom an opaque pointer to the atom... so that the proc returns 'MipsAtom'.
|
||||||
|
#define atombundle_from_array(array) (Slice_MipsAtom){.ptr=array[0],.len=Array_len(array)}
|
||||||
|
|
||||||
|
// Underlying type to an ptr to an array of mips asm words that must terminate with an ac_yield.
|
||||||
|
#define MipsAtom_(sym) MipsCode sym [] align_(4) =
|
||||||
|
|
||||||
|
// Used for atoms with value-args
|
||||||
|
// internal MipsAtom* X_proc(AtomArena_R aa, args) MipsAtom_Proc_(X, aa, { body })
|
||||||
|
// expands to:
|
||||||
|
// internal MipsAtom* X_proc(AtomArena_R aa, args) { MipsCode atom_comp_code[] align_(4) = { body }; return atomarena_push(aa, slice_from_array(MipsCode, atom_comp_code)); }
|
||||||
|
// The atom name is derived by the Lua metaprogram from the preceding
|
||||||
|
// `MipsAtom* X_proc(...)` declaration (backward walk from the macro site,
|
||||||
|
// strips the `_proc` suffix).
|
||||||
|
#define MipsAtom_Proc_(aa, ...) { MipsCode atom_comp_code[] align_(4) = __VA_ARGS__; return atomarena_push(aa, slice_from_array(MipsCode, atom_comp_code)); }
|
||||||
|
|
||||||
|
// Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names).
|
||||||
|
// MipsAtomComp_(ac_X) { body }
|
||||||
|
// expands to:
|
||||||
|
// MipsCode ac_X[] align_(4) = { body };
|
||||||
|
#define MipsAtomComp_(sym) MipsCode sym [] align_(4) =
|
||||||
|
|
||||||
|
// Used for components with value-args (mandatory `ab` (atom-builder) arg).
|
||||||
|
// FI_ void ac_X(MipsAtomBuilder_R ab, args) MipsAtomComp_Proc_(ab, { body })
|
||||||
|
// expands to:
|
||||||
|
// FI_ void ac_X(MipsAtomBuilder_R ab, args) {
|
||||||
|
// MipsCode atom_comp_code[] align_(4) = { body };
|
||||||
|
// atombuilder_push(ab, slice_from_array(MipsCode, atom_comp_code));
|
||||||
|
// }
|
||||||
|
// The body must NOT include mac_yield() (the parent atom yields).
|
||||||
|
// The component name is derived by the Lua metaprogram from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration (backward walk from the macro site).
|
||||||
|
// Inline-only callers (the generated `mac_<name>` aliases) skip the `ab` arg via metaprogram filtering; escape callers (ac_<name> invoked as a function) pass a long-lived builder.
|
||||||
|
#define MipsAtomComp_Proc_(ab, ...) { MipsCode atom_comp_code[] align_(4) = __VA_ARGS__; atombuilder_push(ab, slice_from_array(MipsCode, atom_comp_code)); }
|
||||||
|
|
||||||
|
// Used for trivial mappings from one atom component proc to the command of a more baser (meant for type-mapping)
|
||||||
|
#define MipsAtomComp_ProcMap_(ab, base_command) atom_dbg_skip MipsAtomComp_Proc_(ab, {base_command })
|
||||||
|
|
||||||
|
// WIP: Atoms Assocated closely with each other to form a tape procedure. (Maybe also a phase in a procedure/pipeline?)
|
||||||
|
|
||||||
|
#define AtomBundle_(name) Struct_(tmpl(AtomBundle,name))
|
||||||
|
#define AtomBundle_Len(name) S_(tmpl(AtomBundle,name))/S_(MipsAtom*)
|
||||||
|
#define AtomBundleEntry_(bundle,entry) tmpl(bundle,entry)
|
||||||
|
|
||||||
|
/* Line-table anchor: gcc only adds a file to the .debug_line file table when the contains line-numbered content.
|
||||||
|
Files containing only atoms and atom components.
|
||||||
|
Place `ATOM_FILE_LINE_MARKER();` once at file scope in any `.atom.c` that defines atoms.
|
||||||
|
Macro expands to a file-scope `internal U4 const` declaration keeps the file in the line table.
|
||||||
|
The constant is in `.rodata` so the linker may eliminate it. */
|
||||||
|
#define ATOM_FILE_DEBUGGER_LINE_MARKER(file_name) internal U4 const tmpl(atom_file_debugger_line_marker,file_name) = 0
|
||||||
|
|
||||||
|
typedef Slice_MipsAtom Tape;
|
||||||
|
|
||||||
|
typedef Struct_(TapeHostFrame) {
|
||||||
|
U4 s0;
|
||||||
|
U4 s1;
|
||||||
|
U4 s2;
|
||||||
|
U4 s3;
|
||||||
|
U4 s4;
|
||||||
|
U4 s5;
|
||||||
|
U4 s6;
|
||||||
|
U4 s7;
|
||||||
|
U4 fp;
|
||||||
|
U4 sp;
|
||||||
|
U4 ra;
|
||||||
|
};
|
||||||
|
|
||||||
|
enum {
|
||||||
|
TapeHostFrame_Loc = Scratchpad_End - S_(TapeHostFrame),
|
||||||
|
TapeScratch_Len = TapeHostFrame_Loc - Scratchpad_Loc,
|
||||||
|
};
|
||||||
|
static_assert(S_(TapeHostFrame) == 11 * S_(U4));
|
||||||
|
static_assert(TapeHostFrame_Loc == 0x1F8003D4);
|
||||||
|
|
||||||
|
atom_dbg_skip MipsAtom_(tape_enter) {
|
||||||
|
mac_load_word_imm(R_V0, u4_(TapeHostFrame_Loc)),
|
||||||
|
store_word(R_S0, R_V0, O_(TapeHostFrame,s0)),
|
||||||
|
store_word(R_S1, R_V0, O_(TapeHostFrame,s1)),
|
||||||
|
store_word(R_S2, R_V0, O_(TapeHostFrame,s2)),
|
||||||
|
store_word(R_S3, R_V0, O_(TapeHostFrame,s3)),
|
||||||
|
store_word(R_S4, R_V0, O_(TapeHostFrame,s4)),
|
||||||
|
store_word(R_S5, R_V0, O_(TapeHostFrame,s5)),
|
||||||
|
store_word(R_S6, R_V0, O_(TapeHostFrame,s6)),
|
||||||
|
store_word(R_S7, R_V0, O_(TapeHostFrame,s7)),
|
||||||
|
store_word(R_FP, R_V0, O_(TapeHostFrame,fp)),
|
||||||
|
store_word(R_SP, R_V0, O_(TapeHostFrame,sp)),
|
||||||
|
store_word(R_RA, R_V0, O_(TapeHostFrame,ra)),
|
||||||
|
add_ui(R_TapePtr, R_A0, 0),
|
||||||
|
load_upper_i(R_ScratchBase, u4_hi(Scratchpad_Loc)),
|
||||||
|
load_word(R_AtomJmp, R_TapePtr, 0),
|
||||||
|
add_ui_self( R_TapePtr, S_(MipsAtom)),
|
||||||
|
jump_reg(R_AtomJmp), BdSlot_ nop,
|
||||||
|
};
|
||||||
|
|
||||||
|
atom_dbg_skip MipsAtom_(tape_exit) {
|
||||||
|
mac_load_word_imm(R_V0, u4_(TapeHostFrame_Loc)),
|
||||||
|
load_word(R_S0, R_V0, O_(TapeHostFrame,s0)),
|
||||||
|
load_word(R_S1, R_V0, O_(TapeHostFrame,s1)),
|
||||||
|
load_word(R_S2, R_V0, O_(TapeHostFrame,s2)),
|
||||||
|
load_word(R_S3, R_V0, O_(TapeHostFrame,s3)),
|
||||||
|
load_word(R_S4, R_V0, O_(TapeHostFrame,s4)),
|
||||||
|
load_word(R_S5, R_V0, O_(TapeHostFrame,s5)),
|
||||||
|
load_word(R_S6, R_V0, O_(TapeHostFrame,s6)),
|
||||||
|
load_word(R_S7, R_V0, O_(TapeHostFrame,s7)),
|
||||||
|
load_word(R_RA, R_V0, O_(TapeHostFrame,ra)),
|
||||||
|
load_word(R_FP, R_V0, O_(TapeHostFrame,fp)),
|
||||||
|
load_word(R_SP, R_V0, O_(TapeHostFrame,sp)),
|
||||||
|
jump_reg(R_RA), BdSlot_ nop,
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef void Proc_(TapeEntryFn)(MipsAtom* tape_ptr);
|
||||||
|
|
||||||
|
FI_ void tape_run(Tape tape) { C_(TapeEntryFn*, tape_enter)(tape.ptr); }
|
||||||
|
|
||||||
|
// Procedural authoring of tapes:
|
||||||
typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; };
|
typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; };
|
||||||
FI_ void tb_init(TapeBuilder* tb, FArena* arena) { tb->ptr = arena->start; tb->used = 0; }
|
FI_ void tb_init(TapeBuilder* tb, FArena* arena) { tb->ptr = arena->start; tb->used = 0; }
|
||||||
FI_ TapeBuilder tb_make_old( FArena* arena) { return (TapeBuilder){ arena->start, 0 }; }
|
FI_ TapeBuilder tb_make_old( FArena* arena) { return (TapeBuilder){ arena->start, 0 }; }
|
||||||
FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ mem.ptr, mem.len, 0 }; }
|
FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ u4_(mem.ptr), mem.len, 0 }; } /* capacity in elements (matches used units) */
|
||||||
|
|
||||||
#define tb_emit_(tb, atom) tb_emit(tb, tmpl(code,atom))
|
FI_ void tb_emit(TapeBuilder* tb, MipsAtom* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
|
||||||
FI_ void tb_emit(TapeBuilder* tb, MipsCode* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
|
|
||||||
FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; }
|
FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; }
|
||||||
|
#define tb_emit_(atom) tb_emit(& tb, atom)
|
||||||
|
|
||||||
FI_ Slice_U4 tb_end (TapeBuilder* tb) { tb_emit(tb,code_tape_exit); return (Slice_U4){ C_(U4*,tb->ptr), tb->used }; }
|
FI_ void tb_bind(TapeBuilder* tb, Slice data) { mem_copy(tb->ptr + tb->used * S_(MipsCode), u4_(data.ptr), data.len); tb->used += data.len / S_(MipsCode); }
|
||||||
FI_ Slice_U4 tb_slice(TapeBuilder tb) { return (Slice_U4){ C_(U4*,tb.ptr), tb.used }; }
|
#define tb_bind_(tb,type,...) tb_bind(tb, (Slice){ (B1*)(& (type){__VA_ARGS__}), S_(type) }); static_assert(S_(type) % S_(MipsCode) == 0)
|
||||||
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,code_tape_exit))
|
|
||||||
|
|
||||||
|
// NOTE(Ed): Wip still ideating convention. Possibly will never use a composite.
|
||||||
|
#define tb_emit_wbind_(tb,atom,...) tb_emit(tb,atom); tb_bind_(tb,tmpl(Binds,atom),__VA_ARGS__)
|
||||||
|
#define tb_emit_wbind2_(tb,atom,type,...) tb_emit(tb,atom); tb_bind_(tb,type,__VA_ARGS__)
|
||||||
|
|
||||||
|
FI_ Tape tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Tape){ C_(U4*,tb->ptr), tb->used }; }
|
||||||
|
FI_ Tape tb_slice(TapeBuilder tb) { return (Tape){ C_(U4*,tb.ptr), tb.used }; }
|
||||||
|
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit))
|
||||||
|
|
||||||
|
FI_ void tb_scope_run_end(TapeBuilder* tb) { tb_emit(tb,tape_exit); tape_run(tb_slice(tb[0])); }
|
||||||
|
#define tb_scope_run(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_scope_run_end(tb))
|
||||||
#pragma endregion Tape Drive
|
#pragma endregion Tape Drive
|
||||||
|
|
||||||
#pragma region Macro Mips Atom Components
|
#pragma region Macro Mips Atom Components
|
||||||
@@ -85,204 +270,187 @@ FI_ Slice_U4 tb_slice(TapeBuilder tb) { return (Sli
|
|||||||
* These do NOT yield. They are expanded inline inside Tape Atoms.
|
* These do NOT yield. They are expanded inline inside Tape Atoms.
|
||||||
* ---------------------------------------------------------------------------*/
|
* ---------------------------------------------------------------------------*/
|
||||||
|
|
||||||
/* The 'Yield' sequence for Tape Atoms.
|
// The 'Yield' sequence for Tape Atoms (mac_yield).
|
||||||
* Loads the next pointer from the tape, advances the tape, and jumps.
|
|
||||||
* Cost: ~ 4 cycles */
|
|
||||||
#define mac_yield() \
|
|
||||||
load_word(R_AtomJmp, R_TapePtr, 0) \
|
|
||||||
, add_ui_self( R_TapePtr, S_(MipsCode)) \
|
|
||||||
, jump_reg( R_AtomJmp) \
|
|
||||||
, nop
|
|
||||||
|
|
||||||
/* Words: 3; Loads 3 S2 indices from the face array */
|
atom_dbg_skip MipsAtomComp_(ac_yield) {
|
||||||
#define mac_load_tri_indices(rId_0, rId_1, rId_2) \
|
load_word(R_AtomJmp, R_TapePtr, 0), LdSlot_
|
||||||
load_half_u(rId_0, R_FaceCursor, 0 * S_(S2)) \
|
add_ui_self( R_TapePtr, S_(MipsCode)),
|
||||||
, load_half_u(rId_1, R_FaceCursor, 1 * S_(S2)) \
|
jump_reg( R_AtomJmp), BdSlot_ nop,
|
||||||
, load_half_u(rId_2, R_FaceCursor, 2 * S_(S2))
|
};
|
||||||
|
|
||||||
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
|
atom_dbg_skip MipsAtomComp_(ac_yield_load) {
|
||||||
#define mac_load_tri_verts(rId_0, rId_1, rId_2) \
|
load_word(R_AtomJmp, R_TapePtr, 0),
|
||||||
shift_lleft(R_AT, rId_0, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0) \
|
};
|
||||||
, shift_lleft(R_AT, rId_1, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1) \
|
|
||||||
, shift_lleft(R_AT, rId_2, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2)
|
|
||||||
|
|
||||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list */
|
atom_dbg_skip MipsAtomComp_(ac_yield_tail) {
|
||||||
#define mac_insert_ot_tag(r_otz, prim_length) \
|
add_ui_self(R_TapePtr, S_(MipsCode)),
|
||||||
shift_lleft( R_T1, r_otz, 2) /* T1 = r_otz * S_(U4) */ \
|
jump_reg( R_AtomJmp), BdSlot_ nop,
|
||||||
, add_u( R_T1, R_T1, R_OtBase) /* T1 = & OrderingTable[OTZ] */ \
|
};
|
||||||
, load_word( R_AT, R_T1, O_(PolyTag,bf_addr_len)) /* AT = old_ot_head */ \
|
|
||||||
, load_upper_i(R_V0, prim_length) /* V0 = prim_length << 16 (high 16 bits of a tag) */ \
|
|
||||||
, mask_upper( R_AT, R_AT, S_(polytag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
|
|
||||||
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
|
|
||||||
, store_word( R_AT, R_PrimCursor, O_(PolyTag,bf_addr_len)) /* prim->tag = packed(prim_length, old_addr) */ \
|
|
||||||
, shift_lleft( R_AT, R_PrimCursor, S_(polytag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
|
|
||||||
, shift_lright(R_AT, R_AT, S_(polytag_len_bits)) \
|
|
||||||
, store_word( R_AT, R_T1, O_(PolyTag,bf_addr_len)) /* OrderingTable[OTZ] = PrimCursor */
|
|
||||||
|
|
||||||
/* Words: 3; Emits the F3 command+color word (cmd byte | BLUE | GREEN | RED)
|
|
||||||
* Args: _r, _g, _b are 8-bit RGB byte values (not raw 16-bit fields).
|
|
||||||
* Migrated from hello_gte_tape.c; takes RGB form per the Phase 3
|
|
||||||
* convention. */
|
|
||||||
#define mac_format_f3_color(_r, _g, _b) \
|
|
||||||
load_upper_i( R_AT, gp0_cmd_poly_f3 << 8 | (_b)) \
|
|
||||||
, or_i( R_AT, R_AT, ((_g) << 8) | (_r)) \
|
|
||||||
, store_word(R_AT, R_PrimCursor, O_(Poly_F3,color))
|
|
||||||
|
|
||||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3 */
|
|
||||||
#define mac_gte_store_f3() \
|
|
||||||
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_F3,p0)) \
|
|
||||||
, gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_F3,p1)) \
|
|
||||||
, gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_F3,p2))
|
|
||||||
|
|
||||||
#pragma endregion Macro Atom Components
|
#pragma endregion Macro Atom Components
|
||||||
|
|
||||||
#pragma region Mips Atom Builder
|
#pragma region Atom Builder
|
||||||
// This allows for runtime procedural authoring of mips atoms.
|
// This helps with runtime procedural authoring of mips atoms.
|
||||||
|
typedef Relative_(FArena) Struct_(AtomBuilder) { U4 start; U4 capacity; U4 used; };
|
||||||
|
|
||||||
typedef Struct_(FMipsAtom512) { U4 data[512]; U4 used; };
|
// Usual way to resolve an atom after the bulder is done.
|
||||||
|
#define atom_from_atombuilder(ab) C_(MipsAtom*, (ab).start)
|
||||||
|
|
||||||
typedef Slice_(MipsCode); typedef Slice_MipsCode MipsAtom;
|
FI_ void atombuilder_push(AtomBuilder_R ab, Slice_MipsCode code) {
|
||||||
// FArena Related
|
assert(ab->capacity - ab->used - code.len);
|
||||||
typedef Relative_(FArena) Struct_(MipsAtomBuilder) { U4 start; U4 capacity; U4 used; };
|
U4 dest = ab->start + ab->used * S_(MipsCode); U4 size = S_slice(code);
|
||||||
// Whatever the builder is writting to should most likely coresspond
|
mem_copy(dest, u4_(code.ptr), size); ab->used += size;
|
||||||
// to something that can fit within instruction cache?
|
|
||||||
|
|
||||||
FI_ void atombuilder_unroll(MipsAtomBuilder_R ab, Slice_MipsCode_R code) {
|
|
||||||
assert(ab->capacity - ab->used - code->len);
|
|
||||||
mem_copy(ab->start, u4_(code->ptr), code->len);
|
|
||||||
mem_bump(ab->start, ab->capacity, & ab->used, code->len);
|
|
||||||
}
|
}
|
||||||
#define atombuilder_unroll_mac(ab, mac) atombuilder_unroll(ab, slice_arg_from_array(Slice_MipsCode, mac))
|
#define atombuilder_push_mac(ab, mac) atombuilder_push(ab, slice_arg_from_array(Slice_MipsCode, mac))
|
||||||
|
|
||||||
// When done authoring, utilize this to cap-off the atom
|
// When done authoring, utilize this to cap-off the atom (if not utilizing a MipsAtom_Proc).
|
||||||
FI_ void atombuilder_end(MipsAtomBuilder_R ab) {
|
FI_ void atombuilder_end(AtomBuilder_R ab) { atombuilder_push(ab, slice_from_array(MipsCode, ac_yield)); }
|
||||||
LP_ MipsAtom_(yield) { mac_yield() };
|
|
||||||
mem_copy(ab->start, u4_(code_yield), S_(code_yield));
|
|
||||||
mem_bump(ab->start, ab->capacity, & ab->used, S_(code_yield));
|
|
||||||
}
|
|
||||||
|
|
||||||
#define mipsatom_from_builder(ab) (MipsAtom){ab.start, ab.used}
|
|
||||||
|
|
||||||
|
FI_ void tb_emit_atombuilder(TapeBuilder_R tb, AtomBuilder_R ab) { tb_emit(tb, atom_from_atombuilder(ab[0])); }
|
||||||
#pragma endregion Mips Atom Builder
|
#pragma endregion Mips Atom Builder
|
||||||
|
|
||||||
|
#pragma region Atom Arena
|
||||||
|
// Just a dedicated FArena that is meant to mem_copy and return atom definitions made with MipsAtom_Proc_
|
||||||
|
typedef Relative_(FArena) Struct_(AtomArena) { U4 start; U4 capacity; U4 used; };
|
||||||
|
|
||||||
|
#define atomarena_unused_start(ab) ((ab).start + (ab).used)
|
||||||
|
FI_ void atomarena_init(AtomArena_R arena, Slice mem) { assert(arena != nullptr);
|
||||||
|
arena->start = u4_(mem.ptr);
|
||||||
|
arena->capacity = mem.len;
|
||||||
|
arena->used = 0;
|
||||||
|
}
|
||||||
|
FI_ AtomArena atomarena_make(Slice mem) { AtomArena a; atomarena_init(& a, mem); return a; }
|
||||||
|
FI_ MipsAtom* atomarena_push(AtomArena_R aa, Slice_MipsCode code) {
|
||||||
|
assert(aa->capacity - aa->used - code.len);
|
||||||
|
U4 dest = atomarena_unused_start(aa[0]); U4 size = S_slice(code);
|
||||||
|
mem_copy(dest, u4_(code.ptr), size); aa->used += size;
|
||||||
|
return C_(MipsAtom*, dest);
|
||||||
|
}
|
||||||
|
FI_ void atomarena_reset(AtomArena_R aa) { aa->used = 0; }
|
||||||
|
#pragma endregion Atom Arena
|
||||||
|
|
||||||
|
#pragma region RegFile (Register File Allocator)
|
||||||
|
// A specialized allocator utilized to help the user track which registers are bound to values
|
||||||
|
// that must be preserved for the arena's bounds.
|
||||||
|
// TODO(Ed): Technically we can do this at comp-time with the metaprogram, but we may have namespace conflicts.
|
||||||
|
// Unless we follow a convention for #define <Scope_Prefix> or something per register allocation boundary.
|
||||||
|
|
||||||
|
/* ABI reserves that are never handed out by alloc.
|
||||||
|
* R_AT is the assembler temporary (per the MIPS O32 ABI).
|
||||||
|
* R_K0/K1 are kernel reserves.
|
||||||
|
* R_GP stays the host global pointer.
|
||||||
|
* R_SP/R_FP/R_RA are tape runtime carriers between tape_enter and tape_exit. */
|
||||||
|
U4 const regfile_abi_mask =
|
||||||
|
(1u << R_0) | (1u << R_AT) |
|
||||||
|
(1u << R_K0) | (1u << R_K1) |
|
||||||
|
(1u << R_GP) | (1u << R_SP) |
|
||||||
|
(1u << R_FP) | (1u << R_RA);
|
||||||
|
|
||||||
|
internal Reg const regfile_alloc_order[] = {
|
||||||
|
R_V0, R_V1,
|
||||||
|
R_A0, R_A1, R_A2, R_A3,
|
||||||
|
R_T0, R_T1, R_T2, R_T3, R_T4, R_T5, R_T6, R_T7,
|
||||||
|
R_S0, R_S1, R_S2, R_S3, R_S4, R_S5, R_S6, R_S7,
|
||||||
|
R_T8, R_T9,
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef Struct_(RegFile) {
|
||||||
|
A2_U2 GPR;
|
||||||
|
A2_U2 GTE;
|
||||||
|
};
|
||||||
|
#define regfile(pin_mask) {.GPR={u4_lo(pin_mask), u4_hi(pin_mask)} }
|
||||||
|
FI_ void regfile_init(RegFile_R rf) {
|
||||||
|
/* pack the 32-bit ABI mask into the two U2s */
|
||||||
|
rf->GPR[0] = u4_lo(regfile_abi_mask);
|
||||||
|
rf->GPR[1] = u4_hi(regfile_abi_mask);
|
||||||
|
rf->GTE[0] = rf->GTE[1] = 0;
|
||||||
|
}
|
||||||
|
FI_ RegFile regfile_make(void) { RegFile rf; regfile_init(& rf); return rf; }
|
||||||
|
|
||||||
|
typedef Struct_(RegFile_RInfo) {
|
||||||
|
U2_R section;
|
||||||
|
U2 mask;
|
||||||
|
B2 occupied;
|
||||||
|
};
|
||||||
|
FI_ RegFile_RInfo regfile_rinfo(A2_U2 file, Reg r_id) {
|
||||||
|
U2 s_id = r_id >> 4;
|
||||||
|
U2_R section = & file[s_id];
|
||||||
|
U2 mask = u2_(1u << (r_id & 15));
|
||||||
|
B2 occupied = (section[0] & mask) != 0;
|
||||||
|
return (RegFile_RInfo){section, mask, occupied};
|
||||||
|
}
|
||||||
|
FI_ Reg regfile__alloc_helper(A2_U2 file, Reg r_id) {
|
||||||
|
Reg result = 0; RegFile_RInfo info = regfile_rinfo(file, r_id);
|
||||||
|
if (info.occupied == false) {
|
||||||
|
info.section[0] |= info.mask;
|
||||||
|
result = r_id;
|
||||||
|
}
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
I_ Reg regfile_alloc(RegFile_R rf) {
|
||||||
|
Reg allocated = 0;
|
||||||
|
for index_iter(U4, r_id, R_V0, <, R_T9) {
|
||||||
|
allocated = regfile__alloc_helper(rf->GPR, r_id);
|
||||||
|
Jmp_nZero_(allocated,resolved);
|
||||||
|
}
|
||||||
|
assert(allocated != 0);
|
||||||
|
resolved: return allocated;
|
||||||
|
}
|
||||||
|
FI_ Reg regfile_pin(RegFile_R rf, Reg r_id) {
|
||||||
|
RegFile_RInfo info = regfile_rinfo(rf->GPR, r_id);
|
||||||
|
assert(info.occupied == false);
|
||||||
|
info.section[0] |= info.mask;
|
||||||
|
return r_id;
|
||||||
|
}
|
||||||
|
FI_ void regfile_pin_mask(RegFile_R rf, U4 mask) {
|
||||||
|
B4 occupied = u4_r(rf->GPR)[0] & mask;
|
||||||
|
assert(occupied == false);
|
||||||
|
u4_r(rf->GPR)[0] |= mask;
|
||||||
|
}
|
||||||
|
FI_ void regfile_free_mask(RegFile_R rf, U4 mask) {
|
||||||
|
if (regfile_abi_mask & mask) return;
|
||||||
|
u4_r(rf->GPR)[0] &= ~mask;
|
||||||
|
}
|
||||||
|
FI_ void regfile_free_reg(RegFile_R rf, Reg r_id) {
|
||||||
|
/* never free the ABI set */
|
||||||
|
if (regfile_abi_mask & (1u << r_id)) return;
|
||||||
|
RegFile_RInfo info = regfile_rinfo(rf->GPR, r_id);
|
||||||
|
info.section[0] &= ~info.mask;
|
||||||
|
}
|
||||||
|
FI_ void regfile_reset(RegFile_R rf) {
|
||||||
|
rf->GPR[0] = u4_lo(regfile_abi_mask);
|
||||||
|
rf->GPR[1] = u4_hi(regfile_abi_mask);
|
||||||
|
}
|
||||||
|
FI_ void regfile_reset_to_mask(RegFile_R rf, U4 mask) {
|
||||||
|
rf->GPR[0] = u4_lo(mask);
|
||||||
|
rf->GPR[1] = u4_hi(mask);
|
||||||
|
}
|
||||||
|
#pragma endregion RegFileArena (Register File Allocator)
|
||||||
|
|
||||||
|
#pragma region Mips Atom Procs
|
||||||
|
/* RegUse structs are a convention to organize register allocations for a mips atom procedure.
|
||||||
|
Unlike the usual enum-based declarations, they provide a namespaced scope and have view types via union declarations. */
|
||||||
|
#define RegUse_(proc_name) (tmpl(RegUse,proc_name))
|
||||||
|
|
||||||
|
typedef Struct_(RegUse_example_atom_proc) {
|
||||||
|
Reg const ro_register; // Scratch base carrier.
|
||||||
|
Reg usual_modifiable;
|
||||||
|
union { Reg view_1, view_2, view_3; } t1;
|
||||||
|
};
|
||||||
|
internal MipsAtom* example_atom_proc(AtomArena_R aa, U2 offset, RegUse_example_atom_proc r)
|
||||||
|
MipsAtom_Proc_(aa, {
|
||||||
|
add_si(r.usual_modifiable, r.ro_register, offset),
|
||||||
|
or_u(r.t1.view_1, r.ro_register, 0),
|
||||||
|
branch_lt_zero(r.t1.view_1, atom_offset(example_atom_proc, skip)), BdSlot_ nop,
|
||||||
|
li_s(r.t1.view_2, 100),
|
||||||
|
atom_label(skip)
|
||||||
|
add_si(r.t1.view_3, r.usual_modifiable, 10),
|
||||||
|
mac_yield(),
|
||||||
|
})
|
||||||
|
|
||||||
|
#pragma endregion Mips Atom Procs
|
||||||
|
|
||||||
#pragma region Baked Mips Atoms
|
#pragma region Baked Mips Atoms
|
||||||
// These atoms are resolved at compile time and are (usually) statically linked readonly data.
|
// These atoms are resolved at compile time and are (usually) statically linked readonly data.
|
||||||
|
|
||||||
enum {
|
|
||||||
bios_flushcache = 0x44,
|
|
||||||
bios_table_addr = 0xA0,
|
|
||||||
};
|
|
||||||
|
|
||||||
/* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0).
|
|
||||||
*
|
|
||||||
* Sequence (per MIPS ABI; arguments in arg registers, RA pushed to stack):
|
|
||||||
* 1. sp -= 8; sw $ra, 4($sp) ; save RA
|
|
||||||
* 2. $a0 = bios_flushcache (arg0)
|
|
||||||
* 3. $t0 = bios_table_addr ; t0 = &BIOS A-function table
|
|
||||||
* 4. jalr $t0, $ra ; call BIOS(flushcache)
|
|
||||||
* nop ; branch delay slot
|
|
||||||
* 5. lw $ra, 4($sp); jr $ra ; restore & return
|
|
||||||
* 6. sp += 8
|
|
||||||
*/
|
|
||||||
internal MipsAtom_(mips_flush_icache) {
|
|
||||||
add_ui(rstack_ptr, rstack_ptr, -MipsStackAlignment) /* sp -= 8 */
|
|
||||||
, store_word(rret_addr, rstack_ptr, S_(U4)) /* sw $ra, 4($sp) */
|
|
||||||
, add_ui(rret_0, rdiscard, bios_flushcache) /* addiu $a0, $0, 0x44 */
|
|
||||||
, add_ui(rtmp_0, rdiscard, bios_table_addr) /* addiu $t0, $0, 0xA0 */
|
|
||||||
, jump_link(rtmp_0, rret_addr) /* jalr $t0, $ra */
|
|
||||||
, nop /* BD slot */
|
|
||||||
, load_word(rret_addr, rstack_ptr, S_(U4)) /* lw $ra, 4($sp) */
|
|
||||||
, jump_reg(rret_addr) /* jr $ra */
|
|
||||||
, add_ui(rstack_ptr, rstack_ptr, MipsStackAlignment) /* sp += 8 (BD) */
|
|
||||||
, mac_yield()
|
|
||||||
};
|
|
||||||
|
|
||||||
typedef Struct_(Binds_SetGteWorld) {
|
|
||||||
U4 transform;
|
|
||||||
};
|
|
||||||
// TODO(Ed): Bugged, fix
|
|
||||||
internal MipsAtom_(set_gte_world) {
|
|
||||||
/* Pop matrix address from tape into R_T3 ($11) */
|
|
||||||
load_word(R_T3, R_TapePtr, O_(Binds_SetGteWorld,transform)),
|
|
||||||
add_ui_self( R_TapePtr, S_(Binds_SetGteWorld)),
|
|
||||||
|
|
||||||
// TODO(Ed): Annotate magic offsets.
|
|
||||||
/* Load 3x3 Rotation + 3x1 Translation from R_T3 into GTE CONTROL Regs (ctc2) */
|
|
||||||
load_word(R_T0, R_T3, 0), load_word(R_T1, R_T3, 4),
|
|
||||||
gte_mv_to_ctrl_r( R_T0, gte_cr_RT11), gte_mv_to_ctrl_r( R_T1, gte_cr_RT12),
|
|
||||||
load_word(R_T0, R_T3, 8), load_word(R_T1, R_T3, 12), load_word(R_T2, R_T3, 16),
|
|
||||||
gte_mv_to_ctrl_r( R_T0, gte_cr_RT13), gte_mv_to_ctrl_r( R_T1, gte_cr_RT21), gte_mv_to_ctrl_r( R_T2, gte_cr_RT22),
|
|
||||||
load_word(R_T0, R_T3, 20), load_word(R_T1, R_T3, 24), load_word(R_T2, R_T3, 28),
|
|
||||||
gte_mv_to_ctrl_r( R_T0, gte_cr_TRX), gte_mv_to_ctrl_r( R_T1, gte_cr_TRY), gte_mv_to_ctrl_r( R_T2, gte_cr_TRZ),
|
|
||||||
|
|
||||||
mac_yield()
|
|
||||||
};
|
|
||||||
|
|
||||||
// TODO(Ed): I'm not sure yet if the bindings are redundant with the floortri atom yet.
|
|
||||||
|
|
||||||
/* DIAGNOSTIC 1: Pure tape loop test */
|
|
||||||
internal MipsAtom_(diag_yield) { mac_yield() };
|
|
||||||
|
|
||||||
// TODO(Ed): Reduce magic numbers/offsets
|
|
||||||
/* DIAGNOSTIC 2: Pure memory test (No GTE). Draws a fixed cyan triangle. */
|
|
||||||
internal MipsAtom_(diag_color) {
|
|
||||||
store_word( R_0, R_T7, 0),
|
|
||||||
load_upper_i(R_AT, gp0_cmd_poly_f3 << 8 | 0xFF), /* High: MipsCode Poly_F3(0x20) + Color B:FF */
|
|
||||||
or_i( R_AT, R_AT, 0xFF00), /* Low: Color G:FF, R:00 (Cyan) */
|
|
||||||
store_word( R_AT, R_T7, 4),
|
|
||||||
|
|
||||||
/* Fake coordinates - Swapped winding order to prevent GPU culling! */
|
|
||||||
load_upper_i(R_AT, 0x0010), or_i(R_AT, R_AT, 0x0010), store_word(R_AT, R_T7, 8), /* (16, 16) */
|
|
||||||
load_upper_i(R_AT, 0x0050), or_i(R_AT, R_AT, 0x0010), store_word(R_AT, R_T7, 12), /* (80, 16) */
|
|
||||||
load_upper_i(R_AT, 0x0010), or_i(R_AT, R_AT, 0x0050), store_word(R_AT, R_T7, 16), /* (16, 80) */
|
|
||||||
|
|
||||||
add_ui( R_T1, R_0, 10),
|
|
||||||
shift_lleft(R_T1, R_T1, 2),
|
|
||||||
add_u( R_T1, R_T1, R_T6),
|
|
||||||
|
|
||||||
load_word( R_AT, R_T1, 0),
|
|
||||||
load_upper_i(R_V0, 0x0400), // <--- Fills load delay slot!
|
|
||||||
store_word( R_AT, R_T7, 0),
|
|
||||||
|
|
||||||
shift_lleft( R_AT, R_T7, 8), shift_lright(R_AT, R_AT, 8),
|
|
||||||
or_u( R_AT, R_AT, R_V0),
|
|
||||||
store_word(R_AT, R_T1, 0),
|
|
||||||
|
|
||||||
add_ui(R_T7, R_T7, 20),
|
|
||||||
|
|
||||||
mac_yield()
|
|
||||||
};
|
|
||||||
|
|
||||||
// TODO(Ed): Reduce magic numbers/offsets
|
|
||||||
/* DIAGNOSTIC 3: Pure GTE test (No Memory Writes) */
|
|
||||||
internal MipsAtom_(diag_gte) {
|
|
||||||
/* Load 3 indices */
|
|
||||||
load_half_u(R_T0, R_T4, 0),
|
|
||||||
load_half_u(R_T1, R_T4, 2),
|
|
||||||
load_half_u(R_T2, R_T4, 4),
|
|
||||||
|
|
||||||
/* Load Vertices into GTE */
|
|
||||||
shift_lleft( R_AT, R_T0, 3), add_u( R_AT, R_AT, R_T5),
|
|
||||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
|
||||||
gte_mv_to_data_r( R_V0, C2_VXY0), gte_mv_to_data_r( R_V1, C2_VZ0),
|
|
||||||
|
|
||||||
shift_lleft( R_AT, R_T1, 3), add_u( R_AT, R_AT, R_T5),
|
|
||||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
|
||||||
gte_mv_to_data_r( R_V0, C2_VXY1), gte_mv_to_data_r( R_V1, C2_VZ1),
|
|
||||||
|
|
||||||
shift_lleft( R_AT, R_T2, 3), add_u( R_AT, R_AT, R_T5),
|
|
||||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
|
||||||
gte_mv_to_data_r( R_V0, C2_VXY2), gte_mv_to_data_r( R_V1, C2_VZ2),
|
|
||||||
|
|
||||||
/* Run Math */
|
|
||||||
nop, nop, gte_cmdw_rtpt,
|
|
||||||
nop, nop, gte_cmdw_nclip,
|
|
||||||
nop, nop,
|
|
||||||
|
|
||||||
/* Advance Face Cursor and Yield */
|
|
||||||
add_ui(R_T4, R_T4, 8),
|
|
||||||
|
|
||||||
mac_yield()
|
|
||||||
};
|
|
||||||
|
|
||||||
#pragma endregion Baked Mips Atoms
|
#pragma endregion Baked Mips Atoms
|
||||||
|
|||||||
@@ -0,0 +1,96 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
# include "math.h"
|
||||||
|
# include "lottes_tape.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
ATOM_FILE_DEBUGGER_LINE_MARKER(math_atom_c);
|
||||||
|
|
||||||
|
#define v3s4_R_0() ((Reg_(V3_S4)){R_0,R_0,R_0})
|
||||||
|
|
||||||
|
typedef Struct_(Reg_V3_S2) { Reg x, y, z; };
|
||||||
|
typedef Struct_(Reg_V3_S4) { Reg x, y, z; }; // Register allocation of a V3_S4
|
||||||
|
typedef Struct_(Reg_P3_S4) { Reg x, y, z; }; // Register allocation of a P3_S4
|
||||||
|
|
||||||
|
#pragma region MACs (Mips Atom Component)
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_load_half_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
load_half(tx, base, offset + OA_(U2,[0])),
|
||||||
|
load_half(ty, base, offset + OA_(U2,[1])),
|
||||||
|
load_half(tz, base, offset + OA_(U2,[2])),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_load_v3s2(AtomBuilder_R ab, Reg_(V3_S2) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_half_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_load_v2s2(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
load_half(rs_x, r_base, offset + O_(V3_S2,x)),
|
||||||
|
load_half(rs_y, r_base, offset + O_(V3_S2,y)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_store_v2s2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
store_half(rt_x, base, offset + O_(V2_S2,x)),
|
||||||
|
store_half(rt_y, base, offset + O_(V2_S2,y)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_load_word_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
load_word(tx, base, offset + OA_(U4,[0])),
|
||||||
|
load_word(ty, base, offset + OA_(U4,[1])),
|
||||||
|
load_word(tz, base, offset + OA_(U4,[2])),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_load_v3s4(AtomBuilder_R ab, Reg_(V3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||||
|
FI_ Slice_MipsCode ac_load_p3s4(AtomBuilder_R ab, Reg_(P3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_store_half_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
store_half(tx, base, offset + OA_(U2,[0])),
|
||||||
|
store_half(ty, base, offset + OA_(U2,[1])),
|
||||||
|
store_half(tz, base, offset + OA_(U2,[2])),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_store_v3s2(AtomBuilder_R ab, Reg_(V3_S2) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_store_half_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_store_word_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
store_word(tx, base, offset + OA_(U4,[0])),
|
||||||
|
store_word(ty, base, offset + OA_(U4,[1])),
|
||||||
|
store_word(tz, base, offset + OA_(U4,[2])),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_store_v3s4(AtomBuilder_R ab, Reg_(V3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||||
|
FI_ Slice_MipsCode ac_store_p3s4(AtomBuilder_R ab, Reg_(P3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_add_si_v3s4(AtomBuilder_R ab, Reg rt_x, Reg rt_y, Reg rt_z, Reg base, U2 offset)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
add_si(rt_x, base, O_(V3_S4,x)),
|
||||||
|
add_si(rt_y, base, O_(V3_S4,y)),
|
||||||
|
add_si(rt_z, base, O_(V3_S4,z)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_sub_s_v3(AtomBuilder_R ab
|
||||||
|
, Reg dx, Reg dy, Reg dz
|
||||||
|
, Reg sx, Reg sy, Reg sz
|
||||||
|
, Reg tx, Reg ty, Reg tz
|
||||||
|
) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
sub_s(dx, sx, tx),
|
||||||
|
sub_s(dy, sy, ty),
|
||||||
|
sub_s(dz, sz, tz),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_sub_v3s4(AtomBuilder_R ab, Reg_(V3_S4) d, Reg_(V3_S4) s, Reg_(V3_S4) t) MipsAtomComp_ProcMap_(ab, mac_sub_s_v3(d.x, d.y, d.z, s.x, s.y, s.z, t.x, t.y, t.z))
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_sub_s_v3_self(AtomBuilder_R ab, Reg ds_x, Reg ds_y, Reg ds_z, Reg tx, Reg ty, Reg tz) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
sub_s(ds_x, ds_x, tx),
|
||||||
|
sub_s(ds_y, ds_y, ty),
|
||||||
|
sub_s(ds_z, ds_z, tz),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_sub_v3s4_self(AtomBuilder_R ab, Reg_(V3_S4) ds, Reg_(V3_S4) t) MipsAtomComp_ProcMap_(ab, mac_sub_s_v3_self(ds.x, ds.y, ds.z, t.x, t.y, t.z))
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_store_rects2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
store_half(rt_x, base, offset + O_(Rect_S2,x)),
|
||||||
|
store_half(rt_y, base, offset + O_(Rect_S2,y)),
|
||||||
|
store_half(rt_width, base, offset + O_(Rect_S2,width)),
|
||||||
|
store_half(rt_height, base, offset + O_(Rect_S2,height)),
|
||||||
|
})
|
||||||
|
|
||||||
|
#pragma endregion MACs (Mips Atom Component)
|
||||||
+65
-11
@@ -7,10 +7,24 @@
|
|||||||
#define max(A, B) (((A) > (B)) ? (A) : (B))
|
#define max(A, B) (((A) > (B)) ? (A) : (B))
|
||||||
#define clamp_bot(X, B) max(X, B)
|
#define clamp_bot(X, B) max(X, B)
|
||||||
|
|
||||||
|
/* Convention
|
||||||
|
<Type> ## <Width> _ <Component Type> ## <Component Width>
|
||||||
|
For types with compound data (Ex: Rotation Matrix & Translation):
|
||||||
|
<TypeA> ## <TypeB> ## <Width> _ <ComponentTypeA> ## <ComponentWidthA> ## <ComponentTypeB> ## <ComponentWidthB>
|
||||||
|
|
||||||
|
A: Array
|
||||||
|
V: Vector
|
||||||
|
R: Range
|
||||||
|
M: Matrix
|
||||||
|
T: Translation
|
||||||
|
*/
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
v3s2_byteoff = 3, // log2(8), used with shift_left_logical op for index via byte offset.
|
v3s2_byteoff = 3, // log2(8), used with shift_left_logical op for index via byte offset.
|
||||||
};
|
};
|
||||||
|
|
||||||
|
typedef Array_(U1, 2);
|
||||||
|
typedef Array_(U2, 2);
|
||||||
typedef Array_(U4, 2);
|
typedef Array_(U4, 2);
|
||||||
typedef Array_(S2, 2);
|
typedef Array_(S2, 2);
|
||||||
typedef Array_(S2, 3);
|
typedef Array_(S2, 3);
|
||||||
@@ -22,24 +36,46 @@ typedef S2 A3x3_S2[3][3];
|
|||||||
typedef Struct_(Extent2_S2) { S2 width; S2 height; };
|
typedef Struct_(Extent2_S2) { S2 width; S2 height; };
|
||||||
typedef Struct_(Extent2_S4) { S4 width; S4 height; };
|
typedef Struct_(Extent2_S4) { S4 width; S4 height; };
|
||||||
|
|
||||||
|
typedef Struct_(V2_U1) { U1 x; U1 y; };
|
||||||
typedef Struct_(V2_S2) { S2 x; S2 y; };
|
typedef Struct_(V2_S2) { S2 x; S2 y; };
|
||||||
typedef Struct_(V2_S4) { S4 x; S4 y; };
|
typedef Struct_(V2_S4) { S4 x; S4 y; };
|
||||||
typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; };
|
typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; }; // PSY-Q: SVECTOR
|
||||||
typedef Struct_(V3_S4) { S4 x; S4 y; S4 z; S4 pad; };
|
typedef Struct_(V3_S4) { S4 x; S4 y; S4 z; S4 pad; }; // PSY-Q: VECTOR. RGA(Lengyel): Euclidean vector or direction. A zero-weight RGA point is stored as a V3_S4 with the implicit weight dropped.
|
||||||
typedef Struct_(V4_S2) { S2 x; S2 y; S2 z; S2 w; };
|
typedef Struct_(V4_S2) { S2 x; S2 y; S2 z; S2 w; };
|
||||||
typedef Struct_(V4_S4) { S4 x; S4 y; S4 z; S4 w; };
|
typedef Struct_(V4_S4) { S4 x; S4 y; S4 z; S4 w; };
|
||||||
|
|
||||||
typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; };
|
// typedef Struct_(P3_S4) { S4 x; S4 y; S4 z; S4 w1; }; // RGA(Lengyel): Affine point with implicit weight one. Storage alias of V3_S4. Use P3_S4 when the value is a point.
|
||||||
typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; };
|
typedef V3_S4 P3_S4;
|
||||||
|
|
||||||
typedef Struct_(Rect_S2) { S2 x; S2 y; S2 width; S2 height; };
|
typedef Struct_(R1_U2) { U2 p0; U2 p1; };
|
||||||
typedef Struct_(Rect_S4) { S4 x; S4 y; S4 width; S4 height; };
|
typedef Struct_(R1_S2) { S2 p0; S2 p1; };
|
||||||
|
|
||||||
typedef Struct_(M3_S2) { A3x3_S2 m; A3_S4 t; };
|
typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; }; // Range-2 Signed 2-Byte (16-bit)
|
||||||
|
typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; }; // Range-2 Signed 4-Byte (32-bit)
|
||||||
|
|
||||||
|
typedef Struct_(Rect_S2) { S2 x; S2 y; S2 width; S2 height; };
|
||||||
|
typedef Struct_(Rect_S4) { S4 x; S4 y; S4 width; S4 height; };
|
||||||
|
|
||||||
|
typedef Struct_(MT3_S2S4) { A3x3_S2 m; A3_S4 t; }; // PSY-Q: MATRIX. RGA(Lengyel): Matrix expansion of a rigid transformation. GTE utilizes this representation; corresponding motor not constructed here.
|
||||||
|
|
||||||
|
/* RGA(Lengyel) reserved names (deferred):
|
||||||
|
* P4_S4 - future flat point with explicit weight (Lengyel/TML FlatPoint3D analog).
|
||||||
|
* B3_S4 - future 3D bivector (callers store a Complement(Wedge(...)) as a V3_S4).
|
||||||
|
* Mo8_S4 - future motor. Not introduced until a course operation actually needs composition, interpolation, or inversion. */
|
||||||
|
|
||||||
|
typedef Array_(V2_U1, 2);
|
||||||
|
typedef Array_(V2_S2, 2);
|
||||||
typedef Array_(V2_S2, 3);
|
typedef Array_(V2_S2, 3);
|
||||||
typedef Array_(V2_S2, 4);
|
typedef Array_(V2_S2, 4);
|
||||||
|
|
||||||
|
#define r1u2(p0,p1) (R1_U2){p0,p1}
|
||||||
|
|
||||||
|
enum {
|
||||||
|
fp_one = (1 << 12),
|
||||||
|
};
|
||||||
|
|
||||||
|
#define v3s4_fp_one() v3s4(fp_one, fp_one, fp_one)
|
||||||
|
|
||||||
#define v2s2(x,y) (V2_S2){x,y}
|
#define v2s2(x,y) (V2_S2){x,y}
|
||||||
#define v3s2(x,y,z) (V3_S2){x,y,z,0}
|
#define v3s2(x,y,z) (V3_S2){x,y,z,0}
|
||||||
#define v3s4(x,y,z) (V3_S4){x,y,z,0}
|
#define v3s4(x,y,z) (V3_S4){x,y,z,0}
|
||||||
@@ -58,10 +94,28 @@ FI_ void add_a3s4_fp(A3_S4_R out_a, A3_S4 b) {
|
|||||||
(out_a[0])[2] += b[2] >> 1;
|
(out_a[0])[2] += b[2] >> 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
FI_ void add_v3s4(V3_S4_R out_a, V3_S4 b) {
|
FI_ void sub_a3s4(A3_S4_R out_a, A3_S4 b) {
|
||||||
add_a3s4(pcast(A3_S4_R, out_a), pcast(A3_S4, b));
|
(out_a[0])[0] -= b[0];
|
||||||
|
(out_a[0])[1] -= b[1];
|
||||||
|
(out_a[0])[2] -= b[2];
|
||||||
}
|
}
|
||||||
|
|
||||||
FI_ void add_v3s4_fp(V3_S4_R out_a, V3_S4 b) {
|
FI_ void sub_a3s4_fp(A3_S4_R out_a, A3_S4 b) {
|
||||||
add_a3s4_fp(pcast(A3_S4_R, out_a), pcast(A3_S4, b));
|
(out_a[0])[0] -= b[0] >> 1;
|
||||||
|
(out_a[0])[1] -= b[1] >> 1;
|
||||||
|
(out_a[0])[2] -= b[2] >> 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
FI_ void mul_a3s4(A3_S4_R out_a, A3_S4 b) {
|
||||||
|
(out_a[0])[0] *= b[0];
|
||||||
|
(out_a[0])[1] *= b[1];
|
||||||
|
(out_a[0])[2] *= b[2];
|
||||||
|
}
|
||||||
|
|
||||||
|
FI_ void add_v3s4 (V3_S4_R out_a, V3_S4 b) { add_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
||||||
|
FI_ void add_v3s4_fp(V3_S4_R out_a, V3_S4 b) { add_a3s4_fp(C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
||||||
|
|
||||||
|
FI_ void sub_v3s4 (V3_S4_R out_a, V3_S4 b) { sub_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
||||||
|
FI_ void sub_v3s4_fp(V3_S4_R out_a, V3_S4 b) { sub_a3s4_fp(C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
||||||
|
|
||||||
|
FI_ void mul_v3s4 (V3_S4_R out_a, V3_S4 b) { mul_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
||||||
|
|||||||
+36
-14
@@ -18,7 +18,7 @@ I_ U4 align_pow2(U4 x, U4 b) {
|
|||||||
|
|
||||||
#define align_struct(type_width) ((U4)(((type_width) + 3) & ~3))
|
#define align_struct(type_width) ((U4)(((type_width) + 3) & ~3))
|
||||||
|
|
||||||
FI_ void mem_bump(U4 start, U4 cap, U4*R_ used, U4 amount) {
|
FI_ void mem_bump(U4 cap, U4*R_ used, U4 amount) {
|
||||||
assert(amount <= (cap - used[0]));
|
assert(amount <= (cap - used[0]));
|
||||||
used[0] += amount;
|
used[0] += amount;
|
||||||
}
|
}
|
||||||
@@ -58,36 +58,44 @@ typedef Struct_(Str8) { UTF8* ptr; U4 len; };
|
|||||||
typedef Struct_(Slice_Str8) { Str8* ptr; U4 len; };
|
typedef Struct_(Slice_Str8) { Str8* ptr; U4 len; };
|
||||||
#define slit(string_literal) (Str8){ (UTF8*) string_literal, S_(string_literal) - 1 }
|
#define slit(string_literal) (Str8){ (UTF8*) string_literal, S_(string_literal) - 1 }
|
||||||
|
|
||||||
typedef Struct_(Slice) { U4 ptr, len; }; // Untyped Slice
|
typedef Struct_(Slice) { B1* ptr; U4 len; }; // Untyped Slice (byte-addressable; .len in elements)
|
||||||
FI_ Slice slice_ut_(U4 ptr, U4 len) { return (Slice){ptr, len}; }
|
FI_ Slice slice_ut_(U4 ptr, U4 len) { return (Slice){(B1*)ptr, len}; }
|
||||||
|
|
||||||
#define Slice_(type) Struct_(tmpl(Slice,type)) { type* ptr; U4 len; }
|
#define Slice_(type) Struct_(tmpl(Slice,type)) { type* ptr; U4 len; }
|
||||||
typedef Slice_(B1);
|
typedef Slice_(B1);
|
||||||
#define slice_assert(s) do { assert((s).ptr != 0); assert((s).len > 0); } while(0)
|
#define slice_assert(s) do { assert((s).ptr != 0); assert((s).len > 0); } while(0)
|
||||||
#define slice_end(slice) ((slice).ptr + (slice).len)
|
#define slice_end(slice) ((slice).ptr + S_slice(slice) / S_(B1)) /* byte-ptr arithmetic; .len is in elements per slice convention */
|
||||||
#define S_slice(s) ((s).len * S_((s).ptr[0]))
|
#define S_slice(s) ((s).len * S_((s).ptr[0]))
|
||||||
|
|
||||||
#define slice_ut(ptr,len) slice_ut_(u4_(ptr), u4_(len))
|
#define slice_ut(ptr,len) slice_ut_(u4_(ptr), u4_(len))
|
||||||
#define slice_ut_arr(a) slice_ut_(u4_(a), S_(a))
|
#define slice_ut_arr(a) slice_ut_(u4_(a), S_(a))
|
||||||
#define slice_to_ut(s) slice_ut_(u4_((s).ptr), S_slice(s))
|
#define slice_to_ut(s) slice_ut_(u4_((s).ptr), S_slice(s))
|
||||||
|
|
||||||
#define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter)
|
#define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter)
|
||||||
#define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = array_decl(type,__VA_ARGS__), .len = array_len( array_decl(type,__VA_ARGS__)) }
|
#define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = Array_decl(type,__VA_ARGS__), .len = Array_len( Array_decl(type,__VA_ARGS__)) }
|
||||||
|
#define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = Array_len(array) }
|
||||||
|
|
||||||
FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(s.ptr, s.len); }
|
FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(u4_(s.ptr), s.len); }
|
||||||
#define slice_zero(s) slice_zero_(slice_to_ut(s))
|
#define slice_zero(s) slice_zero_(slice_to_ut(s))
|
||||||
|
|
||||||
FI_ void slice_copy_(Slice dest, Slice src) {
|
FI_ void slice_copy_(Slice dest, Slice src) {
|
||||||
assert(dest.len >= src.len);
|
assert(S_slice(dest) >= S_slice(src));
|
||||||
slice_assert(dest);
|
slice_assert(dest);
|
||||||
slice_assert(src);
|
slice_assert(src);
|
||||||
mem_copy(dest.ptr, src.ptr, src.len);
|
mem_copy(u4_(dest.ptr), u4_(src.ptr), S_slice(src));
|
||||||
}
|
}
|
||||||
#define slice_copy(dest, src) do { \
|
#define slice_copy(dest, src) do { \
|
||||||
static_assert(T_same(dest, src)); \
|
static_assert(T_same(dest, src)); \
|
||||||
slice_copy_(slice_to_ut(dest), slice_to_ut(src)); \
|
slice_copy_(slice_to_ut(dest), slice_to_ut(src)); \
|
||||||
} while(0)
|
} while(0)
|
||||||
|
|
||||||
|
FI_ Slice slice_bump(U4_R used, U4 start, U4 len, U4 amount) {
|
||||||
|
assert(len - used[0] - amount);
|
||||||
|
U4 ptr = start + used[0]; used[0] += amount;
|
||||||
|
return slice_ut(ptr, amount);
|
||||||
|
}
|
||||||
|
|
||||||
|
typedef Slice_(U1);
|
||||||
typedef Slice_(U4);
|
typedef Slice_(U4);
|
||||||
|
|
||||||
#pragma endregion Slice
|
#pragma endregion Slice
|
||||||
@@ -97,18 +105,19 @@ typedef Slice_(U4);
|
|||||||
typedef Opt_(farena) { U4 alignment, type_width; };
|
typedef Opt_(farena) { U4 alignment, type_width; };
|
||||||
typedef Struct_(FArena) { U4 start, capacity, used; };
|
typedef Struct_(FArena) { U4 start, capacity, used; };
|
||||||
FI_ void farena_init(FArena_R arena, Slice mem) { assert(arena != nullptr);
|
FI_ void farena_init(FArena_R arena, Slice mem) { assert(arena != nullptr);
|
||||||
arena->start = mem.ptr;
|
arena->start = u4_(mem.ptr);
|
||||||
arena->capacity = mem.len;
|
arena->capacity = mem.len;
|
||||||
arena->used = 0;
|
arena->used = 0;
|
||||||
}
|
}
|
||||||
FI_ FArena farena_make(Slice mem) { FArena a; farena_init(& a, mem); return a; }
|
FI_ FArena farena_make(Slice mem) { FArena a; farena_init(& a, mem); return a; }
|
||||||
I_ Slice farena_push(FArena_R arena, U4 amount, Opt_farena o) {
|
FI_ Slice farena_bump(FArena_R a, U4 amount) { return slice_bump(& a->used, a->start, a->capacity, amount); }
|
||||||
|
I_ Slice farena_push(FArena_R arena, U4 amount, Opt_farena o) {
|
||||||
if (amount == 0) { return (Slice){}; }
|
if (amount == 0) { return (Slice){}; }
|
||||||
U4 desired = amount * (o.type_width == 0 ? 1 : o.type_width);
|
U4 desired = amount * (o.type_width == 0 ? 1 : o.type_width);
|
||||||
U4 to_commit = align_pow2(desired, o.alignment ? o.alignment : MEM_ALIGNMENT_DEFAULT);
|
U4 to_commit = align_pow2(desired, o.alignment ? o.alignment : MEM_ALIGNMENT_DEFAULT);
|
||||||
U4 ptr = arena->start + arena->used;
|
U4 ptr = arena->start + arena->used;
|
||||||
mem_bump(arena->start, arena->capacity, & arena->used, to_commit);
|
mem_bump(arena->capacity, & arena->used, to_commit);
|
||||||
return (Slice){ ptr, to_commit };
|
return (Slice){ (B1*)ptr, to_commit };
|
||||||
}
|
}
|
||||||
FI_ void farena_reset (FArena_R arena) { arena->used = 0; }
|
FI_ void farena_reset (FArena_R arena) { arena->used = 0; }
|
||||||
FI_ void farena_rewind(FArena_R arena, U4 save_point) {
|
FI_ void farena_rewind(FArena_R arena, U4 save_point) {
|
||||||
@@ -116,8 +125,21 @@ FI_ void farena_rewind(FArena_R arena, U4 save_point) {
|
|||||||
arena->used -= save_point - arena->start;
|
arena->used -= save_point - arena->start;
|
||||||
}
|
}
|
||||||
FI_ U4 farena_save(FArena arena) { return arena.used; }
|
FI_ U4 farena_save(FArena arena) { return arena.used; }
|
||||||
|
FI_ U4 farena_unused_start(FArena arena) { return arena.start + arena.used; }
|
||||||
#define farena_push_(arena, amount, ...) farena_push((arena), (amount), opt_(farena, __VA_ARGS__))
|
#define farena_push_(arena, amount, ...) farena_push((arena), (amount), opt_(farena, __VA_ARGS__))
|
||||||
#define farena_push_type(arena, type, ...) C_(type*, farena_push((arena), 1, opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr)
|
#define farena_push_type(arena, type, ...) C_(type*, farena_push((arena), 1, opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr)
|
||||||
#define farena_push_array(arena, type, amount, ...) (tmpl(Slice,type)){ C_(type*, farena_push((arena), (amount), opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr), (amount) }
|
#define farena_push_array(arena, type, amount, ...) (tmpl(Slice,type)){ C_(type*, farena_push((arena), (amount), opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr), (amount) }
|
||||||
|
|
||||||
#pragma endregion FArena
|
#pragma endregion FArena
|
||||||
|
|
||||||
|
#pragma region BIOS Scratchpad
|
||||||
|
/* BIOS scratchpad location. 1 KB at 0x1F800000.
|
||||||
|
* TapeHostFrame occupies the final 44 bytes while tape code executes.
|
||||||
|
* Atom scratch is bounded by the TapeHostFrame_Loc declaration in lottes_tape.h. */
|
||||||
|
enum {
|
||||||
|
Scratchpad_Loc = 0x1F800000,
|
||||||
|
Scratchpad_Len = 0x400, /* 1 KB */
|
||||||
|
Scratchpad_End = Scratchpad_Loc + Scratchpad_Len, /* 0x1F800400 */
|
||||||
|
};
|
||||||
|
#define C_scratch(type) C_(type, Scratchpad_Loc)
|
||||||
|
#pragma endregion BIOS Scratchpad
|
||||||
|
|||||||
@@ -0,0 +1,80 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
# include "bios.h"
|
||||||
|
# include "mips.h"
|
||||||
|
# include "lottes_tape.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
ATOM_FILE_DEBUGGER_LINE_MARKER(mips_atom_c);
|
||||||
|
|
||||||
|
#pragma region MACs (Mips Atom Components)
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_load_word_imm(AtomBuilder_R ab, Reg dst, U4 imm)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
load_upper_i(dst, u4_hi(imm)),
|
||||||
|
or_i_self( dst, u4_lo(imm)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_shift_aright_v3_self(AtomBuilder_R ab, Reg dt_x, Reg dt_y, Reg dt_z, U2 shift_amount)
|
||||||
|
MipsAtomComp_Proc_( ab, {
|
||||||
|
shift_aright(dt_x, dt_x, shift_amount),
|
||||||
|
shift_aright(dt_y, dt_y, shift_amount),
|
||||||
|
shift_aright(dt_z, dt_z, shift_amount),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_shift_aright_v3s4_self(AtomBuilder_R ab, Reg_(V3_S4) dt, U2 shift) MipsAtomComp_ProcMap_(ab, mac_shift_aright_v3_self(dt.x, dt.y, dt.z, shift))
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_shift_aright_var_v3(AtomBuilder_R ab
|
||||||
|
, Reg rd_v0, Reg rd_v1, Reg rd_v2
|
||||||
|
, Reg rs_v0, Reg rs_v1, Reg rs_v2
|
||||||
|
, Reg r_shift)
|
||||||
|
MipsAtomComp_Proc_(ab, {
|
||||||
|
shift_aright_var(rd_v0, rs_v0, r_shift),
|
||||||
|
shift_aright_var(rd_v1, rs_v1, r_shift),
|
||||||
|
shift_aright_var(rd_v2, rs_v2, r_shift),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_shift_aright_var_v3_self(AtomBuilder_R ab, Reg rds_v0, Reg rds_v1, Reg rds_v2, Reg r_shift)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
shift_aright_var(rds_v0, rds_v0, r_shift),
|
||||||
|
shift_aright_var(rds_v1, rds_v1, r_shift),
|
||||||
|
shift_aright_var(rds_v2, rds_v2, r_shift),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_shift_aright_var_v3s4_self(AtomBuilder_R ab, Reg_(V3_S4) ds, Reg shift) MipsAtomComp_ProcMap_(ab, mac_shift_aright_var_v3_self(ds.x, ds.y, ds.z, shift))
|
||||||
|
|
||||||
|
#pragma endregion MACs (Mips Atom Components)
|
||||||
|
|
||||||
|
#pragma region Baked Atoms
|
||||||
|
|
||||||
|
/* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0).
|
||||||
|
* Sequence (per MIPS ABI; arguments in arg registers, RA pushed to stack):
|
||||||
|
* 1. sp -= 8; sw $ra, 4($sp) ; save RA
|
||||||
|
* 2. $a0 = bios_flushcache (arg0)
|
||||||
|
* 3. $t0 = bios_table_addr ; t0 = &BIOS A-function table
|
||||||
|
* 4. jalr $t0, $ra ; call BIOS(flushcache)
|
||||||
|
* nop ; branch delay slot
|
||||||
|
* 5. lw $ra, 4($sp)
|
||||||
|
* 6. sp += 8 ; load-delay
|
||||||
|
* 7. jr $ra
|
||||||
|
* nop ; BD
|
||||||
|
*/
|
||||||
|
|
||||||
|
#if 0
|
||||||
|
// Note: Can't do this without having a way to do C-Runtime frame call from Tape ABI.
|
||||||
|
// Don't support this without adjusting scratchpad to save tape frame in some way.
|
||||||
|
internal MipsAtom_(mips_flush_icache) {
|
||||||
|
add_ui(R_SP, R_SP, -MipsStackAlignment), // sp -= 8
|
||||||
|
store_word(R_RA, R_SP, S_(U4)), // sw $ra, 4($sp)
|
||||||
|
add_ui(R_V0, R_0, bios_flushcache), // addiu $a0, $0, 0x44
|
||||||
|
add_ui(R_T0, R_0, bios_table_addr), // addiu $t0, $0, 0xA0
|
||||||
|
jump_link(R_T0, R_RA), nop, // jalr $t0, $ra, BD slot
|
||||||
|
load_word(R_RA, R_SP, S_(U4)), // lw $ra, 4($sp)
|
||||||
|
add_ui(R_SP, R_SP, MipsStackAlignment), // sp += 8 (load-delay)
|
||||||
|
jump_reg(R_RA), nop, // jr $ra, BD slot
|
||||||
|
// mac_yield(),
|
||||||
|
};
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#pragma endregion Baked Atoms
|
||||||
+162
-170
@@ -1,38 +1,28 @@
|
|||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
* duffle DSL Suffix Conventions
|
* duffle DSL Suffix Conventions
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
*
|
|
||||||
* Every mnemonic in this header follows the same suffix grammar:
|
* Every mnemonic in this header follows the same suffix grammar:
|
||||||
*
|
* _i: Immediate value (16-bit constant operand).
|
||||||
* _i Immediate value (16-bit constant operand). Combine with
|
* Combine with _u or _s (single-letter modifier + type combined): add_ui, add_si.
|
||||||
* _u or _s (single-letter modifier + type combined): add_ui,
|
* Examples: add_ui, add_si, and_i, or_i, xor_i, load_upper_i. and_i is sign-agnostic (andi zero-extends).
|
||||||
* add_si. Examples: add_ui, add_si, and_i, or_i, xor_i,
|
* load_upper_i is a unique verb; _i is the immediate marker, not a modifier+type combination.
|
||||||
* load_upper_i. and_i is sign-agnostic (andi zero-extends).
|
* _u: Unsigned (no-overflow, no-sign-extension).
|
||||||
* load_upper_i is a unique verb; _i is the immediate marker,
|
* R-type arithmetic examples: add_u, sub_u, mult_u, div_u. I-type (combined with _i): add_ui.
|
||||||
* not a modifier+type combination.
|
* _s: Signed (overflow-traps, sign-extends).
|
||||||
*
|
* R-type: add_s, sub_s, mult_s, div_s, set_lt_s. I-type (combined with _i): add_si.
|
||||||
* _u Unsigned (no-overflow, no-sign-extension). R-type
|
|
||||||
* arithmetic examples: add_u, sub_u, mult_u, div_u. I-type
|
|
||||||
* (combined with _i): add_ui.
|
|
||||||
*
|
|
||||||
* _s Signed (overflow-traps, sign-extends). R-type: add_s,
|
|
||||||
* sub_s, mult_s, div_s, set_lt_s. I-type (combined with _i):
|
|
||||||
* add_si.
|
|
||||||
*
|
*
|
||||||
* --- Shift family (R-type): verb-modifier-direction ---
|
* --- Shift family (R-type): verb-modifier-direction ---
|
||||||
* The shift macros use `shift_<modifier><direction>`. Modifier is
|
* The shift macros use `shift_<modifier><direction>`.
|
||||||
* the single letter `l` (logical) or `a` (arithmetic). Direction
|
* Modifier is the single letter `l` (logical) or `a` (arithmetic).
|
||||||
* is the word `left` or `right`. Combined: `_lleft`, `_lright`,
|
* Direction is the word `left` or `right`. Combined: `_lleft`, `_lright`, `_aright`.
|
||||||
* `_aright`. Examples: shift_lleft( rd, rt, shamt) (= sll)
|
* Examples: shift_lleft( rd, rt, shamt) (= sll)
|
||||||
* shift_lright(rd, rt, shamt) (= srl)
|
* shift_lright(rd, rt, shamt) (= srl)
|
||||||
* shift_aright(rd, rt, shamt) (= sra)
|
* shift_aright(rd, rt, shamt) (= sra)
|
||||||
* (no `_aleft`; MIPS has no `sla` — arithmetic-left is bit-identical
|
* (no `_aleft`; MIPS has no `sla` — arithmetic-left is bit-identical to logical-left, so use shift_lleft for that case)
|
||||||
* to logical-left, so use shift_lleft for that case)
|
|
||||||
*
|
*
|
||||||
* --- Jump/Call family ---
|
* --- Jump/Call family ---
|
||||||
* Simple jumps keep the original short names: jump (j), jump_reg
|
* Simple jumps keep the original short names: jump (j), jump_reg (jr), jump_link (jalr rs, rd).
|
||||||
* (jr), jump_link (jalr rs, rd). The jump-and-link-to variants
|
* The jump-and-link-to variants (jal, jalr rs with default $ra) get the `call_` verb instead:
|
||||||
* (jal, jalr rs with default $ra) get the `call_` verb instead:
|
|
||||||
* call_addr (jal), call_reg (jalr rs, default $ra).
|
* call_addr (jal), call_reg (jalr rs, default $ra).
|
||||||
* Examples: jump(off) (= j)
|
* Examples: jump(off) (= j)
|
||||||
* jump_reg(rs) (= jr)
|
* jump_reg(rs) (= jr)
|
||||||
@@ -40,32 +30,22 @@
|
|||||||
* call_reg(rs) (= jalr rs, default $ra)
|
* call_reg(rs) (= jalr rs, default $ra)
|
||||||
* call_addr(off) (= jal)
|
* call_addr(off) (= jal)
|
||||||
*
|
*
|
||||||
* _r Register marker — used only when the register type needs
|
* _r: Register marker — used only when the register type needs disambiguation (e.g., GTE data register vs control register).
|
||||||
* disambiguation (e.g., GTE data register vs control
|
* NOT used in plain R-type arithmetic (the R-type is implicit). Examples: gte_mv_to_data_r, gte_mv_to_ctrl_r.
|
||||||
* register). NOT used in plain R-type arithmetic (the
|
* _self: Destination equals one source operand.
|
||||||
* R-type is implicit). Examples: gte_mv_to_data_r,
|
* Examples: add_ui_self (I-type, to self), add_u_self (R-type, to self).
|
||||||
* gte_mv_to_ctrl_r.
|
* _mv_to_: Direction: data flows into X.
|
||||||
|
* Example: gte_mv_to_data_r, gte_mv_to_ctrl_r.
|
||||||
|
* _mv_from_: Direction: data flows out of X.
|
||||||
|
* Example: gte_mv_from_data_r, gte_mv_from_ctrl_r.
|
||||||
|
* _str: String-form — emits inline-asm string instead of `.word`.
|
||||||
|
* Example: gte_rtpt_asm_str.
|
||||||
|
* _2w / _1w: Word count of the emitted sequence.
|
||||||
|
* Example: load_imm_2w.
|
||||||
*
|
*
|
||||||
* _self Destination equals one source operand.
|
* _cop2: RESERVED — DO NOT USE in macro names. The `gte_` namespace prefix already implies coprocessor 2. Use `c2` only in:
|
||||||
* Examples: add_ui_self (I-type, to self),
|
* (a) integer opcode enums (op_lwc2 = 0x32, op_swc2 = 0x3A)
|
||||||
* add_u_self (R-type, to self).
|
* (b) vendor-mnemonic macro aliases (gte_mtc2, gte_mfc2)
|
||||||
*
|
|
||||||
* _mv_to_ Direction: data flows into X.
|
|
||||||
* Example: gte_mv_to_data_r, gte_mv_to_ctrl_r.
|
|
||||||
*
|
|
||||||
* _mv_from_ Direction: data flows out of X.
|
|
||||||
* Example: gte_mv_from_data_r, gte_mv_from_ctrl_r.
|
|
||||||
*
|
|
||||||
* _str String-form — emits inline-asm string instead of `.word`.
|
|
||||||
* Example: gte_rtpt_asm_str.
|
|
||||||
*
|
|
||||||
* _2w / _1w Word count of the emitted sequence.
|
|
||||||
* Example: load_imm_2w.
|
|
||||||
*
|
|
||||||
* _cop2 RESERVED — DO NOT USE in macro names. The `gte_` namespace
|
|
||||||
* prefix already implies coprocessor 2. Use `c2` only in:
|
|
||||||
* (a) integer opcode enums (op_lwc2 = 0x32, op_swc2 = 0x3A)
|
|
||||||
* (b) vendor-mnemonic macro aliases (gte_mtc2, gte_mfc2)
|
|
||||||
*
|
*
|
||||||
* Primitive commands: gp0_cmd_poly_f3 = 0x20 (byte opcode)
|
* Primitive commands: gp0_cmd_poly_f3 = 0x20 (byte opcode)
|
||||||
* Packed 32-bit cmd: gp0_word_poly_f3(r, g, b) (32-bit, shifted)
|
* Packed 32-bit cmd: gp0_word_poly_f3(r, g, b) (32-bit, shifted)
|
||||||
@@ -80,9 +60,8 @@
|
|||||||
* gte_lw_v0_xy(base) (gte + lw + v0 + xy)
|
* gte_lw_v0_xy(base) (gte + lw + v0 + xy)
|
||||||
* load_upper_i (load-upper + immediate, unique verb)
|
* load_upper_i (load-upper + immediate, unique verb)
|
||||||
*
|
*
|
||||||
* Vendor mnemonics (sll, srl, sra, jr, j, jal, jalr) are NOT in this
|
* Vendor mnemonics (sll, srl, sra, jr, j, jal, jalr) are NOT in this header.
|
||||||
* header. They live in the opt-in `mips_vendor_sym.h` for users who
|
* They live in the opt-in `mips_vendor_sym.h` for users who prefer the textbook MIPS assembly mnemonics.
|
||||||
* prefer the textbook MIPS assembly mnemonics.
|
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
|
|
||||||
#ifdef INTELLISENSE_DIRECTIVES
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
@@ -98,19 +77,17 @@ enum {
|
|||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
* REGISTER INTEGER IDS (preprocessor-visible)
|
* REGISTER INTEGER IDS (preprocessor-visible)
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
* Every R_* enum below has a parallel R_*_Code `#define` so that the
|
* Every R_* enum below has a parallel R_*_Code `#define` so that the preprocessor can stringify the integer
|
||||||
* preprocessor can stringify the integer (e.g. for asm clobber lists and
|
* (e.g. for asm clobber lists and register-variable declarations via `rgcc(R_X)`).
|
||||||
* register-variable declarations via `rgcc(R_X)`). The enum value is
|
* The enum value is bound to the `#define` so the two forms cannot drift apart.
|
||||||
* bound to the `#define` so the two forms cannot drift apart.
|
|
||||||
*
|
*
|
||||||
* Only registers that get stringified need a `_Code` form; the rest are
|
* Only registers that get stringified need a `_Code` form; the rest are plain enum values.
|
||||||
* plain enum values. If you need to add a new one, follow the pattern:
|
* If you need to add a new one, follow the pattern:
|
||||||
* #define R_T7_Code 15
|
* #define R_T7_Code 15
|
||||||
* R_T7 = R_T7_Code, // in the enum
|
* R_T7 = R_T7_Code, // in the enum
|
||||||
*
|
*
|
||||||
* User code should always reference the enum form (`R_T4`) at arithmetic
|
* User code should always reference the enum form (`R_T4`) at arithmetic sites and let
|
||||||
* sites and let `rlit(R_T4_Code)` / `rgcc(R_T4)` handle the stringify
|
* `rlit(R_T4_Code)` / `rgcc(R_T4)` handle the stringify cases — never write the bare number `12`.
|
||||||
* cases — never write the bare number `12`.
|
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
#define R_0_Code 0
|
#define R_0_Code 0
|
||||||
#define R_AT_Code 1
|
#define R_AT_Code 1
|
||||||
@@ -159,31 +136,31 @@ enum {
|
|||||||
|
|
||||||
/* Semantic Aliases for MIPS Registers (O32 ABI) */
|
/* Semantic Aliases for MIPS Registers (O32 ABI) */
|
||||||
|
|
||||||
, rdiscard = R_0 /* Hardwired to 0 */
|
// , rdiscard = R_0 /* Hardwired to 0 */
|
||||||
, rasm_tmp = R_AT /* Assembler temporary (destroyed by some assembler pseudoinstructions!) */
|
// , rasm_tmp = R_AT /* Assembler temporary (destroyed by some assembler pseudoinstructions!) */
|
||||||
, rret_0 = R_V0 /* Function return value */
|
// , rret_0 = R_V0 /* Function return value */
|
||||||
, rret_1 = R_V1 /* Second return value (e.g., 64-bit) */
|
// , rret_1 = R_V1 /* Second return value (e.g., 64-bit) */
|
||||||
, rarg_0 = R_A0 /* First function argument */
|
// , rarg_0 = R_A0 /* First function argument */
|
||||||
, rarg_1 = R_A1 /* Second function argument */
|
// , rarg_1 = R_A1 /* Second function argument */
|
||||||
, rarg_2 = R_A2 /* Third function argument */
|
// , rarg_2 = R_A2 /* Third function argument */
|
||||||
, rarg_3 = R_A3 /* Fourth function argument */
|
// , rarg_3 = R_A3 /* Fourth function argument */
|
||||||
, rtmp_0 = R_T0 /* Temporary (Caller saved) */
|
// , rtmp_0 = R_T0 /* Temporary (Caller saved) */
|
||||||
, rtmp_1 = R_T1 /* Temporary (Caller saved) */
|
// , rtmp_1 = R_T1 /* Temporary (Caller saved) */
|
||||||
, rtmp_2 = R_T2 /* Temporary (Caller saved) */
|
// , rtmp_2 = R_T2 /* Temporary (Caller saved) */
|
||||||
, rtmp_3 = R_T3 /* Temporary (Caller saved) */
|
// , rtmp_3 = R_T3 /* Temporary (Caller saved) */
|
||||||
, rtmp_4 = R_T4 /* Temporary (Caller saved) — common GTE base pointer */
|
// , rtmp_4 = R_T4 /* Temporary (Caller saved) — common GTE base pointer */
|
||||||
, rtmp_9 = R_T9 /* Temporary (Caller saved) — common GTE base pointer */
|
// , rtmp_9 = R_T9 /* Temporary (Caller saved) — common GTE base pointer */
|
||||||
, rstatic_0 = R_S0 /* Static (Callee saved, preserved across calls) */
|
// , rstatic_0 = R_S0 /* Static (Callee saved, preserved across calls) */
|
||||||
, rstatic_1 = R_S1
|
// , rstatic_1 = R_S1
|
||||||
, rstatic_2 = R_S2
|
// , rstatic_2 = R_S2
|
||||||
, rstatic_3 = R_S3
|
// , rstatic_3 = R_S3
|
||||||
, rstatic_4 = R_S4
|
// , rstatic_4 = R_S4
|
||||||
, rstatic_5 = R_S5
|
// , rstatic_5 = R_S5
|
||||||
, rstatic_6 = R_S6
|
// , rstatic_6 = R_S6
|
||||||
, rstatic_7 = R_S7
|
// , rstatic_7 = R_S7
|
||||||
, rsaved_0 = R_S0 /* Alias for rstatic_0 (alternate vocabulary) */
|
// , rsaved_0 = R_S0 /* Alias for rstatic_0 (alternate vocabulary) */
|
||||||
, rstack_ptr = R_SP /* Stack Pointer */
|
// , rstack_ptr = R_SP /* Stack Pointer */
|
||||||
, rret_addr = R_RA /* Return Address (populated by JAL) */
|
// , rret_addr = R_RA /* Return Address (populated by JAL) */
|
||||||
|
|
||||||
/* --- MIPS CPU Opcodes (Bits 31-26) --- */
|
/* --- MIPS CPU Opcodes (Bits 31-26) --- */
|
||||||
|
|
||||||
@@ -225,7 +202,6 @@ enum {
|
|||||||
/* 2F: N/A */
|
/* 2F: N/A */
|
||||||
// , op_lwc0
|
// , op_lwc0
|
||||||
|
|
||||||
|
|
||||||
// , op_load_addr = op_la
|
// , op_load_addr = op_la
|
||||||
// , op_load_imm = op_li
|
// , op_load_imm = op_li
|
||||||
, op_jump = op_j
|
, op_jump = op_j
|
||||||
@@ -276,29 +252,29 @@ enum {
|
|||||||
enum { _BitOffsets = 0
|
enum { _BitOffsets = 0
|
||||||
/* Bit Offsets for MIPS Instruction Fields */
|
/* Bit Offsets for MIPS Instruction Fields */
|
||||||
|
|
||||||
, OPCODE_SHIFT = 26
|
, OPCODE_POS = 26
|
||||||
, RS_SHIFT = 21
|
, RS_POS = 21
|
||||||
, RT_SHIFT = 16
|
, RT_POS = 16
|
||||||
, RD_SHIFT = 11
|
, RD_POS = 11
|
||||||
, SHAMT_SHIFT = 6 /* Shift Amount */
|
, SHAMT_POS = 6 /* Shift Amount: Offset Position */
|
||||||
, FC_SHIFT = 0
|
, FC_POS = 0
|
||||||
|
|
||||||
/* Bit Masks to prevent overflow into adjacent fields */
|
/* IMM_MASK is the 16-bit two's-complement truncation for the immediate field.
|
||||||
|
* It is NOT a range guard — it is load-bearing for negative branch offsets
|
||||||
|
* (the metaprogram emits raw signed offsets; the mask truncates them to the
|
||||||
|
* 16-bit representation the hardware expects). The static analysis
|
||||||
|
* `immediate_field_width` check validates ranges at build time. */
|
||||||
|
|
||||||
, OPCODE_MASK = 0x3F
|
|
||||||
, REG_MASK = 0x1F
|
|
||||||
, SHAMT_MASK = 0x1F /* Shift Amount */
|
|
||||||
, FC_MASK = 0x3F
|
|
||||||
, IMM_MASK = 0xFFFF
|
, IMM_MASK = 0xFFFF
|
||||||
};
|
};
|
||||||
|
|
||||||
#define enc_op(op) (((op) & OPCODE_MASK) << OPCODE_SHIFT)
|
#define enc_op(op) ((op) << OPCODE_POS)
|
||||||
#define enc_rs(rs) (((rs) & REG_MASK) << RS_SHIFT)
|
#define enc_rs(rs) ((rs) << RS_POS)
|
||||||
#define enc_rt(rt) (((rt) & REG_MASK) << RT_SHIFT)
|
#define enc_rt(rt) ((rt) << RT_POS)
|
||||||
#define enc_rd(rd) (((rd) & REG_MASK) << RD_SHIFT)
|
#define enc_rd(rd) ((rd) << RD_POS)
|
||||||
#define enc_shamt(shamt) (((shamt) & SHAMT_MASK) << SHAMT_SHIFT)
|
#define enc_shamt(shamt) ((shamt) << SHAMT_POS)
|
||||||
#define enc_fc(fc) (((fc) & FC_MASK) << FC_SHIFT)
|
#define enc_fc(fc) ((fc) << FC_POS)
|
||||||
#define enc_imm(imm) (((imm) & IMM_MASK))
|
#define enc_imm(imm) ((imm) & IMM_MASK)
|
||||||
|
|
||||||
/* MIPS R-Type Instruction Format (Register-to-Register) */
|
/* MIPS R-Type Instruction Format (Register-to-Register) */
|
||||||
#define enc_r(op, rs, rt, rd, shamt, fc) (enc_op(op) | enc_rs(rs) | enc_rt(rt) | enc_rd(rd) | enc_shamt(shamt) | enc_fc(fc))
|
#define enc_r(op, rs, rt, rd, shamt, fc) (enc_op(op) | enc_rs(rs) | enc_rt(rt) | enc_rd(rd) | enc_shamt(shamt) | enc_fc(fc))
|
||||||
@@ -327,22 +303,25 @@ enum { _BitOffsets = 0
|
|||||||
* Argument order matches the MIPS assembly syntax:
|
* Argument order matches the MIPS assembly syntax:
|
||||||
* dest-first, then source operands, then immediate last.
|
* dest-first, then source operands, then immediate last.
|
||||||
*
|
*
|
||||||
* load_word(rt, base, off) → lw rt, off(base)
|
* load_word(rt, base, off) → lw rt, off(base)
|
||||||
* store_word(rt, base, off) → sw rt, off(base)
|
* store_word(rt, base, off) → sw rt, off(base)
|
||||||
* add_ui(rt, rs, imm) → addiu rt, rs, imm
|
* add_ui(rt, rs, imm) → addiu rt, rs, imm
|
||||||
* shift_lleft(rd, rt, shamt) → sll rd, rt, shamt
|
* shift_lleft(rd, rt, shamt) → sll rd, rt, shamt
|
||||||
* shift_lright(rd, rt, shamt) → srl rd, rt, shamt
|
* shift_lright(rd, rt, shamt) → srl rd, rt, shamt
|
||||||
* shift_aright(rd, rt, shamt) → sra rd, rt, shamt
|
* shift_aright(rd, rt, shamt) → sra rd, rt, shamt
|
||||||
* jump_reg(rs) → jr rs
|
* jump_reg(rs) → jr rs
|
||||||
* jump_link(rs, rd) → jalr rs (link in rd, default $ra)
|
* jump_link(rs, rd) → jalr rs (link in rd, default $ra)
|
||||||
* nop → sll $0, $0, 0
|
* nop → sll $0, $0, 0
|
||||||
*/
|
*/
|
||||||
#define load_word(rt, base, off) enc_i(op_lw, (base), (rt), (off))
|
#define load_word(rt, base, off) enc_i(op_lw, (base), (rt), (off))
|
||||||
#define load_byte(rt, base, off) enc_i(op_lb, (base), (rt), (off))
|
#define load_byte(rt, base, off) enc_i(op_lb, (base), (rt), (off))
|
||||||
#define load_half(rt, base, off) enc_i(op_lh, (base), (rt), (off))
|
#define load_half(rt, base, off) enc_i(op_lh, (base), (rt), (off))
|
||||||
#define load_byte_u(rt, base, off) enc_i(op_lbu, (base), (rt), (off))
|
#define load_byte_u(rt, base, off) enc_i(op_lbu, (base), (rt), (off))
|
||||||
#define load_half_u(rt, base, off) enc_i(op_lhu, (base), (rt), (off))
|
#define load_half_u(rt, base, off) enc_i(op_lhu, (base), (rt), (off))
|
||||||
|
#define LdSlot_
|
||||||
|
|
||||||
#define store_word(rt, base, off) enc_i(op_sw, (base), (rt), (off))
|
#define store_word(rt, base, off) enc_i(op_sw, (base), (rt), (off))
|
||||||
|
|
||||||
#define add_ui(rt, rs, imm) enc_i(op_addiu, (rs), (rt), (imm))
|
#define add_ui(rt, rs, imm) enc_i(op_addiu, (rs), (rt), (imm))
|
||||||
#define and_i(rt, rs, imm) enc_i(op_andi, (rs), (rt), (imm))
|
#define and_i(rt, rs, imm) enc_i(op_andi, (rs), (rt), (imm))
|
||||||
// #define and_si and_i
|
// #define and_si and_i
|
||||||
@@ -355,20 +334,31 @@ enum { _BitOffsets = 0
|
|||||||
#define load_u4 load_word
|
#define load_u4 load_word
|
||||||
|
|
||||||
// Ergonomic add to the same register.
|
// Ergonomic add to the same register.
|
||||||
|
#define or_i_self(rt_rs, imm) enc_i(op_ori, (rt_rs), (rt_rs), (imm))
|
||||||
#define add_ui_self(rt_rs, imm) enc_i(op_addiu, (rt_rs), (rt_rs), (imm))
|
#define add_ui_self(rt_rs, imm) enc_i(op_addiu, (rt_rs), (rt_rs), (imm))
|
||||||
|
|
||||||
/* Logic Opcodes */
|
/* Logic Opcodes */
|
||||||
|
|
||||||
#define and_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_and)
|
#define and_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_and)
|
||||||
#define or_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_or)
|
#define or_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_or)
|
||||||
#define xor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_xor)
|
#define xor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_xor)
|
||||||
#define nor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_nor)
|
#define nor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_nor)
|
||||||
|
|
||||||
|
#define or_u_self(rd_rs, rt) enc_r(op_special, (rd_rs), (rt), (rd_rs), 0, fc_or)
|
||||||
|
|
||||||
/* Shift family (R-type). shift_lleft/lright/aright: `sll/srl/sra rd, rt, shamt` */
|
/* Shift family (R-type). shift_lleft/lright/aright: `sll/srl/sra rd, rt, shamt` */
|
||||||
#define shift_lleft(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_sll)
|
#define shift_lleft(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_sll)
|
||||||
#define shift_lright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_srl)
|
#define shift_lright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_srl)
|
||||||
#define shift_aright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_sra)
|
#define shift_aright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_sra)
|
||||||
|
|
||||||
|
/* Shift Variable — register-shift forms.
|
||||||
|
* shift_lleft_var(rd, rt, rs) → sllv rd, rt, rs (shamt in low 5 bits of rs)
|
||||||
|
* shift_aright_var(rd, rt, rs) → srav rd, rt, rs */
|
||||||
|
#define shift_lleft_var(rd, rt, rs) enc_r(op_special, (rs), (rt), (rd), 0, fc_sllv)
|
||||||
|
#define shift_aright_var(rd, rt, rs) enc_r(op_special, (rs), (rt), (rd), 0, fc_srav)
|
||||||
|
|
||||||
|
#define shift_lleft_self(rd_rt, shamt) enc_r(op_special, R_0, (rd_rt), (rd_rt), (shamt), fc_sll)
|
||||||
|
|
||||||
#define mask_upper(rd, rt, shamt) shift_lleft(rd, rt, shamt), shift_lright(rd, rt, shamt)
|
#define mask_upper(rd, rt, shamt) shift_lleft(rd, rt, shamt), shift_lright(rd, rt, shamt)
|
||||||
|
|
||||||
/* jr rs — jump to address in rs. */
|
/* jr rs — jump to address in rs. */
|
||||||
@@ -381,10 +371,29 @@ enum { _BitOffsets = 0
|
|||||||
/* call_reg rs — jump-and-link to register-held address; link in $ra. */
|
/* call_reg rs — jump-and-link to register-held address; link in $ra. */
|
||||||
#define call_reg(rs) jump_link((rs), R_RA)
|
#define call_reg(rs) jump_link((rs), R_RA)
|
||||||
|
|
||||||
/* j target — absolute jump within the current 256MB region. */
|
/* j target — absolute jump within the current 256MB region.
|
||||||
|
* WARNING: `jump(off)` CANNOT BE USED for within-atom jumps in the current pipeline.
|
||||||
|
* The MIPS j opcode encodes `(target_addr >> 2)` in its 26-bit immediate field; an ABSOLUTE byte address, not a relative word offset.
|
||||||
|
* The metaprogram computes `off` as a relative word offset (`target_word_idx - branch_word_idx - 1`), which the assembler/linker does NOT resolve.
|
||||||
|
* `jump(off)` is only safe when the BUILD PIPELINE owns the absolute position of the emitted code — i.e. when: s
|
||||||
|
* - the build emits a symbol-relative `.word` expression that the linker resolvess via `R_MIPS_26`, OR
|
||||||
|
* - the code is hand-assembled with explicit absolute targets, OR a custom post-build patcher resolves the 26-bit field.
|
||||||
|
* TODO(Ed): Review this.. technically we can resolve aboslute jumps on baked atoms? (Even proedurally generated ones...)
|
||||||
|
*/
|
||||||
#define jump(off) enc_i(op_j, R_0, R_0, (off))
|
#define jump(off) enc_i(op_j, R_0, R_0, (off))
|
||||||
|
|
||||||
/* call_addr off — jump-and-link to immediate address. */
|
// Annotate an instruction as filling a branch-delay slot.
|
||||||
|
#define BdSlot_
|
||||||
|
|
||||||
|
/* jump_rel off — unconditional relative jump (the within-atom-safe `jump`).
|
||||||
|
* MIPS I R3000A has no "branch always" opcode. The idiom for an unconditional relative jump is `beq $0, $0, off`. */
|
||||||
|
#define jump_rel(off) branch_equal(R_0, R_0, (off))
|
||||||
|
|
||||||
|
/* call_addr off — jump-and-link to immediate address.
|
||||||
|
* Same WARNING as `jump(off)` above: the jal opcode also encodes an absolute 26-bit target.
|
||||||
|
* For within-atom calls, the current pipeline has no equivalent always-taken call-and-link idiom.
|
||||||
|
* Workaround: `branch_link` (always-taken branch + explicit `la $ra, next_word_addr; jr $ra`), or just use `call_reg($tmp)` after loading the target into a register.
|
||||||
|
*/
|
||||||
#define call_addr(off) enc_i(op_jal, R_0, R_0, (off))
|
#define call_addr(off) enc_i(op_jal, R_0, R_0, (off))
|
||||||
|
|
||||||
/* --- Store family (mirrors the load family) --- */
|
/* --- Store family (mirrors the load family) --- */
|
||||||
@@ -398,16 +407,7 @@ enum { _BitOffsets = 0
|
|||||||
* sub_s / sub_u → sub / subu
|
* sub_s / sub_u → sub / subu
|
||||||
* mult_s / mult_u → mult / multu (writes HI/LO; result in LO)
|
* mult_s / mult_u → mult / multu (writes HI/LO; result in LO)
|
||||||
* div_s / div_u → div / divu (LO = quot, HI = rem)
|
* div_s / div_u → div / divu (LO = quot, HI = rem)
|
||||||
*
|
*/
|
||||||
* NOTE: dsl.h defines `add_s`/`sub_s`/`mut_s`/`gt_s`/etc. as
|
|
||||||
* _Generic-based signed integer-arithmetic helpers for U1/U2/U4. Those
|
|
||||||
* live in a different conceptual layer (generic arithmetic on DSL
|
|
||||||
* types) and would collide with the instruction encoders here. The
|
|
||||||
* `#undef` below lets the gas-style names below win; if a file needs
|
|
||||||
* both, the dsl.h versions can be reached via their long forms
|
|
||||||
* (e.g. `def_signed_op`-style or the underlying `add_s1/s2/s4`). */
|
|
||||||
#undef add_s
|
|
||||||
#undef sub_s
|
|
||||||
#define add_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_add)
|
#define add_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_add)
|
||||||
#define add_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_addu)
|
#define add_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_addu)
|
||||||
#define sub_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_sub)
|
#define sub_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_sub)
|
||||||
@@ -417,6 +417,7 @@ enum { _BitOffsets = 0
|
|||||||
#define div_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_div)
|
#define div_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_div)
|
||||||
#define div_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_divu)
|
#define div_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_divu)
|
||||||
|
|
||||||
|
// TODO(Ed): Change convention of 'self' to ds for (destination is source)?
|
||||||
#define add_u_self(rd_rs, rt) add_u(rd_rs, rd_rs, rt)
|
#define add_u_self(rd_rs, rt) add_u(rd_rs, rd_rs, rt)
|
||||||
|
|
||||||
/* --- Arithmetic I-type (immediate) --- */
|
/* --- Arithmetic I-type (immediate) --- */
|
||||||
@@ -436,7 +437,7 @@ enum { _BitOffsets = 0
|
|||||||
#define mov_to_low(rs) enc_r(op_special, (rs), R_0, R_0, 0, fc_mtlo)
|
#define mov_to_low(rs) enc_r(op_special, (rs), R_0, R_0, 0, fc_mtlo)
|
||||||
|
|
||||||
/* --- Atomic branches (no pseudos like bgt/bge; compose with slt_* + branch_ne) ---
|
/* --- Atomic branches (no pseudos like bgt/bge; compose with slt_* + branch_ne) ---
|
||||||
* branch_equal rs, rt, off → beq rs, rt, off
|
* branch_equal rs, rt, off → beq rs, rt, off
|
||||||
* branch_ne rs, rt, off → bne rs, rt, off
|
* branch_ne rs, rt, off → bne rs, rt, off
|
||||||
* branch_lt_zero rs, off → bltz rs, off
|
* branch_lt_zero rs, off → bltz rs, off
|
||||||
* branch_gt_zero rs, off → bgtz rs, off
|
* branch_gt_zero rs, off → bgtz rs, off
|
||||||
@@ -458,30 +459,28 @@ enum { _BitOffsets = 0
|
|||||||
/* --- Shift-amount alias (matches the gas convention `\p3 = shamt`) --- */
|
/* --- Shift-amount alias (matches the gas convention `\p3 = shamt`) --- */
|
||||||
#define shift_amount(rd, rt, n) shift_lleft(rd, rt, n)
|
#define shift_amount(rd, rt, n) shift_lleft(rd, rt, n)
|
||||||
|
|
||||||
/* nop — canonical sll $0, $0, 0 */
|
/* nop — sll $0, $0, 0 */
|
||||||
#define nop shift_lleft(rdiscard, rdiscard, 0)
|
#define nop shift_lleft(R_0, R_0, 0)
|
||||||
|
#define nop2 nop, nop
|
||||||
|
|
||||||
|
// li_s — load signed 16-bit immediate into GPR (addiu rt, $0, imm — sign-extends).
|
||||||
|
#define li_s(rt, imm) add_ui((rt), R_0, (imm))
|
||||||
|
// #define load_imm_s(rt, imm) add_ui((rt), R_0, (imm))
|
||||||
|
|
||||||
#define load_imm_1w(rt, imm) add_ui((rt), R_0, (imm))
|
#define load_imm_1w(rt, imm) add_ui((rt), R_0, (imm))
|
||||||
#define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm))
|
#define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm))
|
||||||
|
|
||||||
/* load_imm_2w — unconditional 2-word `li` form: `lui` + (ori | addi).
|
/* load_imm_2w — unconditional 2-word `li` form: `lui` + (ori | addi).
|
||||||
*
|
* Granular companion to `load_imm`: skips the compile-time range checks and always emits 2 .words. Use this when:
|
||||||
* Granular companion to `load_imm`: skips the compile-time range checks
|
|
||||||
* and always emits 2 .words. Use this when:
|
|
||||||
* - you know `imm` is > 0xFFFF (otherwise you're wasting a word), OR
|
* - you know `imm` is > 0xFFFF (otherwise you're wasting a word), OR
|
||||||
* - `imm` is not a compile-time constant and you want predictable
|
* - `imm` is not a compile-time constant and you want predictable 2-word emission without the `__builtin_constant_p` branches.
|
||||||
* 2-word emission without the `__builtin_constant_p` branches.
|
|
||||||
*
|
*
|
||||||
* The lo16 strategy is still chosen at expansion time on the lo half:
|
* The lo16 strategy is still chosen at expansion time on the lo half:
|
||||||
* lo16 in 0x0000..0x7FFF → addi (sign-ext is harmless, the lui
|
* lo16 in 0x0000..0x7FFF → addi (sign-ext is harmless, the lui already cleared bits 15..0)
|
||||||
* already cleared bits 15..0)
|
* lo16 in 0x8000..0xFFFF → ori (zero-extends to preserve the intended bit pattern)
|
||||||
* lo16 in 0x8000..0xFFFF → ori (zero-extends to preserve the
|
|
||||||
* intended bit pattern)
|
|
||||||
*
|
*
|
||||||
* For situations where you need to bypass even this choice (e.g. to
|
* For situations where you need to bypass even this choice (e.g. to force a specific encoding for a known discontiguous high/low pair),
|
||||||
* force a specific encoding for a known discontiguous high/low pair),
|
|
||||||
* see `load_imm_2w_ori_forced` and `load_imm_2w_addi_forced` below.
|
* see `load_imm_2w_ori_forced` and `load_imm_2w_addi_forced` below.
|
||||||
*
|
|
||||||
* Statement-level (not expression-level): emits its own `asm volatile(...)`.
|
* Statement-level (not expression-level): emits its own `asm volatile(...)`.
|
||||||
*/
|
*/
|
||||||
#define load_imm_2w(rt, imm) do { \
|
#define load_imm_2w(rt, imm) do { \
|
||||||
@@ -512,9 +511,8 @@ enum { _BitOffsets = 0
|
|||||||
} while (0)
|
} while (0)
|
||||||
|
|
||||||
/* load_imm_2w_addi_forced — force the `lui` + `addi` form regardless of lo16 sign.
|
/* load_imm_2w_addi_forced — force the `lui` + `addi` form regardless of lo16 sign.
|
||||||
* Use when you know sign-extension is fine (e.g. lo16 is treated as
|
* Use when you know sign-extension is fine (e.g. lo16 is treated as signed downstream)
|
||||||
* signed downstream) and you want a smaller effective instruction
|
* and you want a smaller effective instruction (the assembler/MIPS hardware will sign-extend the imm16). */
|
||||||
* (the assembler/MIPS hardware will sign-extend the imm16). */
|
|
||||||
#define load_imm_2w_addi_forced(rt, imm) do { \
|
#define load_imm_2w_addi_forced(rt, imm) do { \
|
||||||
/*U4 _li2a_imm_ = (U4)(imm);*/ \
|
/*U4 _li2a_imm_ = (U4)(imm);*/ \
|
||||||
asm volatile(asm_words( \
|
asm volatile(asm_words( \
|
||||||
@@ -526,23 +524,17 @@ enum { _BitOffsets = 0
|
|||||||
|
|
||||||
/* load_imm rt, imm — true `li` semantics (assembler `li` pseudo)
|
/* load_imm rt, imm — true `li` semantics (assembler `li` pseudo)
|
||||||
*
|
*
|
||||||
* Dispatches at compile time on the immediate's range, picking the
|
* Dispatches at compile time on the immediate's range, picking the smallest single-instruction form when possible:
|
||||||
* smallest single-instruction form when possible:
|
* imm in 0 .. 0x7FFF → addi rt, $0, imm (1 word)
|
||||||
*
|
* imm in 0x8000 .. 0xFFFF → ori rt, $0, imm (1 word; sign-bit must be zeroed)
|
||||||
* imm in 0 .. 0x7FFF → addi rt, $0, imm (1 word)
|
* imm in 0x10000 .. 0xFFFFFFFF → lui + (ori | addi) (2 words)
|
||||||
* imm in 0x8000 .. 0xFFFF → ori rt, $0, imm (1 word; sign-bit must be zeroed)
|
|
||||||
* imm in 0x10000 .. 0xFFFFFFFF → lui + (ori | addi) (2 words)
|
|
||||||
*
|
|
||||||
* Statement-level (not expression-level): the macro emits its own
|
|
||||||
* `asm volatile(...)` block with 1 or 2 .word constants. Callers can
|
|
||||||
* group multiple `load_imm` calls in a single volatile by using the
|
|
||||||
* lower-level encoders directly:
|
|
||||||
*
|
*
|
||||||
|
* Statement-level (not expression-level): the macro emits its own `asm volatile(...)` block with 1 or 2 .word constants.
|
||||||
|
* Callers can group multiple `load_imm` calls in a single volatile by using the lower-level encoders directly:
|
||||||
* load_imm(R_T4, 0x12345678); // emits 2 .words
|
* load_imm(R_T4, 0x12345678); // emits 2 .words
|
||||||
*
|
*
|
||||||
* Falls back to a 2-word form if `imm` is not a compile-time constant,
|
* Falls back to a 2-word form if `imm` is not a compile-time constant, but that path is unusual
|
||||||
* but that path is unusual (load_imm is most useful with literal
|
* (load_imm is most useful with literal addresses and magic numbers). */
|
||||||
* addresses and magic numbers). */
|
|
||||||
#define load_imm(rt, imm) do { \
|
#define load_imm(rt, imm) do { \
|
||||||
if (cexpr_(imm) && ((imm) <= 0x7FFFU)) { \
|
if (cexpr_(imm) && ((imm) <= 0x7FFFU)) { \
|
||||||
/* Small positive: addi rt, $0, imm */ \
|
/* Small positive: addi rt, $0, imm */ \
|
||||||
@@ -582,9 +574,8 @@ enum { _BitOffsets = 0
|
|||||||
|
|
||||||
|
|
||||||
/* Standard clobber list for pure-MIPS asm volatile blocks: caller-saved
|
/* Standard clobber list for pure-MIPS asm volatile blocks: caller-saved
|
||||||
* GPRs that the kernel treats as volatile (v0/v1/t0/t1/ra) plus the
|
* GPRs that the kernel treats as volatile (v0/v1/t0/t1/ra) plus the "memory" barrier.
|
||||||
* "memory" barrier. The register ids are passed through `rlit` so
|
* The register ids are passed through `rlit` so the R_*_Code `#define`s are stringified into "$N" at expansion time. */
|
||||||
* the R_*_Code `#define`s are stringified into "$N" at expansion time. */
|
|
||||||
#define clbr_volatile_gprs rlit(R_V0), rlit(R_T0), rlit(R_T1), rlit(R_RA), clb_mem_drain
|
#define clbr_volatile_gprs rlit(R_V0), rlit(R_T0), rlit(R_T1), rlit(R_RA), clb_mem_drain
|
||||||
|
|
||||||
#define asm_mips_flush_icache() asm volatile( asm_words( \
|
#define asm_mips_flush_icache() asm volatile( asm_words( \
|
||||||
@@ -595,6 +586,7 @@ enum { _BitOffsets = 0
|
|||||||
, jump_link(rtmp_0, rret_addr) \
|
, jump_link(rtmp_0, rret_addr) \
|
||||||
, nop \
|
, nop \
|
||||||
, load_word(rret_addr, rstack_ptr, 4) \
|
, load_word(rret_addr, rstack_ptr, 4) \
|
||||||
, jump_reg(rret_addr) \
|
|
||||||
, add_ui(rstack_ptr, rstack_ptr, MipsStackAlignment) \
|
, add_ui(rstack_ptr, rstack_ptr, MipsStackAlignment) \
|
||||||
|
, jump_reg(rret_addr) \
|
||||||
|
, nop \
|
||||||
) asm_clobber: clbr_volatile_gprs )
|
) asm_clobber: clbr_volatile_gprs )
|
||||||
|
|||||||
@@ -2,9 +2,8 @@
|
|||||||
* duffle DSL — MIPS Vendor Mnemonics (opt-in)
|
* duffle DSL — MIPS Vendor Mnemonics (opt-in)
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
*
|
*
|
||||||
* Provides the textbook MIPS assembly mnemonics as thin aliases to the
|
* Provides the textbook MIPS assembly mnemonics as thin aliases to the duffle macros in mips.h.
|
||||||
* canonical duffle macros in mips.h. The duffle names are primary; this
|
* The duffle names are primary; this header is for users who prefer the textbook mnemonics.
|
||||||
* header is for users who prefer the textbook mnemonics.
|
|
||||||
*
|
*
|
||||||
* USAGE: #include "duffle/mips_vendor_sym.h" // after mips.h
|
* USAGE: #include "duffle/mips_vendor_sym.h" // after mips.h
|
||||||
*
|
*
|
||||||
@@ -21,12 +20,6 @@
|
|||||||
* jal -> call_addr (jump-and-link to immediate address)
|
* jal -> call_addr (jump-and-link to immediate address)
|
||||||
* jalr -> call_reg (jump-and-link to register, default $ra)
|
* jalr -> call_reg (jump-and-link to register, default $ra)
|
||||||
* (for the 2-arg `jalr rs, rd`, use `jump_link(rs, rd)` directly)
|
* (for the 2-arg `jalr rs, rd`, use `jump_link(rs, rd)` directly)
|
||||||
*
|
|
||||||
* The vendor mnemonics are NOT registered with the duffle word-count
|
|
||||||
* metadata (tape_atom.metadata.h). They expand to the duffle canonical
|
|
||||||
* macros which DO have word-count entries. Verification: V2 (objdump
|
|
||||||
* byte-identical) holds.
|
|
||||||
*
|
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
|
|
||||||
#ifdef INTELLISENSE_DIRECTIVES
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
|||||||
@@ -0,0 +1,196 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
# include "mips.h"
|
||||||
|
# include "dsl.atom.h"
|
||||||
|
# include "lottes_tape.h"
|
||||||
|
# include "pad.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
ATOM_FILE_DEBUGGER_LINE_MARKER(pad_atom_c);
|
||||||
|
|
||||||
|
#pragma region MACs (Mips Atom Components)
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_pad_set_centered_axes(AtomBuilder_R ab, Reg state, Reg scratch) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
load_upper_i(scratch, (PadAxis_Centered >> 16) & 0xFFFF),
|
||||||
|
or_i_self( scratch, PadAxis_Centered & 0xFFFF), // mac_load_word_imm(scratch, PadAxis_Centered),
|
||||||
|
store_word( scratch, state, O_(PadState,axes)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_pad_set_id_byte(AtomBuilder_R ab, Reg state, Reg r_id, U1 id_value) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
add_ui( r_id, R_0, id_value),
|
||||||
|
store_byte(r_id, state, O_(PadState,id)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_pad_set_status(AtomBuilder_R ab, U4 r_tmp, U1 r_state, U4 pad_status) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
add_ui( r_tmp, R_0, pad_status),
|
||||||
|
store_word(r_tmp, r_state, O_(PadState,status)),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* Invert r_buttons (active-low → active-high) and store to PadState.buttons.
|
||||||
|
* r_buttons must already be loaded (the caller is responsible for filling the load-delay slot of
|
||||||
|
* the preceding load_half_u with an instruction that doesn't read r_buttons). */
|
||||||
|
FI_ Slice_MipsCode ac_pad_store_inverted_buttons(AtomBuilder_R ab, U1 r_buttons, U1 r_pad_state) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
nor_u( r_buttons, r_buttons, R_0),
|
||||||
|
store_half(r_buttons, r_pad_state, O_(PadState,buttons)),
|
||||||
|
})
|
||||||
|
|
||||||
|
#pragma endregion MACs (Mips Atom Components)
|
||||||
|
|
||||||
|
#pragma region Baked Atoms
|
||||||
|
|
||||||
|
/* ----- pad_bios_snapshot -----
|
||||||
|
* Per-frame snapshot of one BIOS pad buffer into PadState.
|
||||||
|
* Decoder (branch ladder on raw[0] status + raw[1] id):
|
||||||
|
* 1. raw[0] == 0xFF -> Disconnected (buttons=0, axes=0x80)
|
||||||
|
* 2. raw[0]==0 && raw[1]==0 -> Pending (buttons=0, axes=0x80)
|
||||||
|
* 3. raw[1] == 0x41 -> Digital (buttons normalized; axes=0x80)
|
||||||
|
* 4. raw[1] == 0x53 -> AnalogStick (buttons normalized; axes from raw[4..7])
|
||||||
|
* 5. raw[1] in 0x7x -> AnalogPad (buttons normalized; axes from raw[4..7])
|
||||||
|
* 6. else -> Unsupported (buttons=0, axes=0x80)
|
||||||
|
*
|
||||||
|
* Buttons normalization: byte_swap16((~raw_buttons) & 0xFFFF).
|
||||||
|
* raw_buttons = load_half_u(raw, 2) = raw[2] | (raw[3] << 8).
|
||||||
|
* byte_swap16(x) = (x >> 8) | (x << 8); nor(x, R_0) = ~x. store_half truncates to 16 bits so the upper-16 mask is implicit in the store.
|
||||||
|
*
|
||||||
|
* Register use (atom-local; no wave-context touched):
|
||||||
|
* R_T0 = raw base : Kept throughout; axes loads read raw[4..7] from R_T0.
|
||||||
|
* R_T1 = state base : Kept throughout; all stores go through R_T1.
|
||||||
|
* R_T2 = raw[0] status : Alive across the disc/pending/id dispatch, then dead.
|
||||||
|
* R_T3 = raw[1] id : Alive across the id dispatch, then dead.
|
||||||
|
* R_T4 = scratch : Shifts, compares, immediate loads, store values.
|
||||||
|
* R_T5 = scratch : Parallel lui + ori for the 0x80808080 axes constant + byte-swap target.
|
||||||
|
*/
|
||||||
|
enum {
|
||||||
|
R_PadRaw = R_T0 atom_reg atom_type(U1),
|
||||||
|
R_PadState = R_T1 atom_reg atom_type(PadState*),
|
||||||
|
R_RawStatus = R_T2 atom_reg,
|
||||||
|
R_RawId = R_T3 atom_reg,
|
||||||
|
};
|
||||||
|
typedef Struct_(Binds_PadBiosSnapshot) {
|
||||||
|
PadBiosRaw* raw;
|
||||||
|
PadState* state;
|
||||||
|
};
|
||||||
|
internal MipsAtom_(pad_bios_snapshot) atom_info(atom_bind(Binds_PadBiosSnapshot)
|
||||||
|
, atom_reads( R_PadRaw, R_PadState, R_RawStatus, R_RawId)
|
||||||
|
, atom_writes(R_PadRaw, R_PadState, R_RawStatus, R_RawId)
|
||||||
|
) {
|
||||||
|
/* === Bind consumption: T0 = raw, T1 = state, advance R_TapePtr by 8. */
|
||||||
|
load_word(R_PadRaw, R_TapePtr, O_(Binds_PadBiosSnapshot,raw)),
|
||||||
|
load_word(R_PadState, R_TapePtr, O_(Binds_PadBiosSnapshot,state)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_PadBiosSnapshot)),
|
||||||
|
|
||||||
|
/* === Read raw[0] (status) + raw[1] (id) */
|
||||||
|
load_byte_u(R_RawStatus, R_PadRaw, O_(PadBiosRaw,status)),
|
||||||
|
load_byte_u(R_RawId, R_PadRaw, O_(PadBiosRaw,id)),
|
||||||
|
|
||||||
|
atom_label(snap_root) /* === Case 1: Disconnected (status == 0xFF). */
|
||||||
|
add_ui(R_T4, R_0, PadRawStatus_Timeout), branch_ne(R_RawStatus, R_T4, atom_offset(snap_root, skip_disconnected)),
|
||||||
|
/* BD-slot: pre-compute PadStatus_Disconnected. Branch reads R_T4=0xFF in EX before this WB completes.
|
||||||
|
* If branch NOT taken (fall through to pending/id_dispatch), R_T4 is overwritten by the next case body's add_ui — harmless. */
|
||||||
|
|
||||||
|
atom_label(disconnected) /* === Disconnected body. */
|
||||||
|
mac_pad_set_status(R_T4, R_PadState, PadStatus_Disconnected),
|
||||||
|
store_half( R_0, R_PadState, O_(PadState,buttons)),
|
||||||
|
mac_pad_set_centered_axes(R_PadState, R_T4),
|
||||||
|
mac_pad_set_id_byte(R_PadState, R_RawId, PadRawStatus_Timeout),
|
||||||
|
jump_rel(atom_offset(disconnected, snap_end)),
|
||||||
|
/* BD-slot: load next atom's entry point (replaces the nop).
|
||||||
|
* Always jumps to snap_end, where mac_yield_tail() transfers control to R_AtomJmp without re-loading it. */
|
||||||
|
mac_yield_load(),
|
||||||
|
atom_label(skip_disconnected)
|
||||||
|
|
||||||
|
/* === Case 2: Pending (status == 0 && id == 0)
|
||||||
|
* Combined check: if (status | id) != 0 then skip to id_dispatch. Falls through to the Pending case only when both are zero. */
|
||||||
|
or_u_self(R_RawStatus, R_RawId), branch_ne(R_RawStatus, R_0, atom_offset(case_2, id_dispatch)),
|
||||||
|
/* BD-slot: pre-compute PadStatus_Pending. Branch reads R_RawStatus in EX before this WB completes.
|
||||||
|
* If branch NOT taken (fall through to id_dispatch), R_T4 is overwritten by the digital/analog body add_ui - harmless. */
|
||||||
|
|
||||||
|
atom_label(pending) /* === Pending body (status=0, id=0 — pre-IRQ-empty buffer). */
|
||||||
|
mac_pad_set_status(R_T4, R_PadState, PadStatus_Pending),
|
||||||
|
store_half( R_0, R_PadState, O_(PadState,buttons)),
|
||||||
|
mac_pad_set_centered_axes(R_PadState, R_T4),
|
||||||
|
store_byte(R_RawId, R_PadState, O_(PadState,id)),
|
||||||
|
jump_rel(atom_offset(pending, snap_end)),
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(id_dispatch) /* === Case 3-6: ID dispatch */
|
||||||
|
add_ui(R_T4, R_0, PadRawId_Digital), branch_ne(R_RawId, R_T4, atom_offset(id_dispatch, try_analog_stick)),
|
||||||
|
/* BD-slot: pre-compute PadStatus_Digital. Branch reads R_RawId in EX before this WB completes.
|
||||||
|
* If branch NOT taken (fall through to try_analog_stick), R_T4 is overwritten by the analog body add_ui. */
|
||||||
|
|
||||||
|
/* === Digital body (status, buttons normalize, axes=0x80, id, branch.
|
||||||
|
* R_T5 holds the 0x80808080 axes constant (loaded into the load-delay slot of the buttons-load).
|
||||||
|
* R_T5 is then "dead" — only consumed at the analog_pad range check downstream. */
|
||||||
|
mac_pad_set_status(R_T4, R_PadState, PadStatus_Digital),
|
||||||
|
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw, buttons)), /* R_T4 = raw_buttons; */
|
||||||
|
mac_load_word_imm( R_T5, PadAxis_Centered), /* fills the buttons-load's delay slot (doesn't read R_T4) */
|
||||||
|
// load_upper_i(R_T5, PadAxis_Centered_Hi), or_i_self(R_T5, PadAxis_Centered_Lo),
|
||||||
|
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
|
||||||
|
store_word(R_T5, R_PadState, O_(PadState, axes)), /* single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y) */
|
||||||
|
mac_pad_set_id_byte(R_PadState, R_T4, PadRawId_Digital),
|
||||||
|
|
||||||
|
jump_rel(atom_offset(id_dispatch, snap_end)),
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(try_analog_stick) /* === Case 4: AnalogStick (id == 0x53)*/
|
||||||
|
add_ui(R_T4, R_0, PadRawId_AnalogStick), branch_ne(R_RawId, R_T4, atom_offset(try_analog_stick, try_analog_pad)),
|
||||||
|
/* BD-slot: pre-compute PadStatus_AnalogStick. Branch reads R_RawId in EX before this WB completes.
|
||||||
|
* If branch NOT taken (fall through to try_analog_pad), R_T4 is overwritten by the analog_pad body add_ui. */
|
||||||
|
|
||||||
|
atom_label(analog_stick) /* === AnalogStick body
|
||||||
|
* R_T5 holds left_xy (loaded into the load-delay slot of the buttons-load via the left-axis load_half_u).
|
||||||
|
* R_T4 holds right_xy (loaded into the load-delay slot of the left-load).
|
||||||
|
* R_T5 is then "dead" — reused for the id-byte value load in mac_pad_write_id_byte.
|
||||||
|
* The buttons invert+store happens BEFORE R_T4 is overwritten by the right_xy load. */
|
||||||
|
mac_pad_set_status(R_T4, R_PadState, PadStatus_AnalogStick),
|
||||||
|
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,buttons)), /* R_T4 = raw_buttons; delay slot at the next instruction */
|
||||||
|
load_half_u( R_T5, R_PadRaw, O_(PadBiosRaw,left)), /* fills the buttons-load's delay slot (doesn't read R_T4) */
|
||||||
|
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
|
||||||
|
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,right)), /* fills R_T5's load-delay slot (doesn't read R_T5); overwrites R_T4 (was buttons) with right_xy */
|
||||||
|
store_half( R_T5, R_PadState, O_(PadState, left)),
|
||||||
|
store_half( R_T4, R_PadState, O_(PadState, right)),
|
||||||
|
mac_pad_set_id_byte(R_PadState, R_T5, PadRawId_AnalogStick),
|
||||||
|
jump_rel(atom_offset(analog_stick, snap_end)),
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(try_analog_pad) /* === Case 5-6: AnalogPad (id & 0xF0 == 0x70) */
|
||||||
|
and_i( R_T4, R_RawId, PadRawId_AnalogPadMask),
|
||||||
|
add_ui( R_T5, R_0, PadRawId_AnalogPadValue),
|
||||||
|
branch_ne(R_T4, R_T5, atom_offset(try_analog_pad, try_unsupported)),
|
||||||
|
/* BD-slot: pre-compute PadStatus_AnalogPad. Branch reads R_T4 in EX before this WB completes.
|
||||||
|
* If branch NOT taken (fall through to try_unsupported), R_T4 is overwritten by the unsupported body add_ui. */
|
||||||
|
|
||||||
|
atom_label(analog_pad) /* === AnalogPad body
|
||||||
|
* Same shape as AnalogStick with AnalogPad status. R_T5 holds left_xy (it's dead on this path).
|
||||||
|
* The id byte is raw id from the BIOS buffer (R_RawId already holds raw[1]).
|
||||||
|
* Buttons invert + store happens before R_T4 is overwritten by the right_xy load. */
|
||||||
|
mac_pad_set_status(R_T4, R_PadState, PadStatus_AnalogPad),
|
||||||
|
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,buttons)), /* R_T4 = raw_buttons; delay slot at the next instruction */
|
||||||
|
load_half_u( R_T5, R_PadRaw, O_(PadBiosRaw,left)), /* fills the buttons-load's delay slot (doesn't read R_T4) */
|
||||||
|
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
|
||||||
|
load_half_u(R_T4, R_PadRaw, O_(PadBiosRaw,right)), /* fills R_T5's load-delay slot (doesn't read R_T5); overwrites R_T4 with right_xy */
|
||||||
|
store_half( R_T5, R_PadState, O_(PadState, left)),
|
||||||
|
store_half( R_T4, R_PadState, O_(PadState, right)),
|
||||||
|
store_byte( R_RawId, R_PadState, O_(PadState, id)),
|
||||||
|
|
||||||
|
jump_rel(atom_offset(analog_pad, snap_end)),
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(try_unsupported) /* === Case 7: Unsupported — fall through from the AnalogPad range-check miss. */
|
||||||
|
add_ui( R_T4, R_0, PadStatus_Unsupported),
|
||||||
|
store_word(R_T4, R_PadState, O_(PadState,status)),
|
||||||
|
store_half(R_0, R_PadState, O_(PadState,buttons)),
|
||||||
|
mac_pad_set_centered_axes(R_PadState, R_T4),
|
||||||
|
mac_pad_set_id_byte(R_PadState, R_RawId, PadUnknownId_Sentinel),
|
||||||
|
/* Fall through to snap_end. */
|
||||||
|
|
||||||
|
atom_label(no_jump_fallthrough)
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(snap_end)
|
||||||
|
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the case-exit branch. */
|
||||||
|
mac_yield_tail(),
|
||||||
|
};
|
||||||
|
|
||||||
|
#pragma endregion Baked Atoms
|
||||||
@@ -0,0 +1,78 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# include "dsl.h"
|
||||||
|
# include "gcc_asm.h"
|
||||||
|
# include "mips.h"
|
||||||
|
# include "bios.h"
|
||||||
|
# include "pad.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
/* Uses ONE 8-byte frame allocated via the compiler's standard prologue.
|
||||||
|
* 4 wasted-arg words for B(12h) InitPAD2 are at [SP+0..15] but are not explicitly allocated.
|
||||||
|
* Compiler handles the MIPS O32 "wasted stack" convention for us by treating the B-call as a 4-arg call.
|
||||||
|
*
|
||||||
|
* The buffer pointers are passed as arguments so the compiler keeps them in callee-saved registers;
|
||||||
|
* The B(12h) asm volatile block does NOT clobber those registers (it clobbers only the volatile GPRs + B-table arg registers explicitly).
|
||||||
|
* The C-level writes after the call re-load the pointers from their callee-saved homes.
|
||||||
|
*
|
||||||
|
* The clobber list for both B-calls names the full BIOS destroy set documented in kernelbios.md:167-174 (R1..R15, R24..R25, R31, HI/LO).
|
||||||
|
* The kernel-ABI "volatile GPRs" subset is clb_mem_drain; the rest of the destroy set is enumerated explicitly here. */
|
||||||
|
NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
|
||||||
|
{
|
||||||
|
/* Pin raw0 + raw1 to $a0 + $a1 via rgcc; the B(12h) call uses these directly.
|
||||||
|
* The `(void)` casts mark them as unread after the call so the compiler doesn't need to move them back. */
|
||||||
|
register PadBiosRaw* p0 rgcc(R_A0) = raw0;
|
||||||
|
register PadBiosRaw* p1 rgcc(R_A1) = raw1;
|
||||||
|
(void)p0; (void)p1;
|
||||||
|
|
||||||
|
// TODO(Ed): Properly annotate the raw values in the inline asm instructions.
|
||||||
|
// Use enums.
|
||||||
|
|
||||||
|
/* B(12h) InitPAD2(raw0, 0x22, raw1, 0x22)
|
||||||
|
* $a0 = raw0 (rgcc-bound; survives the sequence below)
|
||||||
|
* $a1 = raw1 (preserved into $a2 before $a1 is overwritten)
|
||||||
|
* $a2 = raw1 (moved from $a1; survives $a1's overwrite)
|
||||||
|
* $a3 = 0x22 (immediate)
|
||||||
|
* $t1 = 0x12 (function number)
|
||||||
|
* $t2 = 0xB0 (BIOS B-table address) */
|
||||||
|
asm volatile(
|
||||||
|
asm_words(
|
||||||
|
or_u( R_A2, R_A1, R_0), /* $a2 = $a1 = raw1 */
|
||||||
|
add_ui( R_A1, R_0, bios_pad_buffer_size), /* $a1 = 0x22 */
|
||||||
|
add_ui( R_A3, R_0, bios_pad_buffer_size), /* $a3 = 0x22 */
|
||||||
|
add_ui( R_T1, R_0, bios_init_pad_2), /* $t1 = 0x12 */
|
||||||
|
add_ui( R_T2, R_0, bios_btable_addr), /* $t2 = 0xB0 */
|
||||||
|
call_reg(R_T2), /* jalr $t2, $ra */
|
||||||
|
nop /* BD slot */
|
||||||
|
)
|
||||||
|
asm_rpins, r_use(p0), r_use(p1)
|
||||||
|
asm_clobber:
|
||||||
|
rlit(R_AT),
|
||||||
|
rlit(R_V0), rlit(R_V1),
|
||||||
|
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||||
|
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
||||||
|
rlit(R_RA),
|
||||||
|
clb_mem_drain
|
||||||
|
);
|
||||||
|
|
||||||
|
/* The C-level writes re-load the pointers via the parameter names and write 0xFF to each
|
||||||
|
* buffer's status byte to mark the initial-state hazard documented in kernelbios.md:1621-1624. */
|
||||||
|
u1_v(raw0)[0] = 0xFF;
|
||||||
|
u1_v(raw1)[0] = 0xFF;
|
||||||
|
|
||||||
|
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */
|
||||||
|
asm volatile(
|
||||||
|
asm_words(
|
||||||
|
add_ui( R_T1, R_0, bios_start_pad_2), /* $t1 = 0x13 */
|
||||||
|
add_ui( R_T2, R_0, bios_btable_addr), /* $t2 = 0xB0 (re-load) */
|
||||||
|
call_reg(R_T2), /* jalr $t2, $ra */
|
||||||
|
nop /* BD slot */
|
||||||
|
)
|
||||||
|
asm_clobber:
|
||||||
|
rlit(R_AT),
|
||||||
|
rlit(R_V0), rlit(R_V1),
|
||||||
|
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||||
|
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
||||||
|
rlit(R_RA),
|
||||||
|
clb_mem_drain
|
||||||
|
);
|
||||||
|
}
|
||||||
@@ -0,0 +1,108 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# pragma once
|
||||||
|
# include "dsl.h"
|
||||||
|
# include "math.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// PSX button bit positions: PSX-SPX docs/psx-spx/docs/controllersandmemorycards.md:405-421.
|
||||||
|
typedef Enum_(U2, PadBtns) {
|
||||||
|
Bit_(Pad_Select, 0),
|
||||||
|
Bit_(Pad_L3, 1),
|
||||||
|
Bit_(Pad_R3, 2),
|
||||||
|
Bit_(Pad_Start, 3),
|
||||||
|
Bit_(Pad_Up, 4),
|
||||||
|
Bit_(Pad_Right, 5),
|
||||||
|
Bit_(Pad_Down, 6),
|
||||||
|
Bit_(Pad_Left, 7),
|
||||||
|
Bit_(Pad_L2, 8),
|
||||||
|
Bit_(Pad_R2, 9),
|
||||||
|
Bit_(Pad_L1, 10),
|
||||||
|
Bit_(Pad_R1, 11),
|
||||||
|
Bit_(Pad_Triangle, 12),
|
||||||
|
Bit_(Pad_Circle, 13),
|
||||||
|
Bit_(Pad_Cross, 14),
|
||||||
|
Bit_(Pad_Square, 15),
|
||||||
|
};
|
||||||
|
|
||||||
|
enum {
|
||||||
|
PadId_Offset = 4,
|
||||||
|
|
||||||
|
Pad0 = 0 << PadId_Offset,
|
||||||
|
Pad1 = 1 << PadId_Offset,
|
||||||
|
};
|
||||||
|
|
||||||
|
/* =============================================================================
|
||||||
|
* BIOS pad-buffer subsystem: docs/psx-spx/docs/kernelbios.md (B(12h) + B(13h))
|
||||||
|
* ============================================================================= */
|
||||||
|
|
||||||
|
enum {
|
||||||
|
PAD_BIOS_RAW_SIZE = 0x22,
|
||||||
|
};
|
||||||
|
// BIOS pad buffer layout (docs/psx-spx/docs/kernelbios.md (InitPAD2 returns 0x22 = 34 bytes per port)).
|
||||||
|
// Bytes 0..7 are the named snapshot region; bytes 8..33 are reserved (the BIOS writes the buffer raw; we only read bytes 0..7 via O_(PadBiosRaw, ...)).
|
||||||
|
typedef Struct_(PadBiosRaw) {
|
||||||
|
U1 status; /* offset 0 (PadRawStatus_Ok / PadRawStatus_Timeout) */
|
||||||
|
U1 id; /* offset 1 (PadRawId_Digital / PadRawId_AnalogStick / 0x7x AnalogPad) */
|
||||||
|
U2 buttons; /* offset 2-3 (active-low 16-bit button map) */
|
||||||
|
V2_U1 right; /* offset 4-5 (right stick x, y) */
|
||||||
|
V2_U1 left; /* offset 6-7 (left stick x, y) */
|
||||||
|
U1 reserved[PAD_BIOS_RAW_SIZE - 8]; /* offset 8..33 */
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef Enum_(U4, PadStatus) {
|
||||||
|
PadStatus_Disconnected,
|
||||||
|
PadStatus_Digital,
|
||||||
|
PadStatus_AnalogStick,
|
||||||
|
PadStatus_AnalogPad,
|
||||||
|
PadStatus_Unsupported,
|
||||||
|
PadStatus_Pending,
|
||||||
|
PadStatus_Invalid,
|
||||||
|
};
|
||||||
|
|
||||||
|
// Distinct from the game-facing PadStatus enum: PadRawStatus_Ok and PadRawStatus_Timeout are raw BIOS values
|
||||||
|
typedef Enum_(U1, PadRawStatus) {
|
||||||
|
PadRawStatus_Ok = 0x00,
|
||||||
|
PadRawStatus_Timeout = 0xFF,
|
||||||
|
};
|
||||||
|
typedef Enum_(U1, PadRawId) {
|
||||||
|
PadRawId_Digital = 0x41,
|
||||||
|
PadRawId_AnalogStick = 0x53,
|
||||||
|
PadRawId_AnalogPadMask = 0xF0,
|
||||||
|
PadRawId_AnalogPadValue = 0x70,
|
||||||
|
};
|
||||||
|
typedef Enum_(U1, PadUnknownId) {
|
||||||
|
PadUnknownId_Sentinel = 0xFF,
|
||||||
|
};
|
||||||
|
typedef Enum_(U4, PadAxisCentered) {
|
||||||
|
PadAxis_Centered_Hi = 0x8080,
|
||||||
|
PadAxis_Centered_Lo = 0x8080,
|
||||||
|
PadAxis_Centered = 0x80808080U,
|
||||||
|
};
|
||||||
|
typedef Enum_(U1, PadDeadZone) {
|
||||||
|
PadDeadZone_LowBound = 0x70, /* left_x < LowBound → active; delta = 0x80 - left_x > 0 (rightward pull) */
|
||||||
|
PadDeadZone_Center = 0x80, /* analog rest position; left_x == Center → delta = 0 (no rotation) */
|
||||||
|
PadDeadZone_HighBound = 0x90, /* left_x > HighBound → active; delta = 0x80 - left_x < 0 (leftward pull) */
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
typedef Struct_(PadAxes) {
|
||||||
|
V2_U1 left; /* offset 8-9 */
|
||||||
|
V2_U1 right; /* offset 10-11 */
|
||||||
|
};
|
||||||
|
// Field order is chosen so that the 4 axes (left_x, left_y, right_x, right_y)
|
||||||
|
// form a contiguous 4-byte block at offset 8, allowing a single `store_word` to clear-or-write all 4 axes in one MIPS instruction.
|
||||||
|
typedef Struct_(PadState) {
|
||||||
|
PadStatus status; /* offset 0, (U4) */
|
||||||
|
PadBtns buttons; /* offset 4, */
|
||||||
|
U1 id; /* offset 6, */
|
||||||
|
byte_pad(1); /* offset 7, explicit pad to align the axes block */
|
||||||
|
union {
|
||||||
|
A2_V2_U1 axes; /* offset 8-11 store_target (4-byte aligned)*/
|
||||||
|
struct {
|
||||||
|
V2_U1 left; /* offset 8-9 */
|
||||||
|
V2_U1 right; /* offset 10-11 */
|
||||||
|
};
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
internal void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1);
|
||||||
@@ -0,0 +1,7 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
# include "psyq.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
ATOM_FILE_DEBUGGER_LINE_MARKER(pysq_atom_c);
|
||||||
@@ -0,0 +1,121 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# pragma once
|
||||||
|
# include "dsl.h"
|
||||||
|
# include "math.h"
|
||||||
|
# include "gp.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
typedef Struct_(DrawEnv_Packed) { U4 tag; U4 code[15]; };
|
||||||
|
typedef Struct_(DrawEnv) {
|
||||||
|
Rect_S2 clip_area;
|
||||||
|
V2_S2 drawing_offset[2];
|
||||||
|
Rect_S2 texture_window;
|
||||||
|
S2 texture_page;
|
||||||
|
B1 flag_dither;
|
||||||
|
B1 flag_draw_on_display;
|
||||||
|
B1 enable_auto_clear;
|
||||||
|
RGB8 initial_bg_color;
|
||||||
|
DrawEnv_Packed dr_env; // reserved
|
||||||
|
};
|
||||||
|
typedef Struct_(DisplayEnv) {
|
||||||
|
Rect_S2 display_area;
|
||||||
|
Rect_S2 screen;
|
||||||
|
B1 vinterlace;
|
||||||
|
B1 color24;
|
||||||
|
B1 pad0;
|
||||||
|
B1 pad1;
|
||||||
|
};
|
||||||
|
typedef Array_(DrawEnv, 2);
|
||||||
|
typedef Array_(DisplayEnv, 2);
|
||||||
|
|
||||||
|
typedef Struct_(DoubleBuffer) {
|
||||||
|
A2_DrawEnv draw;
|
||||||
|
A2_DisplayEnv display;
|
||||||
|
};
|
||||||
|
|
||||||
|
DisplayEnv* displayenv_init(DisplayEnv* env, S4 x, S4 y, S4 w, S4 h) asm("SetDefDispEnv");
|
||||||
|
DrawEnv* drawenv_init (DrawEnv* env, S4 x, S4 y, S4 w, S4 h) asm("SetDefDrawEnv");
|
||||||
|
|
||||||
|
DisplayEnv* displayenv_put(DisplayEnv* env) asm("PutDispEnv");
|
||||||
|
DrawEnv* drawenv_put (DrawEnv* env) asm("PutDrawEnv");
|
||||||
|
|
||||||
|
U4 geom_init(void) asm("InitGeom");
|
||||||
|
void geom_set_offset(U4 x, U4 y) asm("SetGeomOffset");
|
||||||
|
void geom_set_screen(U4 h) asm("SetGeomScreen");
|
||||||
|
|
||||||
|
U4* orderingtbl_clear_reverse(U4* ot, U4 len) asm("ClearOTagR");
|
||||||
|
|
||||||
|
U4 reset_graph(U4 mode) asm("ResetGraph");
|
||||||
|
void set_display_enabled(U4 mask) asm("SetDispMask");
|
||||||
|
|
||||||
|
U4 draw_sync(U4 mode) asm("DrawSync");
|
||||||
|
U4 vsync(U4 mode) asm("VSync");
|
||||||
|
|
||||||
|
void draw_orderingtbl(U4* buf) asm("DrawOTag");
|
||||||
|
|
||||||
|
typedef Struct_(Tile) {
|
||||||
|
U4 tag;
|
||||||
|
RGB8 color;
|
||||||
|
B1 code;
|
||||||
|
Rect_S2 rect;
|
||||||
|
};
|
||||||
|
|
||||||
|
/*
|
||||||
|
Linear Algebra
|
||||||
|
*/
|
||||||
|
|
||||||
|
MT3_S2S4* mt3s2s4_rotation (V3_S2* vec, MT3_S2S4* mat) asm("RotMatrix");
|
||||||
|
MT3_S2S4* mt3s2s4_translation(MT3_S2S4* mat, V3_S4* vec) asm("TransMatrix");
|
||||||
|
MT3_S2S4* mt3s2s4_scale (MT3_S2S4* mat, V3_S4* vec) asm("ScaleMatrix");
|
||||||
|
|
||||||
|
// Rotation, Translation, Perspective
|
||||||
|
|
||||||
|
S4 rtp_v3s2_raw(V3_S2* vec, S4* xy, S4* pp, S4* flag) asm("RotTransPers");
|
||||||
|
FI_ S4 rtp_v3s2(V3_S2* vec, V2_S2* xy, A2_S2* pp, S4* flag) { return rtp_v3s2_raw(vec, C_(S4*R_, & xy->x), C_(S4*R_, pp), r_(flag)); }
|
||||||
|
|
||||||
|
S4 rtp_avg_nclip_a3_v3s2_raw(V3_S2* v0, V3_S2* v1, V3_S2* v2, S4* xy1, S4* xy2, S4* xy3, S4* pp, S4* otz, S4* flag) asm("RotAverageNclip3");
|
||||||
|
FI_ S4 rtp_avg_nclip_a3_v3s2(
|
||||||
|
V3_S2* v0, V3_S2* v1, V3_S2* v2,
|
||||||
|
V2_S2* xy0, V2_S2* xy1, V2_S2* xy2,
|
||||||
|
A2_S2* pp, S4* otz, S4* flag
|
||||||
|
){
|
||||||
|
return rtp_avg_nclip_a3_v3s2_raw(
|
||||||
|
v0, v1, v2,
|
||||||
|
C_(S4*R_, xy0), C_(S4*R_, xy1), C_(S4*R_, xy2),
|
||||||
|
C_(S4*R_, pp), C_(S4*R_, otz), C_(S4*R_, flag)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
S4 rtp_avg_nclip_a4_v3s2_raw(V3_S2* v0, V3_S2* v1, V3_S2* v2, V3_S2* v3, S4* xy1, S4* xy2, S4* xy3, S4* xy4, S4* pp, S4* otz, S4* flag) asm("RotAverageNclip4");
|
||||||
|
FI_ S4 rtp_avg_nclip_a4_v3s2(
|
||||||
|
V3_S2* v0, V3_S2* v1, V3_S2* v2, V3_S2* v3,
|
||||||
|
V2_S2* xy0, V2_S2* xy1, V2_S2* xy2, V2_S2* xy3,
|
||||||
|
A2_S2* pp, S4* otz, S4* flag
|
||||||
|
){
|
||||||
|
return rtp_avg_nclip_a4_v3s2_raw(
|
||||||
|
v0, v1, v2, v3,
|
||||||
|
C_(S4*R_, xy0), C_(S4*R_, xy1), C_(S4*R_, xy2), C_(S4*R_, xy3),
|
||||||
|
C_(S4*R_, pp), C_(S4*R_, otz), C_(S4*R_, flag)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
void gte_matrix_set_rotation (MT3_S2S4* mat) asm("SetRotMatrix");
|
||||||
|
void gte_matrix_set_translation(MT3_S2S4* mat) asm("SetTransMatrix");
|
||||||
|
|
||||||
|
// Einheit, Metrication to unit vector. "Normalization", not Orthogonal "Normal, Normalis". Directionalization.
|
||||||
|
// RGA(Lengyel): Normalize the bulk of a zero-weight direction. This is not finite-point unitization (which forces w=1).
|
||||||
|
S4 psy_normalize_v3s4(V3_S4* v0, V3_S4* v1) asm("VectorNormal");
|
||||||
|
|
||||||
|
// RGA(Lengyel): Apply the matrix expansion of a rigid transformation.
|
||||||
|
// Motor antiproduct is equivalent for unitized points; LA form is what GTE consumes.
|
||||||
|
V3_S4* mul_m3s2_v3s4(MT3_S2S4* m, V3_S4* v, V3_S4* result) asm("ApplyMatrixLV");
|
||||||
|
|
||||||
|
// RGA(Lengyel): Store the full translation column. The motor translator would store half this displacement in m.xyz.
|
||||||
|
MT3_S2S4* trans_m3s2(MT3_S2S4* m, V3_S4* off) asm("TransMatrix");
|
||||||
|
|
||||||
|
MT3_S2S4* gte_comp_coord_m3s2(MT3_S2S4* m0, MT3_S2S4* m1, MT3_S2S4* result) asm("CompMatrixLV");
|
||||||
|
|
||||||
|
// RGA(Lengyel): Complement(Wedge(a,b)), i.e. the Euclidean 3D complement of the exterior product, stored as a V3_S4.
|
||||||
|
// The underlying GTE OP is a specialized signed-16-bit D x IR command; the wedge interpretation is a 3D dual of the same 3 scalars.
|
||||||
|
void cross_v3s4(V3_S4* v0, V3_S4* v1, V3_S4* result) asm("OuterProduct12");
|
||||||
|
|
||||||
@@ -1,15 +1,22 @@
|
|||||||
// tape_atom.metadata.h
|
// word_count.metadata.h
|
||||||
// Single source of truth for instruction-word counts.
|
// Single source of truth for instruction-word counts.
|
||||||
// Used by C (to define compile-time constants) AND Python (to count positions).
|
// Used by C (to define compile-time constants) AND Python (to count positions).
|
||||||
//
|
//
|
||||||
// Format: WORD_COUNT(MACRO_NAME, COUNT)
|
// Format: WORD_COUNT(MACRO_NAME, COUNT)
|
||||||
// One line per macro that appears in your atom sources.
|
// One line per macro that appears in your atom sources.
|
||||||
//
|
//
|
||||||
|
// This file is encoding-macros-only.
|
||||||
|
// The auto-generated component macros (mac_X) live in the source directory's own gen/macs.h (per-directory aggregation; included separately by the unity build).
|
||||||
|
// The unity build should include THIS file and the .macs.h file in the same TU, with both wrapped
|
||||||
|
// (or the include guard order handled) to avoid WORD_COUNT redeclaration.
|
||||||
|
//
|
||||||
// To regenerate: hand-count the instructions in each macro definition.
|
// To regenerate: hand-count the instructions in each macro definition.
|
||||||
// (You'll only need to do this once per macro — they don't change often.)
|
// (You'll only need to do this once per macro — they don't change often.)
|
||||||
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
||||||
|
|
||||||
WORD_COUNT(nop, 1)
|
WORD_COUNT(nop, 1)
|
||||||
|
WORD_COUNT(atom_label, 0)
|
||||||
|
WORD_COUNT(atom_offset, 0)
|
||||||
WORD_COUNT(load_upper_i, 1)
|
WORD_COUNT(load_upper_i, 1)
|
||||||
WORD_COUNT(jump_reg, 1)
|
WORD_COUNT(jump_reg, 1)
|
||||||
WORD_COUNT(jump_link, 1)
|
WORD_COUNT(jump_link, 1)
|
||||||
@@ -17,37 +24,49 @@ WORD_COUNT(call_reg, 1)
|
|||||||
WORD_COUNT(call_addr, 1)
|
WORD_COUNT(call_addr, 1)
|
||||||
WORD_COUNT(branch_le_zero, 1)
|
WORD_COUNT(branch_le_zero, 1)
|
||||||
WORD_COUNT(branch_equal, 1)
|
WORD_COUNT(branch_equal, 1)
|
||||||
|
WORD_COUNT(branch_ne, 1)
|
||||||
WORD_COUNT(add_ui, 1)
|
WORD_COUNT(add_ui, 1)
|
||||||
WORD_COUNT(set_lt_u, 1)
|
WORD_COUNT(set_lt_u, 1)
|
||||||
WORD_COUNT(set_lt_s, 1)
|
WORD_COUNT(set_lt_s, 1)
|
||||||
WORD_COUNT(set_lt_si, 1)
|
WORD_COUNT(set_lt_si, 1)
|
||||||
WORD_COUNT(set_lt_ui, 1)
|
WORD_COUNT(set_lt_ui, 1)
|
||||||
WORD_COUNT(load_ui, 1)
|
|
||||||
WORD_COUNT(load_word, 1)
|
WORD_COUNT(load_word, 1)
|
||||||
WORD_COUNT(load_half_u, 1)
|
WORD_COUNT(load_half_u, 1)
|
||||||
|
WORD_COUNT(load_byte_u, 1)
|
||||||
WORD_COUNT(store_word, 1)
|
WORD_COUNT(store_word, 1)
|
||||||
|
WORD_COUNT(store_byte, 1)
|
||||||
WORD_COUNT(add_ui_self, 1)
|
WORD_COUNT(add_ui_self, 1)
|
||||||
WORD_COUNT(add_u_self, 1)
|
WORD_COUNT(add_u_self, 1)
|
||||||
WORD_COUNT(add_u, 1)
|
WORD_COUNT(add_u, 1)
|
||||||
WORD_COUNT(or_i, 1)
|
WORD_COUNT(or_i, 1)
|
||||||
|
WORD_COUNT(or_i_self, 1)
|
||||||
WORD_COUNT(or_u, 1)
|
WORD_COUNT(or_u, 1)
|
||||||
|
WORD_COUNT(or_u_self, 1)
|
||||||
|
WORD_COUNT(nor_u, 1)
|
||||||
WORD_COUNT(shift_lleft, 1)
|
WORD_COUNT(shift_lleft, 1)
|
||||||
|
WORD_COUNT(shift_lleft_self, 1)
|
||||||
WORD_COUNT(shift_lright, 1)
|
WORD_COUNT(shift_lright, 1)
|
||||||
WORD_COUNT(shift_aright, 1)
|
WORD_COUNT(shift_aright, 1)
|
||||||
WORD_COUNT(mask_upper, 2)
|
WORD_COUNT(mask_upper, 2)
|
||||||
WORD_COUNT(gte_mv_from_data_r, 1)
|
WORD_COUNT(gte_mv_from_data_r, 1)
|
||||||
WORD_COUNT(gte_mv_from_ctrl_r, 1)
|
WORD_COUNT(gte_mv_from_ctrl_r, 1)
|
||||||
WORD_COUNT(gte_mv_to_data_r, 1)
|
WORD_COUNT(gte_mv_to_data_r, 1)
|
||||||
WORD_COUNT(gte_mv_to_ctrl_r, 1)
|
WORD_COUNT(gte_mv_to_ctrl_r, 1)
|
||||||
WORD_COUNT(gte_sw, 1)
|
WORD_COUNT(gte_sw, 1)
|
||||||
WORD_COUNT(gte_cmdw_rtpt, 1)
|
WORD_COUNT(gte_cmdw_rtpt, 1)
|
||||||
WORD_COUNT(gte_cmdw_nclip, 1)
|
WORD_COUNT(gte_cmdw_nclip, 1)
|
||||||
|
WORD_COUNT(gte_cmdw_op, 1)
|
||||||
WORD_COUNT(gte_avg_sort_z3, 1)
|
WORD_COUNT(gte_avg_sort_z3, 1)
|
||||||
WORD_COUNT(mac_load_tri_indices, 3)
|
WORD_COUNT(gte_cmdw_sqr, 1)
|
||||||
WORD_COUNT(mac_load_tri_verts, 18)
|
WORD_COUNT(gte_cmdw_gpf, 1)
|
||||||
WORD_COUNT(mac_format_f3_color, 3)
|
WORD_COUNT(shift_lleft_var, 1)
|
||||||
WORD_COUNT(mac_gte_store_f3, 3)
|
WORD_COUNT(shift_aright_var, 1)
|
||||||
WORD_COUNT(mac_insert_ot_tag, 11)
|
WORD_COUNT(li_s, 1)
|
||||||
WORD_COUNT(mac_yield, 4)
|
WORD_COUNT(and_i, 1)
|
||||||
|
WORD_COUNT(add_si, 1)
|
||||||
|
WORD_COUNT(branch_lt_zero, 1)
|
||||||
|
WORD_COUNT(sub_s, 1)
|
||||||
|
WORD_COUNT(sub_u, 1)
|
||||||
|
WORD_COUNT(nop2, 2)
|
||||||
|
|
||||||
#undef WORD_COUNT
|
#undef WORD_COUNT
|
||||||
@@ -17,19 +17,19 @@ enum {
|
|||||||
};
|
};
|
||||||
|
|
||||||
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
||||||
typedef def_farray(OrderingTable_Buffer, 2);
|
typedef Array_(OrderingTable_Buffer, 2);
|
||||||
|
|
||||||
typedef B1 PrimitiveBuffer[PrimitiveBuff_Len];
|
typedef B1 PrimitiveBuffer[PrimitiveBuff_Len];
|
||||||
typedef def_farray(PrimitiveBuffer, 2);
|
typedef Array_(PrimitiveBuffer, 2);
|
||||||
typedef def_struct(PrimitiveArena) {
|
typedef Struct_(PrimitiveArena) {
|
||||||
A2_PrimitiveBuffer buf;
|
A2_PrimitiveBuffer buf;
|
||||||
U4 used;
|
U4 used;
|
||||||
};
|
};
|
||||||
|
|
||||||
#define Cube_num_verts 8
|
#define Cube_num_verts 8
|
||||||
typedef def_farray(V3_S2, Cube_num_verts);
|
typedef Array_(V3_S2, Cube_num_verts);
|
||||||
#define Cube_num_faces 6
|
#define Cube_num_faces 6
|
||||||
typedef def_farray(V4_S2, Cube_num_faces);
|
typedef Array_(V4_S2, Cube_num_faces);
|
||||||
void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
||||||
memory_copy(verts, & (A8_V3_S2) {
|
memory_copy(verts, & (A8_V3_S2) {
|
||||||
{ -128, -128, -128 },
|
{ -128, -128, -128 },
|
||||||
@@ -40,7 +40,7 @@ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
|||||||
{ 128, 128, -128 },
|
{ 128, 128, -128 },
|
||||||
{ 128, 128, 128 },
|
{ 128, 128, 128 },
|
||||||
{ -128, 128, 128 }
|
{ -128, 128, 128 }
|
||||||
}, size_of(A8_V3_S2) );
|
}, S_(A8_V3_S2) );
|
||||||
memory_copy(faces, & (A6_V4_S2) {
|
memory_copy(faces, & (A6_V4_S2) {
|
||||||
{ 3, 2, 0, 1 },
|
{ 3, 2, 0, 1 },
|
||||||
{ 0, 1, 4, 5 },
|
{ 0, 1, 4, 5 },
|
||||||
@@ -48,10 +48,10 @@ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
|||||||
{ 1, 2, 5, 6 },
|
{ 1, 2, 5, 6 },
|
||||||
{ 2, 3, 6, 7 },
|
{ 2, 3, 6, 7 },
|
||||||
{ 3, 0, 7, 4 },
|
{ 3, 0, 7, 4 },
|
||||||
}, size_of(A6_V4_S2) );
|
}, S_(A6_V4_S2) );
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
typedef def_struct(Ent_Cube) {
|
typedef Struct_(Ent_Cube) {
|
||||||
V3_S4 accel;
|
V3_S4 accel;
|
||||||
V3_S4 vel;
|
V3_S4 vel;
|
||||||
V3_S4 pos;
|
V3_S4 pos;
|
||||||
@@ -62,22 +62,22 @@ typedef def_struct(Ent_Cube) {
|
|||||||
};
|
};
|
||||||
|
|
||||||
#define Floor_num_verts 4
|
#define Floor_num_verts 4
|
||||||
typedef def_farray(V3_S2, Floor_num_verts);
|
typedef Array_(V3_S2, Floor_num_verts);
|
||||||
#define Floor_num_faces 2
|
#define Floor_num_faces 2
|
||||||
typedef def_farray(V3_S2, Floor_num_faces);
|
typedef Array_(V3_S2, Floor_num_faces);
|
||||||
void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
|
void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
|
||||||
memory_copy(verts, &(A4_V3_S2) {
|
memory_copy(verts, &(A4_V3_S2) {
|
||||||
{ -900, 0, -900 },
|
{ -900, 0, -900 },
|
||||||
{ -900, 0, 900 },
|
{ -900, 0, 900 },
|
||||||
{ 900, 0, -900 },
|
{ 900, 0, -900 },
|
||||||
{ 900, 0, 900 },
|
{ 900, 0, 900 },
|
||||||
}, size_of(A8_V3_S2));
|
}, S_(A8_V3_S2));
|
||||||
memory_copy(faces, & (A2_V3_S2) {
|
memory_copy(faces, & (A2_V3_S2) {
|
||||||
{ 0, 1, 2 },
|
{ 0, 1, 2 },
|
||||||
{ 1, 3, 2 },
|
{ 1, 3, 2 },
|
||||||
}, size_of(A2_V3_S2));
|
}, S_(A2_V3_S2));
|
||||||
};
|
};
|
||||||
typedef def_struct(Ent_Floor) {
|
typedef Struct_(Ent_Floor) {
|
||||||
V3_S4 accel;
|
V3_S4 accel;
|
||||||
V3_S4 pos;
|
V3_S4 pos;
|
||||||
V3_S4 scale;
|
V3_S4 scale;
|
||||||
@@ -86,7 +86,7 @@ typedef def_struct(Ent_Floor) {
|
|||||||
A2_V3_S2 faces;
|
A2_V3_S2 faces;
|
||||||
};
|
};
|
||||||
|
|
||||||
typedef def_struct(SMemory) {
|
typedef Struct_(SMemory) {
|
||||||
DoubleBuffer screen_buf;
|
DoubleBuffer screen_buf;
|
||||||
A2_OrderingTable_Buffer ordering_tbl;
|
A2_OrderingTable_Buffer ordering_tbl;
|
||||||
PrimitiveArena primitives;
|
PrimitiveArena primitives;
|
||||||
@@ -108,7 +108,7 @@ B1* prim__alloc(U4 type_width, Str8 type_name) {
|
|||||||
pa->used += type_width;
|
pa->used += type_width;
|
||||||
return next;
|
return next;
|
||||||
}
|
}
|
||||||
#define prim_alloc(type) (type*)prim__alloc(size_of(type), txt( stringify(type)))
|
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
|
||||||
|
|
||||||
void gp_screen_init_c11(DoubleBuffer* screen_buf, S2* active_buf_id)
|
void gp_screen_init_c11(DoubleBuffer* screen_buf, S2* active_buf_id)
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -5,8 +5,8 @@
|
|||||||
# include "duffle/gp.h"
|
# include "duffle/gp.h"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
typedef def_struct(DrawEnv_Packed) { U4 tag; U4 code[15]; };
|
typedef Struct_(DrawEnv_Packed) { U4 tag; U4 code[15]; };
|
||||||
typedef def_struct(DrawEnv) {
|
typedef Struct_(DrawEnv) {
|
||||||
Rect_S2 clip_area;
|
Rect_S2 clip_area;
|
||||||
A2_S2 drawing_offset;
|
A2_S2 drawing_offset;
|
||||||
Rect_S2 texture_window;
|
Rect_S2 texture_window;
|
||||||
@@ -17,7 +17,7 @@ typedef def_struct(DrawEnv) {
|
|||||||
RGB8 initial_bg_color;
|
RGB8 initial_bg_color;
|
||||||
DrawEnv_Packed dr_env; // reserved
|
DrawEnv_Packed dr_env; // reserved
|
||||||
};
|
};
|
||||||
typedef def_struct(DisplayEnv) {
|
typedef Struct_(DisplayEnv) {
|
||||||
Rect_S2 display_area;
|
Rect_S2 display_area;
|
||||||
Rect_S2 screen;
|
Rect_S2 screen;
|
||||||
B1 vinterlace;
|
B1 vinterlace;
|
||||||
@@ -25,9 +25,9 @@ typedef def_struct(DisplayEnv) {
|
|||||||
B1 pad0;
|
B1 pad0;
|
||||||
B1 pad1;
|
B1 pad1;
|
||||||
};
|
};
|
||||||
typedef def_farray(DrawEnv, 2);
|
typedef Array_(DrawEnv, 2);
|
||||||
typedef def_farray(DisplayEnv, 2);
|
typedef Array_(DisplayEnv, 2);
|
||||||
typedef def_struct(DoubleBuffer) {
|
typedef Struct_(DoubleBuffer) {
|
||||||
A2_DrawEnv draw;
|
A2_DrawEnv draw;
|
||||||
A2_DisplayEnv display;
|
A2_DisplayEnv display;
|
||||||
};
|
};
|
||||||
@@ -58,7 +58,7 @@ U4 vsync(U4 mode) __asm__("VSync");
|
|||||||
|
|
||||||
void draw_orderingtbl(U4* buf) __asm__("DrawOTag");
|
void draw_orderingtbl(U4* buf) __asm__("DrawOTag");
|
||||||
|
|
||||||
typedef def_struct(PolyTag) {
|
typedef Struct_(PolyTag) {
|
||||||
U4 addr: 24;
|
U4 addr: 24;
|
||||||
U4 len: 8;
|
U4 len: 8;
|
||||||
RGB8 color;
|
RGB8 color;
|
||||||
@@ -106,7 +106,7 @@ typedef def_struct(PolyTag) {
|
|||||||
// #define setLineF4(p) set_len(p, 6), set_code(p, 0x4c),(p)->pad = 0x55555555
|
// #define setLineF4(p) set_len(p, 6), set_code(p, 0x4c),(p)->pad = 0x55555555
|
||||||
// #define setLineG4(p) set_len(p, 9), set_code(p, 0x5c),(p)->pad = 0x55555555, (p)->p2 = 0, (p)->p3 = 0
|
// #define setLineG4(p) set_len(p, 9), set_code(p, 0x5c),(p)->pad = 0x55555555, (p)->p2 = 0, (p)->p3 = 0
|
||||||
|
|
||||||
typedef def_struct(Poly_F3) {
|
typedef Struct_(Poly_F3) {
|
||||||
U4 tag;
|
U4 tag;
|
||||||
RGB8 color;
|
RGB8 color;
|
||||||
B1 code;
|
B1 code;
|
||||||
@@ -120,14 +120,14 @@ typedef def_struct(Poly_F3) {
|
|||||||
};
|
};
|
||||||
};
|
};
|
||||||
|
|
||||||
typedef def_struct(Poly_G3) {
|
typedef Struct_(Poly_G3) {
|
||||||
U4 tag; RGB8 c0; B1 code;
|
U4 tag; RGB8 c0; B1 code;
|
||||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||||
V2_S2 p2;
|
V2_S2 p2;
|
||||||
};
|
};
|
||||||
|
|
||||||
typedef def_struct(Poly_F4) {
|
typedef Struct_(Poly_F4) {
|
||||||
U4 tag;
|
U4 tag;
|
||||||
RGB8 color;
|
RGB8 color;
|
||||||
B1 code;
|
B1 code;
|
||||||
@@ -142,7 +142,7 @@ typedef def_struct(Poly_F4) {
|
|||||||
};
|
};
|
||||||
};
|
};
|
||||||
|
|
||||||
typedef def_struct(Poly_G4) {
|
typedef Struct_(Poly_G4) {
|
||||||
U4 tag; RGB8 c0; B1 code;
|
U4 tag; RGB8 c0; B1 code;
|
||||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||||
@@ -150,7 +150,7 @@ typedef def_struct(Poly_G4) {
|
|||||||
V2_S2 p3;
|
V2_S2 p3;
|
||||||
};
|
};
|
||||||
|
|
||||||
typedef def_struct(Tile) {
|
typedef Struct_(Tile) {
|
||||||
U4 tag;
|
U4 tag;
|
||||||
RGB8 color;
|
RGB8 color;
|
||||||
B1 code;
|
B1 code;
|
||||||
@@ -169,7 +169,7 @@ M3_S2* m3s2_scale (M3_S2* mat, V3_S4* vec) __asm__("ScaleMatrix");
|
|||||||
// Rotation, Translation, Perspective
|
// Rotation, Translation, Perspective
|
||||||
|
|
||||||
S4 rtp_v3s2_raw(V3_S2* vec, S4* xy, S4* pp, S4* flag) __asm__("RotTransPers");
|
S4 rtp_v3s2_raw(V3_S2* vec, S4* xy, S4* pp, S4* flag) __asm__("RotTransPers");
|
||||||
FI_ S4 rtp_v3s2(V3_S2* vec, V2_S2* xy, A2_S2* pp, S4* flag) { return rtp_v3s2_raw(vec, cast(S4*R_, & xy->x), cast(S4*R_, pp), r_(flag)); }
|
FI_ S4 rtp_v3s2(V3_S2* vec, V2_S2* xy, A2_S2* pp, S4* flag) { return rtp_v3s2_raw(vec, C_(S4*R_, & xy->x), C_(S4*R_, pp), r_(flag)); }
|
||||||
|
|
||||||
S4 rtp_avg_nclip_a3_v3s2_raw(V3_S2* v0, V3_S2* v1, V3_S2* v2, S4* xy1, S4* xy2, S4* xy3, S4* pp, S4* otz, S4* flag) __asm__("RotAverageNclip3");
|
S4 rtp_avg_nclip_a3_v3s2_raw(V3_S2* v0, V3_S2* v1, V3_S2* v2, S4* xy1, S4* xy2, S4* xy3, S4* pp, S4* otz, S4* flag) __asm__("RotAverageNclip3");
|
||||||
FI_ S4 rtp_avg_nclip_a3_v3s2(
|
FI_ S4 rtp_avg_nclip_a3_v3s2(
|
||||||
@@ -179,8 +179,8 @@ FI_ S4 rtp_avg_nclip_a3_v3s2(
|
|||||||
){
|
){
|
||||||
return rtp_avg_nclip_a3_v3s2_raw(
|
return rtp_avg_nclip_a3_v3s2_raw(
|
||||||
v0, v1, v2,
|
v0, v1, v2,
|
||||||
cast(S4*R_, xy0), cast(S4*R_, xy1), cast(S4*R_, xy2),
|
C_(S4*R_, xy0), C_(S4*R_, xy1), C_(S4*R_, xy2),
|
||||||
cast(S4*R_, pp), cast(S4*R_, otz), cast(S4*R_, flag)
|
C_(S4*R_, pp), C_(S4*R_, otz), C_(S4*R_, flag)
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -192,8 +192,8 @@ FI_ S4 rtp_avg_nclip_a4_v3s2(
|
|||||||
){
|
){
|
||||||
return rtp_avg_nclip_a4_v3s2_raw(
|
return rtp_avg_nclip_a4_v3s2_raw(
|
||||||
v0, v1, v2, v3,
|
v0, v1, v2, v3,
|
||||||
cast(S4*R_, xy0), cast(S4*R_, xy1), cast(S4*R_, xy2), cast(S4*R_, xy3),
|
C_(S4*R_, xy0), C_(S4*R_, xy1), C_(S4*R_, xy2), C_(S4*R_, xy3),
|
||||||
cast(S4*R_, pp), cast(S4*R_, otz), cast(S4*R_, flag)
|
C_(S4*R_, pp), C_(S4*R_, otz), C_(S4*R_, flag)
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -1,19 +0,0 @@
|
|||||||
// Auto-generated by tape_atom_offset_gen.meta.lua — DO NOT EDIT
|
|
||||||
// Source: C:\projects\Pikuma\ps1\code\gte_hello\hello_gte_tape.c
|
|
||||||
#pragma once
|
|
||||||
|
|
||||||
#pragma region hello_gte_tape
|
|
||||||
|
|
||||||
|
|
||||||
// --- atom: floor_tri (49 words) ---
|
|
||||||
|
|
||||||
#define _atom_offset_culling_floor_tri_exit 15
|
|
||||||
#define _atom_offset_bounds_chk_floor_tri_exit 3
|
|
||||||
|
|
||||||
enum {
|
|
||||||
atom_offset_culling_floor_tri_exit = _atom_offset_culling_floor_tri_exit,
|
|
||||||
atom_offset_bounds_chk_floor_tri_exit = _atom_offset_bounds_chk_floor_tri_exit,
|
|
||||||
};
|
|
||||||
|
|
||||||
#pragma endregion hello_gte_tape
|
|
||||||
|
|
||||||
@@ -1,234 +0,0 @@
|
|||||||
#ifdef INTELLISENSE_DIRECTIVES
|
|
||||||
# include "duffle/lottes_tape.h"
|
|
||||||
# include "duffle/atom_dsl.h"
|
|
||||||
# include "hello_gte.h"
|
|
||||||
# include "tape_atom.metadata.h"
|
|
||||||
# include "gen/hello_gte_tape.offsets.h"
|
|
||||||
#endif
|
|
||||||
|
|
||||||
#pragma region MACs (Mips Atom components)
|
|
||||||
/* The macros mac_format_f3_color and mac_gte_store_f3 moved to
|
|
||||||
* lottes_tape.h during the Phase 3 gp.h overhaul. Both are now RGB-form
|
|
||||||
* (mac_format_f3_color takes _r, _g, _b byte values rather than raw
|
|
||||||
* 16-bit half-words). */
|
|
||||||
|
|
||||||
enum fack {
|
|
||||||
ah = gp0_cmd_poly_f3 << 8 | 0xFF,
|
|
||||||
};
|
|
||||||
void fk() {
|
|
||||||
(void*)ah;
|
|
||||||
}
|
|
||||||
|
|
||||||
#pragma endregion MACs
|
|
||||||
|
|
||||||
#pragma region Baked Atoms
|
|
||||||
|
|
||||||
typedef Struct_(Binds_CubeTri) {
|
|
||||||
U4 PrimCursor;
|
|
||||||
U4 FaceCursor;
|
|
||||||
U4 VertBase;
|
|
||||||
U4 OtBase;
|
|
||||||
};
|
|
||||||
internal MipsAtom_(rbind_cube_tri) {
|
|
||||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
|
||||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
|
|
||||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
|
|
||||||
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
|
|
||||||
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
|
|
||||||
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
|
||||||
// Note(Ed): This entire thing is argument shuffle?
|
|
||||||
// TODO(Ed): Eliminate
|
|
||||||
mac_yield()
|
|
||||||
};
|
|
||||||
|
|
||||||
/* ============================================================================
|
|
||||||
* cube_tri — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
|
|
||||||
* ============================================================================
|
|
||||||
*
|
|
||||||
* Reads 4 indices from R_FaceCur (V4_S2 = 8 bytes), loads 4 vertices into
|
|
||||||
* the GTE, runs the PsyQ RotAverageNclip4 sequence, and renders a Poly_G4.
|
|
||||||
*/
|
|
||||||
atom_region (cube_tri, REGION_PRIM_ARENA)
|
|
||||||
atom_group (cube_tri, GROUP_RENDER_PRIMS)
|
|
||||||
atom_cadence (cube_tri, CADENCE_FRAME)
|
|
||||||
atom_annot(cube_tri, phase_work,
|
|
||||||
tape_regs(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
|
||||||
tape_regs(R_PrimCursor, R_FaceCursor))
|
|
||||||
internal
|
|
||||||
MipsAtom_(cube_tri) {
|
|
||||||
/* ── 1. Load 4 face indices from R_FaceCur ──────────────────────────── */
|
|
||||||
load_half_u(R_T0, R_FaceCursor, 0), /* T0 = face->x (vertex 0 index) */
|
|
||||||
load_half_u(R_T1, R_FaceCursor, 2), /* T1 = face->y (vertex 1 index) */
|
|
||||||
load_half_u(R_T2, R_FaceCursor, 4), /* T2 = face->z (vertex 2 index) */
|
|
||||||
load_half_u(R_T3, R_FaceCursor, 6), /* T3 = face->w (vertex 3 index) */
|
|
||||||
|
|
||||||
/* ── 2. Load V0, V1, V2 into GTE ────────────────────────────────────── */
|
|
||||||
/* V0 = verts[face->x] */
|
|
||||||
shift_lleft(R_AT, R_T0, 3), add_u(R_AT, R_AT, R_VertBase),
|
|
||||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
|
||||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
|
||||||
|
|
||||||
/* V1 = verts[face->y] */
|
|
||||||
shift_lleft(R_AT, R_T1, 3), add_u(R_AT, R_AT, R_VertBase),
|
|
||||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
|
||||||
gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
|
||||||
|
|
||||||
/* V2 = verts[face->z] */
|
|
||||||
shift_lleft(R_AT, R_T2, 3), add_u(R_AT, R_AT, R_VertBase),
|
|
||||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
|
||||||
gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
|
||||||
|
|
||||||
/* ── 3. RTPT — transforms V0/V1/V2 → SXY0/SXY1/SXY2 + SZ1/SZ2/SZ3 ─── */
|
|
||||||
nop, nop, gte_cmdw_rtpt,
|
|
||||||
|
|
||||||
/* ── 4. NCLIP — backface culling on SXY0/SXY1/SXY2 (p0,p1,p2) ──────── */
|
|
||||||
/* MUST be done BEFORE RTPS overwrites SXY0 with p3! */
|
|
||||||
nop, nop, gte_cmdw_nclip,
|
|
||||||
nop, nop,
|
|
||||||
|
|
||||||
/* ── 5. Cull check: skip format/insert if MAC0 ≤ 0 (backface) ───────── */
|
|
||||||
gte_mv_from_data_r(R_T0, C2_MAC0),
|
|
||||||
nop,
|
|
||||||
branch_le_zero(R_T0, 49), /* Skip 49 if MAC0 ≤ 0 (backface) → cull */
|
|
||||||
nop, /* BD slot */
|
|
||||||
|
|
||||||
/* ── 6. Store p0,p1,p2 to primitive buffer (BEFORE RTPS overwrites) ─── */
|
|
||||||
store_word(R_0, R_PrimCursor, 0),
|
|
||||||
|
|
||||||
/* Word 1: c0 (BGR) + code = 0x38FF00FF (magenta, opcode 0x38) */
|
|
||||||
load_upper_i(R_AT, 0x38FF), or_i(R_AT, R_AT, 0x00FF),
|
|
||||||
store_word(R_AT, R_PrimCursor, 4),
|
|
||||||
|
|
||||||
/* Word 2: p0 = SXY0 (stored BEFORE RTPS overwrites it) */
|
|
||||||
gte_sw(C2_SXY0, R_PrimCursor, 8),
|
|
||||||
|
|
||||||
/* Word 3: c1 (BGR) + pad = 0x0000FFFF (yellow) */
|
|
||||||
load_upper_i(R_AT, 0x0000), or_i(R_AT, R_AT, 0xFFFF),
|
|
||||||
store_word(R_AT, R_PrimCursor, 12),
|
|
||||||
|
|
||||||
/* Word 4: p1 = SXY1 */
|
|
||||||
gte_sw(C2_SXY1, R_PrimCursor, 16),
|
|
||||||
|
|
||||||
/* Word 5: c2 (BGR) + pad = 0x00FFFF00 (cyan) */
|
|
||||||
load_upper_i(R_AT, 0x00FF), or_i(R_AT, R_AT, 0xFF00),
|
|
||||||
store_word(R_AT, R_PrimCursor, 20),
|
|
||||||
|
|
||||||
/* Word 6: p2 = SXY2 */
|
|
||||||
gte_sw(C2_SXY2, R_PrimCursor, 24),
|
|
||||||
|
|
||||||
/* Word 7: c3 (BGR) + pad = 0x0000FF00 (green) */
|
|
||||||
load_upper_i(R_AT, 0x0000), or_i(R_AT, R_AT, 0xFF00),
|
|
||||||
store_word(R_AT, R_PrimCursor, 28),
|
|
||||||
|
|
||||||
/* ── 7. Load V3 = verts[face->w] into V0 ─────────────────────────────── */
|
|
||||||
shift_lleft(R_AT, R_T3, 3), add_u(R_AT, R_AT, R_VertBase),
|
|
||||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
|
||||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
|
||||||
|
|
||||||
/* ── 8. RTPS — transforms V0 (now V3) → SXY0 (p3) + SZ0 ─────────────── */
|
|
||||||
nop, nop, gte_cmdw_rtps,
|
|
||||||
|
|
||||||
/* Word 8: p3 = SXY0 (written AFTER RTPS with V3's screen coords) */
|
|
||||||
gte_sw(C2_SXY0, R_PrimCursor, 32),
|
|
||||||
|
|
||||||
/* ── 9. AVSZ4 — average Z from SZ0/SZ1/SZ2/SZ3 ────────────── */
|
|
||||||
nop, nop, gte_cmdw_avsz4,
|
|
||||||
nop, nop,
|
|
||||||
gte_mv_from_data_r(R_T1, C2_OTZ),
|
|
||||||
|
|
||||||
/* ── 10. Bounds check OTZ < 2048 ─────────────────────────────────────── */
|
|
||||||
add_ui( R_AT, R_0, 2048),
|
|
||||||
set_lt_u( R_AT, R_T1, R_AT),
|
|
||||||
branch_equal(R_AT, R_0, 13), /* Skip 13 → land at add_ui(R_FaceCur,...) */
|
|
||||||
nop, /* BD slot */
|
|
||||||
|
|
||||||
/* ── 11. Insert into Ordering Table (length = 8 for Poly_G4) ─────────── */
|
|
||||||
mac_insert_ot_tag(R_T1, 0x0800), /* 0x0800 = 8 << 8 = length 8 in tag */
|
|
||||||
|
|
||||||
/* ── 12. Advance cursors & yield ─────────────────────────────────────── */
|
|
||||||
add_ui(R_PrimCursor, R_PrimCursor, 36), /* 9 words × 4 bytes */
|
|
||||||
add_ui(R_FaceCursor, R_FaceCursor, 8), /* 4 × S2 = 8 bytes */
|
|
||||||
mac_yield()
|
|
||||||
};
|
|
||||||
|
|
||||||
typedef Struct_(Binds_FloorTri) {
|
|
||||||
U4 PrimCursor;
|
|
||||||
U4 FaceCursor;
|
|
||||||
U4 VertBase;
|
|
||||||
U4 OtBase;
|
|
||||||
};
|
|
||||||
atom_region(rbind_floor_tri, REGION_PRIM_ARENA)
|
|
||||||
atom_group(rbind_floor_tri, GROUP_RENDER_FLOOR)
|
|
||||||
atom_cadence(rbind_floor_tri, CADENCE_FRAME)
|
|
||||||
atom_annot(rbind_floor_tri, phase_bind
|
|
||||||
, atom_reads()
|
|
||||||
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase))
|
|
||||||
internal
|
|
||||||
MipsAtom_(rbind_floor_tri) {
|
|
||||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
|
||||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
|
|
||||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
|
|
||||||
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
|
|
||||||
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
|
|
||||||
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
|
||||||
mac_yield()
|
|
||||||
};
|
|
||||||
|
|
||||||
atom_region( floor_tri, REGION_PRIM_ARENA)
|
|
||||||
atom_group( floor_tri, GROUP_RENDER_FLOOR)
|
|
||||||
atom_cadence(floor_tri, CADENCE_FRAME)
|
|
||||||
atom_annot( floor_tri, phase_work,
|
|
||||||
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
|
||||||
atom_writes(R_PrimCursor, R_FaceCursor))
|
|
||||||
internal
|
|
||||||
MipsAtom_(floor_tri) {
|
|
||||||
mac_load_tri_indices(R_T0, R_T1, R_T2),
|
|
||||||
mac_load_tri_verts( R_T0, R_T1, R_T2),
|
|
||||||
nop, nop, gte_cmdw_rotate_translate_perspective_triple,
|
|
||||||
nop, nop, gte_cmdw_nclip,
|
|
||||||
nop, nop,
|
|
||||||
/* Culling (Branch forward if Backface) */
|
|
||||||
gte_mv_from_data_r(R_T0, C2_MAC0),
|
|
||||||
nop, branch_le_zero(R_T0, atom_offset(culling, floor_tri_exit)),
|
|
||||||
nop,
|
|
||||||
/* Format Primitive */
|
|
||||||
// mac_format_f3_color(0x20FF, 0xFFFF), // works
|
|
||||||
mac_format_f3_color(0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
|
|
||||||
mac_gte_store_f3(),
|
|
||||||
/* Calculate Depth */
|
|
||||||
nop, nop, gte_avg_sort_z3,
|
|
||||||
nop, nop, gte_mv_from_data_r(R_T1, C2_OTZ),
|
|
||||||
/* Bounds Check OTZ < 2048 (Branch forward to skip insertion) */
|
|
||||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
|
||||||
set_lt_u( R_AT, R_T1, R_AT),
|
|
||||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_tri_exit)),
|
|
||||||
nop,
|
|
||||||
/* Insert into Ordering Table Linked List */
|
|
||||||
mac_insert_ot_tag(R_T1, 0x0400),
|
|
||||||
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
|
|
||||||
// Note(Ed): No bounds checking, should be checked before atom runs.
|
|
||||||
|
|
||||||
/* Advance Input Cursor & Yield (Both branch targets land here) */
|
|
||||||
atom_label(floor_tri_exit)
|
|
||||||
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
|
|
||||||
mac_yield()
|
|
||||||
};
|
|
||||||
|
|
||||||
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
|
|
||||||
atom_region( sync_primitive_arena, REGION_PRIM_ARENA)
|
|
||||||
atom_group( sync_primitive_arena, GROUP_RENDER_FLOOR)
|
|
||||||
atom_cadence(sync_primitive_arena, CADENCE_FRAME)
|
|
||||||
atom_annot( sync_primitive_arena, phase_work,
|
|
||||||
atom_reads( R_TapePtr, R_PrimCursor),
|
|
||||||
atom_writes(R_TapePtr))
|
|
||||||
internal MipsAtom_(sync_primitive_arena) {
|
|
||||||
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
|
|
||||||
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
|
|
||||||
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
|
|
||||||
/* Calculate byte offset and store directly back to RAM */
|
|
||||||
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
|
|
||||||
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
|
|
||||||
mac_yield()
|
|
||||||
};
|
|
||||||
|
|
||||||
#pragma endregion Baked Atoms
|
|
||||||
@@ -0,0 +1,13 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
#pragma once
|
||||||
|
#endif
|
||||||
|
// Auto-generated by ps1_meta.lua (passes/auto_reg.lua) — DO NOT EDIT
|
||||||
|
// Directory: C:\projects\Pikuma\ps1\code\hello_camera
|
||||||
|
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.c
|
||||||
|
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.h
|
||||||
|
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.atom.c
|
||||||
|
// Per-phase register allocations resolved by the lua pass.
|
||||||
|
// R_<Sym>_Code = <chosen GPR's _Code constant> for every marker in this directory.
|
||||||
|
|
||||||
|
#define R_GpTmp_Code R_V0_Code
|
||||||
|
|
||||||
@@ -0,0 +1,41 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
#pragma once
|
||||||
|
#endif
|
||||||
|
// Auto-generated by ps1_meta.lua — DO NOT EDIT
|
||||||
|
// Directory: C:\projects\Pikuma\ps1\code\hello_camera/
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.atom.c
|
||||||
|
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
|
||||||
|
|
||||||
|
#ifndef WORD_COUNT
|
||||||
|
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#define mac_put_disp_env(reg_transfer, reg_base, port) \
|
||||||
|
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port)
|
||||||
|
WORD_COUNT(mac_put_disp_env, 5)
|
||||||
|
|
||||||
|
#define mac_put_draw_env(reg_transfer, reg_base, port) \
|
||||||
|
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port) /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */ \
|
||||||
|
, mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port) /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */ \
|
||||||
|
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[1] TextureWindow (tw=(0,0)) */ \
|
||||||
|
, mac_gcmd_push(enc_gp0_draw_area_tl_word(0, ScreenRes_Y), reg_transfer, reg_base, port) /* code[2] DrawArea top-left (clip.x=0, clip.y=ScreenRes_Y=240) */ \
|
||||||
|
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) /* code[3] DrawArea bottom-right (clip.x+w=320, clip.y+h=480) */ \
|
||||||
|
, mac_gcmd_push(gp0_word_set_draw_offset(), reg_transfer, reg_base, port) /* code[4] DrawOffset (ofs=(0,0)) — bare-cmd word; the GPU uses the current state machine. */ \
|
||||||
|
, mac_gcmd_push(gp0_word_dr_env_mask(), reg_transfer, reg_base, port) /* code[5] Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit. */ \
|
||||||
|
, mac_gcmd_push(gp0_word_dr_env_bg_color_cmd(1, 7, 7, 7), reg_transfer, reg_base, port) /* code[6] Initial-bg-color + auto-clear (isbg=1, r=7, g=7, b=7). */ \
|
||||||
|
, mac_gcmd_push(gp0_word_dr_env_draw_mode(1), reg_transfer, reg_base, port) /* code[7] Re-assert DrawMode with isbg=1 (isbg-flag set; the 0xE1 cmd byte plus isbg only). */ /* code[8..10] Padding (NOP — GPU discards; the DR_ENV requires 16 words total). */ \
|
||||||
|
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) /* code[11..12] TextureWindow bottom-right (tw.x+tw.w=0, tw.y+tw.h=0) — libpsyx emits twice. */ \
|
||||||
|
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[13..14] Padding (NOP) — completes the 16-word packet. */ \
|
||||||
|
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port)
|
||||||
|
WORD_COUNT(mac_put_draw_env, 16)
|
||||||
|
|
||||||
@@ -0,0 +1,68 @@
|
|||||||
|
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
|
||||||
|
// Directory: C:\projects\Pikuma\ps1\code\hello_camera\
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.atom.c
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#pragma region hello_camera
|
||||||
|
|
||||||
|
|
||||||
|
// --- atom: pad_input_cube_rotation (60 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_dpad_left_exit_dpad_left 6
|
||||||
|
#define _atom_offset_dpad_right_exit_dpad_right 6
|
||||||
|
#define _atom_offset_dead_zone_low_check_dead_low_active 8
|
||||||
|
#define _atom_offset_dead_zone_high_check_dead_high_active 15
|
||||||
|
#define _atom_offset_dead_zone_skip_exit_stick 24
|
||||||
|
#define _atom_offset_end_low_exit_stick 12
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_dpad_left_exit_dpad_left = _atom_offset_dpad_left_exit_dpad_left,
|
||||||
|
atom_offset_dpad_right_exit_dpad_right = _atom_offset_dpad_right_exit_dpad_right,
|
||||||
|
atom_offset_dead_zone_low_check_dead_low_active = _atom_offset_dead_zone_low_check_dead_low_active,
|
||||||
|
atom_offset_dead_zone_high_check_dead_high_active = _atom_offset_dead_zone_high_check_dead_high_active,
|
||||||
|
atom_offset_dead_zone_skip_exit_stick = _atom_offset_dead_zone_skip_exit_stick,
|
||||||
|
atom_offset_end_low_exit_stick = _atom_offset_end_low_exit_stick,
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- atom: pad_input_cam (39 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_left_x_exit_left_x 3
|
||||||
|
#define _atom_offset_right_x_exit_right_x 3
|
||||||
|
#define _atom_offset_up_y_exit_up_y 3
|
||||||
|
#define _atom_offset_down_y_exit_down_y 3
|
||||||
|
#define _atom_offset_cross_z_exit_cross_z 3
|
||||||
|
#define _atom_offset_circle_z_exit_circle_z 3
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_left_x_exit_left_x = _atom_offset_left_x_exit_left_x,
|
||||||
|
atom_offset_right_x_exit_right_x = _atom_offset_right_x_exit_right_x,
|
||||||
|
atom_offset_up_y_exit_up_y = _atom_offset_up_y_exit_up_y,
|
||||||
|
atom_offset_down_y_exit_down_y = _atom_offset_down_y_exit_down_y,
|
||||||
|
atom_offset_cross_z_exit_cross_z = _atom_offset_cross_z_exit_cross_z,
|
||||||
|
atom_offset_circle_z_exit_circle_z = _atom_offset_circle_z_exit_circle_z,
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- atom: cube_g4_face (73 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_cull_cube_g4_face_exit 41
|
||||||
|
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
|
||||||
|
atom_offset_bounds_chk_cube_g4_face_exit = _atom_offset_bounds_chk_cube_g4_face_exit,
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- atom: floor_f3_face (56 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_culling_floor_f3_face_exit 25
|
||||||
|
#define _atom_offset_bounds_chk_floor_f3_face_exit 16
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
|
||||||
|
atom_offset_bounds_chk_floor_f3_face_exit = _atom_offset_bounds_chk_floor_f3_face_exit,
|
||||||
|
};
|
||||||
|
|
||||||
|
#pragma endregion hello_camera
|
||||||
|
|
||||||
@@ -0,0 +1,643 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# pragma once
|
||||||
|
# include "duffle/gen/macs.h"
|
||||||
|
# include "duffle/gen/offsets.h"
|
||||||
|
# include "duffle/dsl.atom.h"
|
||||||
|
# include "duffle/lottes_tape.h"
|
||||||
|
# include "duffle/mips.h"
|
||||||
|
# include "duffle/gte.h"
|
||||||
|
# include "duffle/gp.h"
|
||||||
|
# include "duffle/pad.h"
|
||||||
|
# include "duffle/word_count.metadata.h"
|
||||||
|
# include "duffle/psyq.h"
|
||||||
|
# include "duffle/math.atom.h"
|
||||||
|
# include "duffle/mips.atom.c"
|
||||||
|
# include "duffle/gte.atom.c"
|
||||||
|
# include "duffle/gp.atom.c"
|
||||||
|
# include "duffle/psyq.atom.c"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/auto_reg.h"
|
||||||
|
# include "hello_camera.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c);
|
||||||
|
|
||||||
|
#pragma region MACs (Mips Atom components)
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_put_disp_env(AtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
||||||
|
MipsAtomComp_Proc_(ab, {
|
||||||
|
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
|
||||||
|
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
|
||||||
|
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
||||||
|
})
|
||||||
|
|
||||||
|
I_ Slice_MipsCode ac_put_draw_env(AtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
||||||
|
MipsAtomComp_Proc_(ab, {
|
||||||
|
/*
|
||||||
|
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
|
||||||
|
* References:
|
||||||
|
* - libpsyx source: `toolchain/psyq-4_7/lib/libgpu.a` (binary, function `PutDrawEnv`)
|
||||||
|
* - PSX-SPX doc: https://problemkaputt.de/psx-spx.htm#gputdrawingcommands
|
||||||
|
* - PSYQ SDK: `setdrawenv` / `makelongdr_env` source
|
||||||
|
* - NOCASH PSX spec: §"GP0(E1h) Draw Mode setting" through §"DR_ENV"
|
||||||
|
*
|
||||||
|
* The 16-word format is documented in the PSYQ SDK manual and on NOCASH's PSX-spec.txt. The libpsyx reference is at:
|
||||||
|
* ./toolchain/psyq-4_7/lib/libgpu.a
|
||||||
|
* (binary; the PutDrawEnv implementation builds the 16-word DR_ENV from the user's DRAWENV struct and emits it via GP0 GPU commands.)
|
||||||
|
*
|
||||||
|
* Word indices (libpsyx PutDrawEnv / SetDrawEnv order):
|
||||||
|
* tag = (length << 24) | addr — 16-word packet (1 tag + 15 code)
|
||||||
|
* code[0] = DrawMode (dfe=1, dtd=0, tpage=0) — must come first per libpsyx
|
||||||
|
* code[1] = TextureWindow (tw=(0,0)) — bare-cmd word; GPU uses current state
|
||||||
|
* code[2] = DrawArea top-left (clip.x=0, clip.y=240)
|
||||||
|
* code[3] = DrawArea bottom-right (clip.x+w=320, clip.y+h=480)
|
||||||
|
* code[4] = DrawOffset (ofs=(0,0)) — bare-cmd word
|
||||||
|
* code[5] = Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit
|
||||||
|
* code[6] = Initial-bg-color (isbg=1, r=7, g=7, b=7)
|
||||||
|
* code[7] = DrawMode (isbg=1, tpage=0) — re-asserts DrawMode with isbg
|
||||||
|
* code[8..10] = padding (NOP) — 3 words to fill the packet
|
||||||
|
* code[11..12] = TextureWindow bottom-right — defaults to (0,0,0,0)
|
||||||
|
* code[13..14] = padding (NOP) — completes the 16-word packet
|
||||||
|
*/
|
||||||
|
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port), /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */
|
||||||
|
mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port), /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */
|
||||||
|
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port), /* code[1] TextureWindow (tw=(0,0)) */
|
||||||
|
mac_gcmd_push(enc_gp0_draw_area_tl_word(0, ScreenRes_Y), reg_transfer, reg_base, port), /* code[2] DrawArea top-left (clip.x=0, clip.y=ScreenRes_Y=240) */
|
||||||
|
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port), /* code[3] DrawArea bottom-right (clip.x+w=320, clip.y+h=480) */
|
||||||
|
|
||||||
|
mac_gcmd_push(gp0_word_set_draw_offset(), reg_transfer, reg_base, port), /* code[4] DrawOffset (ofs=(0,0)) — bare-cmd word; the GPU uses the current state machine. */
|
||||||
|
mac_gcmd_push(gp0_word_dr_env_mask(), reg_transfer, reg_base, port), /* code[5] Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit. */
|
||||||
|
mac_gcmd_push(gp0_word_dr_env_bg_color_cmd(1, 7, 7, 7), reg_transfer, reg_base, port), /* code[6] Initial-bg-color + auto-clear (isbg=1, r=7, g=7, b=7). */
|
||||||
|
mac_gcmd_push(gp0_word_dr_env_draw_mode(1), reg_transfer, reg_base, port), /* code[7] Re-assert DrawMode with isbg=1 (isbg-flag set; the 0xE1 cmd byte plus isbg only). */
|
||||||
|
|
||||||
|
/* code[8..10] Padding (NOP — GPU discards; the DR_ENV requires 16 words total). */
|
||||||
|
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||||
|
|
||||||
|
/* code[11..12] TextureWindow bottom-right (tw.x+tw.w=0, tw.y+tw.h=0) — libpsyx emits twice. */
|
||||||
|
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port),
|
||||||
|
|
||||||
|
/* code[13..14] Padding (NOP) — completes the 16-word packet. */
|
||||||
|
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||||
|
})
|
||||||
|
|
||||||
|
#pragma endregion MACs
|
||||||
|
|
||||||
|
#pragma region Atom Procs
|
||||||
|
|
||||||
|
#pragma region resolve_look_at
|
||||||
|
/* ─── resolve_look_at bundle chain atoms ──────────────────────────── */
|
||||||
|
|
||||||
|
typedef AtomBundle_(resolve_look_at) { MipsAtom
|
||||||
|
*input_and_sub,
|
||||||
|
*normalize_fwd_uz,
|
||||||
|
*cross_to_right,
|
||||||
|
*normalize_right_ux,
|
||||||
|
*cross_to_up,
|
||||||
|
*normalize_up_uy,
|
||||||
|
*populate_mt3s4s2;
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef Struct_(ResolveLookAtScratch) {
|
||||||
|
V3_S4 fwd;
|
||||||
|
V3_S4 uz;
|
||||||
|
V3_S4 right;
|
||||||
|
V3_S4 ux;
|
||||||
|
V3_S4 up;
|
||||||
|
V3_S4 uy;
|
||||||
|
P3_S4 eye;
|
||||||
|
P3_S4 target;
|
||||||
|
V3_S4 up_in;
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef Struct_(Binds_ResolveLookAtSub) {
|
||||||
|
P3_S4* target;
|
||||||
|
P3_S4* eye;
|
||||||
|
V3_S4* up_in;
|
||||||
|
};
|
||||||
|
typedef Struct_(RegUse_resolve_look_at_input_and_sub) {
|
||||||
|
Reg target_ptr;
|
||||||
|
Reg eye_ptr;
|
||||||
|
Reg up_in_ptr;
|
||||||
|
union { Reg_(V3_S4) r012, up_in, eye; };
|
||||||
|
union { Reg_(V3_S4) r345, target, fwd; };
|
||||||
|
};
|
||||||
|
/* Atom 0 in the bundle: input_and_sub. Stages C-side inputs into the scratchpad and computes fwd = target - eye. */
|
||||||
|
internal MipsAtom* AtomBundleEntry_(resolve_look_at,input_and_sub)(AtomArena_R aa, RegUse_resolve_look_at_input_and_sub r)
|
||||||
|
atom_info(atom_bind(Binds_ResolveLookAtSub)) MipsAtom_Proc_(aa, {
|
||||||
|
load_word(r.target_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,target)),
|
||||||
|
load_word(r.eye_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)),
|
||||||
|
load_word(r.up_in_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)),
|
||||||
|
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtSub)),
|
||||||
|
|
||||||
|
/* Stage up_in.x/y/z into the scratchpad. R_ScratchBase = R_SP = 0x1F800000. */
|
||||||
|
mac_load_v3s4( r.up_in, r.up_in_ptr, 0), LdSlot_
|
||||||
|
mac_store_v3s4(r.up_in, R_ScratchBase, O_(ResolveLookAtScratch,up_in)),
|
||||||
|
|
||||||
|
// Stage eye.x/y/z into the scratchpad (atom 6 reads these for the translation column).
|
||||||
|
mac_load_v3s4( r.eye, r.eye_ptr, 0), LdSlot_
|
||||||
|
mac_store_v3s4(r.eye, R_ScratchBase, O_(ResolveLookAtScratch,eye)),
|
||||||
|
|
||||||
|
/* Compute fwd = target - eye. */
|
||||||
|
mac_load_v3s4( r.target, r.target_ptr, 0), LdSlot_
|
||||||
|
mac_sub_v3s4_self(r.fwd, r.eye),
|
||||||
|
mac_store_v3s4( r.fwd, R_ScratchBase, O_(ResolveLookAtScratch,fwd)),
|
||||||
|
|
||||||
|
mac_yield()
|
||||||
|
})
|
||||||
|
|
||||||
|
typedef Struct_(Binds_ResolveLookAt_PopulateMT3S4S2) {
|
||||||
|
MT3_S2S4* look_at; /* MT3_S2S4* — destination matrix address */
|
||||||
|
};
|
||||||
|
typedef Struct_(RegUse_resolve_look_at_populate_mt3s4s2) {
|
||||||
|
Reg look_at;
|
||||||
|
Reg eye; /* matrix_vector phase: load -eye */
|
||||||
|
Reg_(V3_S4) row; /* populate phase: load ux/uy/uz */
|
||||||
|
union { Reg r0, ux, vx; }; /* populate addr → matrix_vector v_x */
|
||||||
|
union { Reg r1, uy, vy; }; /* populate uy → matrix_vector v_y */
|
||||||
|
union { Reg r2, uz, vz; }; /* populate uz → matrix_vector v_z */
|
||||||
|
};
|
||||||
|
/* write look_at->m[][] from ux/uy/uz as packed S2 (populate),
|
||||||
|
* ctc2 RT chain into C2[0..4] (matrix_vector), MVMVA RT*(-eye)>>12, store off
|
||||||
|
* directly to look_at->t[] (trans_matrix).
|
||||||
|
*
|
||||||
|
* C11 ApplyMatrixLV semantics (gte.atom.c ac_apply_matrix_lv; libgte reference):
|
||||||
|
* 1. ctc2 RT matrix (5 ctc2s to C2[0..4])
|
||||||
|
* 2. lw -eye from memory
|
||||||
|
* 3. S15 decomposition (eliminated here — the fused body takes the >>12 path
|
||||||
|
* directly via mtc2 IR + MVMVA pass2, matching the libgte canonical output)
|
||||||
|
* 4. mtc2 to IR1/2/3, nop2, MVMVA pass2 (sf=1, mx=0, v=3, cv=3)
|
||||||
|
* 5. mfc2 MACs → off
|
||||||
|
* 6. store off to look_at->t[] (skip scratch.eye intermediate)
|
||||||
|
*/
|
||||||
|
internal MipsAtom* AtomBundleEntry_(resolve_look_at,populate_mt3s4s2)(AtomArena_R aa, RegUse_resolve_look_at_populate_mt3s4s2 r)
|
||||||
|
atom_info(atom_bind(Binds_ResolveLookAt_PopulateMT3S4S2)) MipsAtom_Proc_(aa, {
|
||||||
|
/* --- Tape pop: look_at pointer --- */
|
||||||
|
load_word(r.look_at, R_TapePtr, O_(Binds_ResolveLookAt_PopulateMT3S4S2,look_at)),
|
||||||
|
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_ResolveLookAt_PopulateMT3S4S2)),
|
||||||
|
|
||||||
|
add_si(r.ux, R_ScratchBase, O_(ResolveLookAtScratch, ux)), LdSlot_
|
||||||
|
add_si(r.uy, R_ScratchBase, O_(ResolveLookAtScratch, uy)),
|
||||||
|
add_si(r.uz, R_ScratchBase, O_(ResolveLookAtScratch, uz)),
|
||||||
|
add_si(r.eye, R_ScratchBase, O_(ResolveLookAtScratch, eye)),
|
||||||
|
|
||||||
|
/* write look_at->m[][] from ux/uy/uz as packed S2 */
|
||||||
|
mac_load_v3s4(r.row, r.ux, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4, m[0])),
|
||||||
|
mac_load_v3s4(r.row, r.uy, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4, m[1])),
|
||||||
|
mac_load_v3s4(r.row, r.uz, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4, m[2])),
|
||||||
|
|
||||||
|
/* ctc2 RT chain + MVMVA RT * (-eye) >> 12 */
|
||||||
|
/* C2[0] = (RT12<<16)|RT11 ← ctc2 RT11 from m[0][0..1]
|
||||||
|
* C2[1] = (RT21<<16)|RT13 ← ctc2 RT12 from m[0][2..3]
|
||||||
|
* C2[2] = (RT23<<16)|RT22 ← ctc2 RT13 from m[1][1..2]
|
||||||
|
* C2[3] = (RT32<<16)|RT31 ← ctc2 RT21 from m[2][0..1]
|
||||||
|
* C2[4] = (RT33<<16)|junk ← ctc2 RT22 from m[2][2] (half) */
|
||||||
|
load_word( r.vx, r.look_at, O_(MT3_S2S4, m[0][0])), /* RT11|RT12 */ LdSlot_
|
||||||
|
load_word( r.vy, r.look_at, O_(MT3_S2S4, m[0][2])), /* RT13|RT21 */ LdSlot_ gte_mv_to_ctrl_r(r.vx, gte_cr_RT11),
|
||||||
|
load_word( r.vz, r.look_at, O_(MT3_S2S4, m[1][1])), /* RT22|RT23 */ LdSlot_ gte_mv_to_ctrl_r(r.vy, gte_cr_RT12),
|
||||||
|
load_word( r.vx, r.look_at, O_(MT3_S2S4, m[2][0])), /* RT31|RT32 */ LdSlot_ gte_mv_to_ctrl_r(r.vz, gte_cr_RT13),
|
||||||
|
load_half_u(r.vy, r.look_at, O_(MT3_S2S4, m[2][2])), /* RT33 */ LdSlot_ gte_mv_to_ctrl_r(r.vx, gte_cr_RT21),
|
||||||
|
|
||||||
|
GteDelay_ mac_load_word_v3(r.vx, r.vy, r.vz, r.eye, 0), LdSlot_
|
||||||
|
mac_sub_s_v3(r.vx, r.vy, r.vz, R_0, R_0, R_0, r.vx, r.vy, r.vz),
|
||||||
|
|
||||||
|
gte_mv_to_data_r(r.vx, C2_IR1),
|
||||||
|
gte_mv_to_data_r(r.vy, C2_IR2),
|
||||||
|
gte_mv_to_data_r(r.vz, C2_IR3),
|
||||||
|
GteDelay_ nop2,
|
||||||
|
|
||||||
|
/* MVMVA pass 2 — C11 ApplyMatrixLV command. sf=1, mx=0 (RT), v=3 (IR), cv=3. Reads RT × IR >> 12. */
|
||||||
|
gte_cmdw_mvmva_c11_pass2, GteDelay_ load_word(R_AtomJmp, R_TapePtr, 0), // ac_yield: word 1
|
||||||
|
mac_gte_mv_from_data_r_mac123(r.vx, r.vy, r.vz), GteDelay_ add_ui_self( R_TapePtr, S_(MipsCode)), // ac_yield: word 2
|
||||||
|
|
||||||
|
/* store off directly to look_at->t[] (skip scratch.eye intermediate) */
|
||||||
|
mac_store_word_v3(r.vx, r.vy, r.vz, r.look_at, O_(MT3_S2S4, t)),
|
||||||
|
|
||||||
|
jump_reg(R_AtomJmp), BdSlot_ nop, // ac_yield: word 3-4
|
||||||
|
})
|
||||||
|
#pragma endregion resolve_look_at
|
||||||
|
|
||||||
|
#pragma endregion Atom Procs
|
||||||
|
|
||||||
|
#pragma region Baked Atoms
|
||||||
|
|
||||||
|
enum {
|
||||||
|
R_ScreenX = R_T5 atom_reg atom_type(U2),
|
||||||
|
R_ScreenY = R_T6 atom_reg atom_type(U2),
|
||||||
|
R_ScreenBuf = R_T7 atom_reg, /* Caller-pinned: & smem.screen_buf */
|
||||||
|
#define R_ScreenBuf_Code R_T7_Code
|
||||||
|
};
|
||||||
|
//screen_env_init. Mirrors the libpsyx's SetDefDispEnv + SetDefDrawEnv + the manual enable_auto_clear / initial_bg_color writes.
|
||||||
|
internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
|
||||||
|
, atom_reads(R_T0, R_ScreenX, R_ScreenY, R_ScreenBuf)
|
||||||
|
, atom_writes(R_T0, R_ScreenX, R_ScreenY)
|
||||||
|
) {
|
||||||
|
/* display[0] = (0, 0, 320, 240); rest of struct zeroed. */
|
||||||
|
add_ui(R_ScreenX, R_0, ScreenRes_X), add_ui(R_ScreenY, R_0, ScreenRes_Y),
|
||||||
|
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area.width) + O_(DoubleBuffer,display[0])),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,display_area) + O_(DoubleBuffer,display[0])),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + O_(DoubleBuffer,display[0])),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + O_(DoubleBuffer,display[0])),
|
||||||
|
|
||||||
|
/* display[1] = (0, 240, 320, 240); rest of struct zeroed. */
|
||||||
|
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area) + O_(DoubleBuffer,display[1])),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + O_(DoubleBuffer,display[1])),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + O_(DoubleBuffer,display[1])),
|
||||||
|
|
||||||
|
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + O_(DoubleBuffer,draw[0])), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
|
||||||
|
mac_store_v2s2( R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + O_(DoubleBuffer,draw[0])), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
|
||||||
|
|
||||||
|
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + O_(DoubleBuffer,draw[1])),
|
||||||
|
|
||||||
|
/* draw[0].texture_window = (0, 0, 0, 0); two word-zeroes cover the full 8-byte tw field. */
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + O_(DoubleBuffer,draw[0])),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + O_(DoubleBuffer,draw[0])),
|
||||||
|
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,drawing_offset[0].x) + O_(DoubleBuffer,draw[1])),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + O_(DoubleBuffer,draw[1])),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + O_(DoubleBuffer,draw[1])),
|
||||||
|
|
||||||
|
/* draw[0].texture_page = 10 (gp0_tpage_default). C11 SetDefDrawEnv at C11_only.elf:0x8001273C writes the same 0x0A. . */
|
||||||
|
add_ui(R_T0, R_0, gp0_tpage_default),
|
||||||
|
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + O_(DoubleBuffer,draw[0])),
|
||||||
|
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + O_(DoubleBuffer,draw[1])),
|
||||||
|
|
||||||
|
/* draw[0] control bytes: flag_dither=1, flag_draw_on_display=1 (the dfe bit per psx-spx; libpsyx sets it via `SetDefDrawEnv`'s conditional at C11_only.elf:0x80012728), enable_auto_clear=1. Each byte is named;
|
||||||
|
* the previous `store_word(R_0, ..., +20)` overwrote all four with zero. */
|
||||||
|
add_ui(R_T0, R_0, 1),
|
||||||
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + O_(DoubleBuffer,draw[0])),
|
||||||
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + O_(DoubleBuffer,draw[0])),
|
||||||
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + O_(DoubleBuffer,draw[0])),
|
||||||
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + O_(DoubleBuffer,draw[1])),
|
||||||
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + O_(DoubleBuffer,draw[1])),
|
||||||
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + O_(DoubleBuffer,draw[1])),
|
||||||
|
|
||||||
|
/* draw[0].initial_bg_color = (r=7, g=7, b=7). */
|
||||||
|
add_ui(R_T0, R_0, 7),
|
||||||
|
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + O_(DoubleBuffer,draw[0])),
|
||||||
|
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + O_(DoubleBuffer,draw[1])),
|
||||||
|
|
||||||
|
mac_yield(),
|
||||||
|
};
|
||||||
|
|
||||||
|
enum {
|
||||||
|
R_IO_BaseAddr = R_T4 atom_reg, /* Caller-pinned: IO_BASE_ADDR = 0x1F800000 */
|
||||||
|
R_GP1_Offset = R_T2 atom_reg, /* Caller-pinned: GPIO_PORT1_OFFSET = 0x10 */
|
||||||
|
atom_auto_reg(gp_screen_init, R_GpTmp), /* Auto-allocated scratch; resolved to a free pool GPR by the lua pass. C-preprocessor expands to R_GpTmp = R_GpTmp_Code with an atom_auto_reg trailing comment. */
|
||||||
|
#define R_IO_BaseAddr_Code R_T4_Code
|
||||||
|
#define R_GP1_Offset_Code R_T2_Code
|
||||||
|
};
|
||||||
|
internal MipsAtom_(gp_screen_init) atom_info(atom_phase(screen_init), atom_reads(R_IO_BaseAddr)) {
|
||||||
|
store_word(R_0, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(00h) Reset */
|
||||||
|
mac_gcmd_push(gp1_word_ResetCmdBuffer(), R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(01h) ClearFIFO; uses pinned R_ScreenX as the transfer reg. */
|
||||||
|
mac_gcmd_push(gp1_word_AcknowledgeIRQ(), R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(02h) AckIRQ; uses pinned R_ScreenX as the transfer reg. */
|
||||||
|
mac_gcmd_push(gp1_word_DisplayOn(), R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(03h) Display ON; uses pinned R_ScreenX as the transfer reg. */
|
||||||
|
mac_gcmd_push(gp1_word_dma_to_gpu(), R_GpTmp, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(04h) DMADirection=2 (CPU->GPU). libpsyx's per-frame PutDrawEnv/DrawOTag use DMA2; without this the DMA queue never drains. Uses auto-allocated R_GpTmp. */
|
||||||
|
mac_gcmd_push(gp1_word_StartDisplayArea(), R_GpTmp, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(05h) StartDisplayArea (X=0, Y=0); uses auto-allocated R_GpTmp. */
|
||||||
|
|
||||||
|
/* GP1: DisplayMode + Display Ranges. */
|
||||||
|
mac_gcmd_push(gp1_word_display_mode_320x240_15bit_ntsc, R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||||
|
mac_gcmd_push(gp1_word_horizontal_range_ntsc, R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||||
|
mac_gcmd_push(gp1_word_vertical_range_ntsc, R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||||
|
|
||||||
|
/* GTE: SetGeomOffset (OFX, OFY) — ScreenRes_CenterX, ScreenRes_CenterY. */
|
||||||
|
load_upper_i(R_ScreenX, ScreenRes_CenterX), gte_mv_to_ctrl_r(R_ScreenX, gte_cr_OFX_Code),
|
||||||
|
load_upper_i(R_ScreenX, ScreenRes_CenterY), gte_mv_to_ctrl_r(R_ScreenX, gte_cr_OFY_Code),
|
||||||
|
|
||||||
|
/* GTE: SetGeomScreen (H) — CR26 (per PSX-SPX / libpsyx), value is the raw projection-plane distance, NOT shifted. */
|
||||||
|
add_ui(R_ScreenX, R_0, ScreenZ), gte_mv_to_ctrl_r(R_ScreenX, gte_cr_H_Code),
|
||||||
|
|
||||||
|
/* GP1: DisplayEnable — bit 0 = 0 (Display ON). */
|
||||||
|
mac_gcmd_push(gp1_word_DisplayOn(), R_GpTmp, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* Uses auto-allocated R_GpTmp. */
|
||||||
|
mac_yield(),
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef Struct_(Binds_PadApplyInput) {
|
||||||
|
PadState* state;
|
||||||
|
V3_S2* cube_rot;
|
||||||
|
V3_S2* floor_rot;
|
||||||
|
};
|
||||||
|
enum {
|
||||||
|
R_PadStateT5 = R_T5 atom_reg,
|
||||||
|
R_CubeRot = R_T1 atom_reg,
|
||||||
|
R_FloorRot = R_T2 atom_reg,
|
||||||
|
};
|
||||||
|
internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyInput)
|
||||||
|
, atom_reads(R_T0, R_CubeRot, R_FloorRot, R_T3, R_T4, R_PadStateT5, R_TapePtr)
|
||||||
|
, atom_writes( R_CubeRot, R_FloorRot)
|
||||||
|
) {
|
||||||
|
/* Pop Binds from tape (state, cube_rot, floor_rot) */
|
||||||
|
load_word(R_PadStateT5, R_TapePtr, O_(Binds_PadApplyInput,state)),
|
||||||
|
load_word(R_CubeRot, R_TapePtr, O_(Binds_PadApplyInput,cube_rot)),
|
||||||
|
load_word(R_FloorRot, R_TapePtr, O_(Binds_PadApplyInput,floor_rot)),
|
||||||
|
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_PadApplyInput)),
|
||||||
|
|
||||||
|
/* Load pad[0].buttons into R_T0. */
|
||||||
|
load_word(R_T0, R_PadStateT5, O_(PadState,buttons)), LdSlot_ nop,
|
||||||
|
// Note(Ed): Potential op with delay slot?
|
||||||
|
|
||||||
|
/* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */
|
||||||
|
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)), BdSlot_
|
||||||
|
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), LdSlot_
|
||||||
|
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
add_si( R_T4, R_T4, 30),
|
||||||
|
add_si( R_T3, R_T3, 5),
|
||||||
|
store_half(R_T4, R_CubeRot, O_(V3_S2,y)),
|
||||||
|
store_half(R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
atom_label(exit_dpad_left)
|
||||||
|
|
||||||
|
/* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */
|
||||||
|
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)), BdSlot_
|
||||||
|
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), LdSlot_
|
||||||
|
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
add_si( R_T4, R_T4, -30),
|
||||||
|
add_si( R_T3, R_T3, -5),
|
||||||
|
store_half(R_T4, R_CubeRot, O_(V3_S2,y)),
|
||||||
|
store_half(R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
atom_label(exit_dpad_right)
|
||||||
|
|
||||||
|
/* Analog left-stick X: dead zone 0x70..0x90.
|
||||||
|
* Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */
|
||||||
|
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), LdSlot_ //?
|
||||||
|
|
||||||
|
/* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly).
|
||||||
|
* set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */
|
||||||
|
add_ui(R_T4, R_0, PadDeadZone_HighBound), set_lt_u(R_T4, R_T3, R_T4), branch_ne(R_T4, R_0, atom_offset(dead_zone_low_check, dead_low_active)),
|
||||||
|
add_ui(R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_low_active */
|
||||||
|
|
||||||
|
atom_label(dead_check_upper)
|
||||||
|
/* left_x >= 0x70 → check upper bound. */
|
||||||
|
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), /* reload */ LdSlot_ //?
|
||||||
|
add_ui( R_T4, R_0, PadDeadZone_HighBound),
|
||||||
|
|
||||||
|
/* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */
|
||||||
|
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)), BdSlot_
|
||||||
|
add_ui( R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_high_active */
|
||||||
|
jump_rel(atom_offset(dead_zone_skip, exit_stick)),
|
||||||
|
BdSlot_ mac_yield_load(), LdSlot_
|
||||||
|
|
||||||
|
atom_label(dead_low_active)
|
||||||
|
/* R_T3 = left_x (from line 632 lbu; not clobbered between dead_zone_low_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
||||||
|
* The earlier `load_byte_u(R_T3, ...)` reload was redundant and introduced a load-use hazard on the next `sub_u`.
|
||||||
|
* R_T4 = 0x80 from the BD-slot of `dead_zone_low_check`'s branch_ne. */
|
||||||
|
sub_u( R_T3, R_T4, R_T3), /* R_T3 = 0x80 - left_x */
|
||||||
|
/* delta = 0x80 - left_x (positive). */
|
||||||
|
|
||||||
|
/* R_T4 = cube_delta */
|
||||||
|
shift_aright(R_T4, R_T3, 2),
|
||||||
|
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), LdSlot_ nop,
|
||||||
|
add_u( R_T0, R_T0, R_T4),
|
||||||
|
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||||
|
/* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap;
|
||||||
|
* doesn't read R_T0; R_T4 settles by the subsequent add_u). */
|
||||||
|
load_half( R_T0, R_FloorRot, O_(V3_S2,y)), LdSlot_
|
||||||
|
shift_aright(R_T4, R_T3, 5),
|
||||||
|
add_u( R_T0, R_T0, R_T4),
|
||||||
|
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
|
||||||
|
jump_rel(atom_offset(end_low, exit_stick)),
|
||||||
|
BdSlot_ mac_yield_load(), LdSlot_
|
||||||
|
|
||||||
|
atom_label(dead_high_active)
|
||||||
|
/* R_T3 = left_x (from line 641 lbu in dead_check_upper; not clobbered between dead_zone_high_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
||||||
|
* The earlier `load_byte_u(R_T3, ...)` reload was redundant and introduced a load-use hazard on the next `sub_u`.
|
||||||
|
* R_T4 = 0x80 from the BD-slot of `dead_zone_high_check`'s branch_ne. */
|
||||||
|
sub_u( R_T3, R_T4, R_T3),
|
||||||
|
/* delta = 0x80 - left_x (signed negative). */
|
||||||
|
|
||||||
|
shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */
|
||||||
|
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), LdSlot_ nop,
|
||||||
|
add_u( R_T0, R_T0, R_T4),
|
||||||
|
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||||
|
|
||||||
|
/* R_T4 = floor_delta (signed) — moved into the load-delay slot of the floor load below. */
|
||||||
|
load_half( R_T0, R_FloorRot, O_(V3_S2,y)), LdSlot_
|
||||||
|
shift_aright(R_T4, R_T3, 5),
|
||||||
|
add_u( R_T0, R_T0, R_T4),
|
||||||
|
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
|
||||||
|
atom_label(no_jump_fallthrough)
|
||||||
|
mac_yield_load(), LdSlot_
|
||||||
|
|
||||||
|
atom_label(exit_stick)
|
||||||
|
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the dead-zone/exit branch. */
|
||||||
|
mac_yield_tail(),
|
||||||
|
};
|
||||||
|
|
||||||
|
enum {
|
||||||
|
R_Cam = R_T4 atom_reg,
|
||||||
|
R_CamPadState = R_T5 atom_reg,
|
||||||
|
};
|
||||||
|
typedef Struct_(Binds_PadInputCam) {
|
||||||
|
PadState* state;
|
||||||
|
Camera* cam;
|
||||||
|
};
|
||||||
|
internal MipsAtom_(pad_input_cam) atom_info(atom_bind(Binds_PadInputCam)
|
||||||
|
, atom_reads( R_Cam, R_CamPadState, R_TapePtr)
|
||||||
|
, atom_writes(R_Cam)
|
||||||
|
) {
|
||||||
|
/* Bind pop: state → R_CamPadState (R_T5), cam → R_Cam (R_T4), advance R_TapePtr by 8. */
|
||||||
|
load_word(R_CamPadState, R_TapePtr, O_(Binds_PadInputCam,state)),
|
||||||
|
load_word(R_Cam, R_TapePtr, O_(Binds_PadInputCam,cam)),
|
||||||
|
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_PadInputCam)),
|
||||||
|
|
||||||
|
/* Load pad[0].buttons into R_T0; nop fills the load-delay slot. */
|
||||||
|
load_word(R_T0, R_CamPadState, O_(PadState,buttons)), LdSlot_
|
||||||
|
load_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
||||||
|
|
||||||
|
// D-pad Left → cam.pos.x -= 50. and_i fulfills BD-slot for load on R_Cam.
|
||||||
|
LdSlot_ and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(left_x, exit_left_x)), BdSlot_ nop,
|
||||||
|
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
||||||
|
atom_label(exit_left_x)
|
||||||
|
|
||||||
|
/* D-pad Right → cam.pos.x += 50. Reuses R_T1 from Left. */
|
||||||
|
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(right_x, exit_right_x)), BdSlot_ nop,
|
||||||
|
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
||||||
|
atom_label(exit_right_x)
|
||||||
|
|
||||||
|
/* D-pad Up → cam.pos.y -= 50. Load pos.y BEFORE the andi. */
|
||||||
|
load_word(R_T1, R_Cam, O_(Camera,pos.y)), LdSlot_
|
||||||
|
and_i(R_T3, R_T0, Pad_Up), branch_le_zero(R_T3, atom_offset(up_y, exit_up_y)), BdSlot_ nop,
|
||||||
|
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
|
||||||
|
atom_label(exit_up_y)
|
||||||
|
|
||||||
|
/* D-pad Down → cam.pos.y += 50. Reuses R_T1 from Up. */
|
||||||
|
and_i(R_T3, R_T0, Pad_Down), branch_le_zero(R_T3, atom_offset(down_y, exit_down_y)), BdSlot_ nop,
|
||||||
|
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
|
||||||
|
atom_label(exit_down_y)
|
||||||
|
|
||||||
|
/* D-pad Cross → cam.pos.z -= 50. Load pos.z BEFORE the andi. */
|
||||||
|
load_word(R_T1, R_Cam, O_(Camera,pos.z)), LdSlot_
|
||||||
|
and_i(R_T3, R_T0, Pad_Cross), branch_le_zero(R_T3, atom_offset(cross_z, exit_cross_z)), BdSlot_ load_word(R_AtomJmp, R_TapePtr, 0), LdSlot_ // ac_yield: word 1
|
||||||
|
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
|
||||||
|
atom_label(exit_cross_z)
|
||||||
|
|
||||||
|
/* D-pad Circle → cam.pos.z += 50. Reuses R_T1 from Cross. */
|
||||||
|
and_i(R_T3, R_T0, Pad_Circle), branch_le_zero(R_T3, atom_offset(circle_z, exit_circle_z)), BdSlot_ add_ui_self(R_TapePtr, S_(MipsCode)), // ac_yield: word 2
|
||||||
|
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
|
||||||
|
atom_label(exit_circle_z)
|
||||||
|
|
||||||
|
jump_reg(R_AtomJmp), BdSlot_ nop // ac_yield: word 3-4
|
||||||
|
};
|
||||||
|
|
||||||
|
enum {
|
||||||
|
R_PrimCursor = R_T7 atom_reg atom_type(U4*), /* Output cursor (primitive buffer) */
|
||||||
|
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2*), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
|
||||||
|
R_VertBase = R_T5 atom_reg atom_type(V3_S2*), /* Base address of the vertex array */
|
||||||
|
R_OtBase = R_T6 atom_reg atom_type(U4*), /* Base address of the Ordering Table */
|
||||||
|
#define R_PrimCursor_Code R_T7_Code
|
||||||
|
#define R_FaceCursor_Code R_T4_Code
|
||||||
|
#define R_VertBase_Code R_T5_Code
|
||||||
|
#define R_OtBase_Code R_T6_Code
|
||||||
|
};
|
||||||
|
typedef Struct_(Binds_CubeTri) {
|
||||||
|
U4 PrimCursor;
|
||||||
|
V4_S2* FaceCursor;
|
||||||
|
V3_S2* VertBase;
|
||||||
|
U4* OtBase;
|
||||||
|
};
|
||||||
|
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)
|
||||||
|
, atom_reads(R_TapePtr)
|
||||||
|
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
|
||||||
|
){
|
||||||
|
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||||
|
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
|
||||||
|
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
|
||||||
|
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
|
||||||
|
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
|
||||||
|
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
|
||||||
|
internal
|
||||||
|
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||||
|
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
||||||
|
atom_writes(R_PrimCursor, R_FaceCursor)
|
||||||
|
){
|
||||||
|
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||||
|
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
||||||
|
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
||||||
|
// load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
||||||
|
|
||||||
|
LdSlot_ mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
||||||
|
GteDelay_ load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)), LdSlot_
|
||||||
|
GteDelay_ load_word(R_AtomJmp, R_TapePtr, 0), LdSlot_ //ac_yield: word 2,
|
||||||
|
gte_cmdw_rotate_translate_perspective_triple,
|
||||||
|
gte_cmdw_nclip,
|
||||||
|
|
||||||
|
gte_mv_from_data_r(R_T0, C2_MAC0), GteDelay_ add_ui_self(R_TapePtr, S_(MipsCode)), // ac_yield: word 1
|
||||||
|
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
|
||||||
|
/* BD-slot: Write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
|
||||||
|
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
|
||||||
|
* harmless because the OT entry that points to this prim is created later. */
|
||||||
|
BdSlot_ store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
||||||
|
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
||||||
|
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)), LdSlot_
|
||||||
|
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||||
|
|
||||||
|
mac_gte_store_g4_p012(R_PrimCursor),
|
||||||
|
gte_cmdw_rotate_translate_perspective_single,
|
||||||
|
mac_gte_store_g4_p3(R_PrimCursor),
|
||||||
|
|
||||||
|
gte_cmdw_avg_sort_z4,
|
||||||
|
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||||
|
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||||
|
set_lt_u( R_AT, R_T1, R_AT),
|
||||||
|
|
||||||
|
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), BdSlot_ nop,
|
||||||
|
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_G4)),
|
||||||
|
mac_format_g4_color(R_PrimCursor,
|
||||||
|
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
||||||
|
/* c1 yellow */ 0xFF, 0xFF, 0x00,
|
||||||
|
/* c2 cyan */ 0x00, 0xFF, 0xFF,
|
||||||
|
/* c3 green */ 0x00, 0xFF, 0x00),
|
||||||
|
// end: branch(bounds_chk)
|
||||||
|
// end: branch(cull)
|
||||||
|
|
||||||
|
atom_label(cube_g4_face_exit)
|
||||||
|
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
|
||||||
|
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
|
||||||
|
jump_reg(R_AtomJmp), BdSlot_ nop // ac_yield: word 3-4
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef Struct_(Binds_FloorTri) {
|
||||||
|
U4 PrimCursor;
|
||||||
|
V3_S2* FaceCursor;
|
||||||
|
V3_S2* VertBase;
|
||||||
|
U4* OtBase;
|
||||||
|
};
|
||||||
|
internal
|
||||||
|
MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(floor_f3)
|
||||||
|
, atom_reads(R_TapePtr)
|
||||||
|
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
|
||||||
|
){
|
||||||
|
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||||
|
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
|
||||||
|
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
|
||||||
|
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
|
||||||
|
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
|
||||||
|
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
// atom_dbg_skip
|
||||||
|
internal
|
||||||
|
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
||||||
|
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||||
|
, atom_writes(R_PrimCursor, R_FaceCursor)
|
||||||
|
) {
|
||||||
|
mac_load_tri_indices(R_FaceCursor, R_T0, R_T1, R_T2),
|
||||||
|
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2), GteDelay_ nop2,
|
||||||
|
gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
|
||||||
|
gte_cmdw_nclip,
|
||||||
|
|
||||||
|
/* Culling (Branch forward if Backface) */
|
||||||
|
gte_mv_from_data_r(R_T0, C2_MAC0), GteDelay_ load_word(R_AtomJmp, R_TapePtr, 0), // ac_yield: word 1
|
||||||
|
branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), BdSlot_ add_ui_self(R_TapePtr, S_(MipsCode)), // ac_yield: word 2
|
||||||
|
/* Format Primitive */
|
||||||
|
mac_gte_store_f3(R_PrimCursor),
|
||||||
|
|
||||||
|
/* Calculate Depth */
|
||||||
|
gte_avg_sort_z3,
|
||||||
|
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||||
|
/* Bounds Check OTZ < 2048 (Branch forward to skip insertion) */
|
||||||
|
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||||
|
set_lt_u( R_AT, R_T1, R_AT),
|
||||||
|
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), BdSlot_ nop,
|
||||||
|
mac_format_f3_color(R_PrimCursor, 0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
|
||||||
|
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_F3)), /* Insert into Ordering Table Linked List */
|
||||||
|
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
|
||||||
|
// Note(Ed): No bounds checking, should be checked before atom runs.
|
||||||
|
// end: branch(bounds_chk)
|
||||||
|
// end: branch(culling)
|
||||||
|
|
||||||
|
/* Advance Input Cursor & Yield (Both branch targets land here) */
|
||||||
|
atom_label(floor_f3_face_exit)
|
||||||
|
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
|
||||||
|
jump_reg(R_AtomJmp), BdSlot_ nop // ac_yield: word 3-4
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
|
||||||
|
internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitiveArena)
|
||||||
|
, atom_reads( R_TapePtr, R_PrimCursor)
|
||||||
|
, atom_writes(R_TapePtr)
|
||||||
|
){
|
||||||
|
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
|
||||||
|
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)), LdSlot_
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
|
||||||
|
/* Calculate byte offset and store directly back to RAM */
|
||||||
|
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
|
||||||
|
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
#pragma endregion Baked Atoms
|
||||||
@@ -0,0 +1,436 @@
|
|||||||
|
#pragma region Vendors
|
||||||
|
#include <stdio.h>
|
||||||
|
#include <stdlib.h>
|
||||||
|
// #include <assert.h>
|
||||||
|
// #include "libgpu.h"
|
||||||
|
// #include "libetc.h"
|
||||||
|
// #include "libgte.h"
|
||||||
|
#pragma endregion Vendors
|
||||||
|
|
||||||
|
#pragma region Duffle Headers
|
||||||
|
# include "duffle/gen/macs.h"
|
||||||
|
# include "duffle/gen/offsets.h"
|
||||||
|
|
||||||
|
#include "duffle/word_count.metadata.h"
|
||||||
|
|
||||||
|
#include "duffle/dsl.h"
|
||||||
|
#include "duffle/memory.h"
|
||||||
|
#include "duffle/math.h"
|
||||||
|
|
||||||
|
#include "duffle/gcc_asm.h"
|
||||||
|
#include "duffle/mips.h"
|
||||||
|
#include "duffle/gp.h"
|
||||||
|
#include "duffle/gte.h"
|
||||||
|
#include "duffle/pad.h"
|
||||||
|
|
||||||
|
#include "duffle/dsl.atom.h"
|
||||||
|
#include "duffle/lottes_tape.h"
|
||||||
|
|
||||||
|
#include "duffle/bios.h"
|
||||||
|
#include "duffle/psyq.h"
|
||||||
|
#pragma endregion Duffle Headers
|
||||||
|
|
||||||
|
#pragma region Duffle TUs
|
||||||
|
#include "duffle/pad.c"
|
||||||
|
#include "duffle/math.atom.h"
|
||||||
|
#include "duffle/mips.atom.c"
|
||||||
|
#include "duffle/gte.atom.c"
|
||||||
|
#include "duffle/gp.atom.c"
|
||||||
|
#include "duffle/pad.atom.c"
|
||||||
|
#include "duffle/psyq.atom.c"
|
||||||
|
#pragma endregion Duffle TUs
|
||||||
|
|
||||||
|
#pragma region Hello Camera Headers
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
# include "gen/auto_reg.h"
|
||||||
|
|
||||||
|
#include "hello_camera.h"
|
||||||
|
#pragma endregion Hello Camera Headers
|
||||||
|
|
||||||
|
#pragma region Hello Joypad TUs
|
||||||
|
#include "hello_camera.atom.c"
|
||||||
|
#pragma endregion Hello Joypad TUs
|
||||||
|
|
||||||
|
enum {
|
||||||
|
MemTape_Len = 512,
|
||||||
|
|
||||||
|
ResolveLookAtArena_Words = 1024,
|
||||||
|
ResolveLookAtArena_Size = ResolveLookAtArena_Words * S_(MipsCode),
|
||||||
|
|
||||||
|
CT_InitAtomMem_Words = Kilo_(4),
|
||||||
|
CT_InitAtomMem_Size = CT_InitAtomMem_Words * S_(MipsCode),
|
||||||
|
};
|
||||||
|
typedef Struct_(SMemory) {
|
||||||
|
PrimitiveArena primitives;
|
||||||
|
A2_OrderingTable_Buffer ordering_tbl;
|
||||||
|
DoubleBuffer screen_buf;
|
||||||
|
S4 active_buf_id;
|
||||||
|
|
||||||
|
U4 MemTape[MemTape_Len];
|
||||||
|
|
||||||
|
MT3_S2S4 tform_world;
|
||||||
|
MT3_S2S4 tform_view;
|
||||||
|
|
||||||
|
Camera cam;
|
||||||
|
|
||||||
|
Ent_Cube cube;
|
||||||
|
Ent_Floor floor;
|
||||||
|
|
||||||
|
PadBiosRaw pad_raw[2];
|
||||||
|
PadState pad[2];
|
||||||
|
|
||||||
|
U1 ct_init_atom_mem[CT_InitAtomMem_Size];
|
||||||
|
MipsAtom* normalize_v3s4;
|
||||||
|
MipsAtom* gte_cross_v3s4;
|
||||||
|
|
||||||
|
U1 resolve_look_at_mem[ResolveLookAtArena_Size];
|
||||||
|
MipsAtom* resolve_look_at_bundle[AtomBundle_Len(resolve_look_at)];
|
||||||
|
};
|
||||||
|
global SMemory smem;
|
||||||
|
extern SMemory smem;
|
||||||
|
|
||||||
|
#define pad0_btn_(btn) btn & smem.pad[0].buttons
|
||||||
|
#define pad1_btn_(btn) btn & smem.pad[1].buttons
|
||||||
|
|
||||||
|
I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
|
||||||
|
gknown PrimitiveArena* pa = & smem.primitives;
|
||||||
|
gknown B1* buf = (B1*) r_(smem.primitives.buf)[smem.active_buf_id];
|
||||||
|
assert(pa->used + type_width < PrimitiveBuff_Len);
|
||||||
|
B1* next = buf + pa->used;
|
||||||
|
pa->used += type_width;
|
||||||
|
return next;
|
||||||
|
}
|
||||||
|
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
|
||||||
|
|
||||||
|
I_ void resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4* up_in) {
|
||||||
|
// RGA(Lengyel): Build matrix expansion of a rigid transformation. Corresponding motor is not constructed; we write the LA form for GTE.
|
||||||
|
// Preconditions: eye != target, up_in not collinear with (target - eye).
|
||||||
|
V3_S4 right, up, forward;
|
||||||
|
V3_S4 ux, uy, uz;
|
||||||
|
V3_S4 pos, off;
|
||||||
|
|
||||||
|
forward = target[0]; sub_v3s4(& forward, eye[0]); // RGA(Lengyel): Affine point - point = zero-weight direction.
|
||||||
|
psy_normalize_v3s4(& forward, & uz); // RGA(Lengyel): Normalize the direction bulk. Not finite-point unitization.
|
||||||
|
|
||||||
|
cross_v3s4(& uz, up_in, & right); psy_normalize_v3s4(& right, & ux); // RGA(Lengyel): Complement(Wedge(forward, up_in)) -> right axis.
|
||||||
|
cross_v3s4(& uz, & ux, & up); psy_normalize_v3s4(& up, & uy); // RGA(Lengyel): Complement(Wedge(forward, right)) -> up axis.
|
||||||
|
|
||||||
|
// RGA(Lengyel): matrix expansion of the world-to-camera rotation (basis rows).
|
||||||
|
look_at->m[0][0] = ux.x; look_at->m[0][1] = ux.y; look_at->m[0][2] = ux.z;
|
||||||
|
look_at->m[1][0] = uy.x; look_at->m[1][1] = uy.y; look_at->m[1][2] = uy.z;
|
||||||
|
look_at->m[2][0] = uz.x; look_at->m[2][1] = uz.y; look_at->m[2][2] = uz.z;
|
||||||
|
|
||||||
|
pos = eye[0]; mul_v3s4(& pos, v3s4(-1,-1,-1)); // RGA(Lengyel): -eye in world coordinates (spatial bulk only; implicit weight is dropped).
|
||||||
|
|
||||||
|
// RGA(Lengyel): R * (-eye) is the full matrix translation column.
|
||||||
|
// Motor translator would store half this displacement in m.xyz; GTE consumes full column.
|
||||||
|
mul_m3s2_v3s4(look_at, & pos, & off);
|
||||||
|
trans_m3s2( look_at, & off);
|
||||||
|
}
|
||||||
|
FI_ void camera_look_at_c11(Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at_c11(& c->look_at, & c->pos, target, up_in); }
|
||||||
|
|
||||||
|
internal void compile_init_atoms(void) {
|
||||||
|
AtomArena ab = atomarena_make(slice_ut_arr(smem.ct_init_atom_mem));
|
||||||
|
RegFile rf = regfile(regfile_abi_mask);
|
||||||
|
#define ralloc() regfile_alloc(& rf)
|
||||||
|
#define ralloc_v3() { ralloc(), ralloc(), ralloc() }
|
||||||
|
|
||||||
|
smem.gte_cross_v3s4 = gte_cross_v3s4(& ab,
|
||||||
|
RegUse_(gte_cross_v3s4) {
|
||||||
|
.a = ralloc_v3(),
|
||||||
|
.b = ralloc_v3(),
|
||||||
|
.x = ralloc(),
|
||||||
|
.y = ralloc(),
|
||||||
|
.z = ralloc(),
|
||||||
|
});
|
||||||
|
regfile_reset(& rf);
|
||||||
|
|
||||||
|
smem.normalize_v3s4 = normalize_v3s4(& ab,
|
||||||
|
RegUse_(normalize_v3s4) {
|
||||||
|
.res = ralloc_v3(),
|
||||||
|
.r0 = ralloc(),
|
||||||
|
.r1 = ralloc(),
|
||||||
|
.r2 = ralloc(),
|
||||||
|
.r3 = ralloc(),
|
||||||
|
.r4 = ralloc(),
|
||||||
|
.r5 = ralloc(),
|
||||||
|
});
|
||||||
|
regfile_reset(& rf);
|
||||||
|
|
||||||
|
assert(ab.used <= CT_InitAtomMem_Size);
|
||||||
|
#undef ralloc
|
||||||
|
#undef ralloc_v3
|
||||||
|
}
|
||||||
|
|
||||||
|
internal void compile_resolve_look_at(void) {
|
||||||
|
AtomBundle_resolve_look_at_R bundle = C_(void*, smem.resolve_look_at_bundle);
|
||||||
|
AtomArena ab = atomarena_make(slice_ut_arr(smem.resolve_look_at_mem));
|
||||||
|
RegFile rf = regfile(regfile_abi_mask);
|
||||||
|
#define ralloc() regfile_alloc(& rf)
|
||||||
|
#define ralloc_v3() { ralloc(), ralloc(), ralloc() }
|
||||||
|
bundle->input_and_sub = AtomBundleEntry_(resolve_look_at, input_and_sub)(& ab,
|
||||||
|
RegUse_(resolve_look_at_input_and_sub) {
|
||||||
|
.target_ptr = ralloc(),
|
||||||
|
.eye_ptr = ralloc(),
|
||||||
|
.up_in_ptr = ralloc(),
|
||||||
|
.up_in = ralloc_v3(),
|
||||||
|
.r012 = ralloc_v3(),
|
||||||
|
.r345 = {ralloc(), R_AT, ralloc() },
|
||||||
|
});
|
||||||
|
regfile_reset(& rf);
|
||||||
|
|
||||||
|
bundle->normalize_fwd_uz = smem.normalize_v3s4;
|
||||||
|
bundle->cross_to_right = smem.gte_cross_v3s4;
|
||||||
|
bundle->normalize_right_ux = smem.normalize_v3s4;
|
||||||
|
bundle->cross_to_up = smem.gte_cross_v3s4;
|
||||||
|
bundle->normalize_up_uy = smem.normalize_v3s4;
|
||||||
|
|
||||||
|
bundle->populate_mt3s4s2 = AtomBundleEntry_(resolve_look_at,populate_mt3s4s2)(& ab,
|
||||||
|
RegUse_(resolve_look_at_populate_mt3s4s2){
|
||||||
|
.look_at = ralloc(),
|
||||||
|
.eye = ralloc(),
|
||||||
|
.row = ralloc_v3(),
|
||||||
|
.r0 = ralloc(),
|
||||||
|
.r1 = ralloc(),
|
||||||
|
.r2 = ralloc(),
|
||||||
|
});
|
||||||
|
|
||||||
|
assert(ab.used <= ResolveLookAtArena_Size); // Sanity check: arena didn't overflow.
|
||||||
|
#undef ralloc
|
||||||
|
}
|
||||||
|
|
||||||
|
// Emit the resolve_look_at bundle into the tape. Called once per frame from update().
|
||||||
|
I_ void resolve_look_at(TapeBuilder_R tb, MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4* up_in) {
|
||||||
|
/* Typed view of the scratchpad for field-address arithmetic. */
|
||||||
|
ResolveLookAtScratch* sp = C_scratch(ResolveLookAtScratch*);
|
||||||
|
AtomBundle_resolve_look_at_R bundle = C_(void*, smem.resolve_look_at_bundle);
|
||||||
|
tb_emit(tb, bundle->input_and_sub); tb_bind_(tb, Binds_ResolveLookAtSub,
|
||||||
|
.target = target,
|
||||||
|
.eye = eye,
|
||||||
|
.up_in = up_in,
|
||||||
|
);
|
||||||
|
tb_emit(tb, bundle->normalize_fwd_uz); tb_bind_(tb, Binds_normalize_v3s4,
|
||||||
|
.src_offset = O_(ResolveLookAtScratch,fwd),
|
||||||
|
.dst_offset = O_(ResolveLookAtScratch,uz),
|
||||||
|
);
|
||||||
|
tb_emit(tb, bundle->cross_to_right); tb_bind_(tb, Binds_gte_cross_v3s4,
|
||||||
|
.src_a = & sp->uz,
|
||||||
|
.src_b = & sp->up_in,
|
||||||
|
.out = & sp->right,
|
||||||
|
);
|
||||||
|
tb_emit(tb, bundle->normalize_right_ux); tb_bind_(tb, Binds_normalize_v3s4,
|
||||||
|
.src_offset = O_(ResolveLookAtScratch,right),
|
||||||
|
.dst_offset = O_(ResolveLookAtScratch,ux),
|
||||||
|
);
|
||||||
|
tb_emit(tb, bundle->cross_to_up); tb_bind_(tb, Binds_gte_cross_v3s4,
|
||||||
|
.src_a = & sp->uz,
|
||||||
|
.src_b = & sp->ux,
|
||||||
|
.out = & sp->up,
|
||||||
|
);
|
||||||
|
tb_emit(tb, bundle->normalize_up_uy); tb_bind_(tb, Binds_normalize_v3s4,
|
||||||
|
.src_offset = O_(ResolveLookAtScratch,up),
|
||||||
|
.dst_offset = O_(ResolveLookAtScratch,uy),
|
||||||
|
);
|
||||||
|
tb_emit(tb, bundle->populate_mt3s4s2); tb_bind_(tb, Binds_ResolveLookAt_PopulateMT3S4S2,
|
||||||
|
.look_at = look_at,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
FI_ void camera_look_at(TapeBuilder_R tb, Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at(tb, & c->look_at, & c->pos, target, up_in); }
|
||||||
|
|
||||||
|
void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||||
|
{
|
||||||
|
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
|
||||||
|
|
||||||
|
// Pad Input
|
||||||
|
{
|
||||||
|
tb.used = 0; tb_scope_run(& tb) {
|
||||||
|
// Grab latest state from bios.
|
||||||
|
tb_emit_(pad_bios_snapshot);
|
||||||
|
tb_data(& tb, u4_(& smem.pad_raw[0]));
|
||||||
|
tb_data(& tb, u4_(& smem.pad[0]));
|
||||||
|
// tb_emit_(pad_bios_snapshot);
|
||||||
|
// tb_data_(raw, & smem.pad_raw[1]);
|
||||||
|
// tb_data_(state, & smem.pad[1]);
|
||||||
|
|
||||||
|
tb_emit_(pad_input_cam);
|
||||||
|
tb_data(& tb, u4_(& smem.pad[0]));
|
||||||
|
tb_data(& tb, u4_(& smem.cam));
|
||||||
|
|
||||||
|
// tb_emit_(pad_input_cube_rotation);
|
||||||
|
// tb_data_(state, & smem.pad[0]);
|
||||||
|
// tb_data_(cube_rot, & smem.cube.rot);
|
||||||
|
// tb_data_(floor_rot, & smem.floor.rot);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
orderingtbl_clear_reverse(ordering_buf, OrderingTbl_Len);
|
||||||
|
|
||||||
|
// Update the position based on acceleration and velocity
|
||||||
|
gknown V3_S4_R pos = & smem.cube.pos;
|
||||||
|
gknown V3_S4_R vel = & smem.cube.vel;
|
||||||
|
gknown V3_S4_R acc = & smem.cube.accel;
|
||||||
|
add_v3s4(vel, acc[0]);
|
||||||
|
add_v3s4_fp(pos, vel[0]);
|
||||||
|
|
||||||
|
if (pos->y + 150 > smem.floor.pos.y) vel->y *= -1;
|
||||||
|
|
||||||
|
// Prep
|
||||||
|
S4 nclip = 0;
|
||||||
|
S4 orderingtbl_z = 0;
|
||||||
|
A2_S2 p; //???
|
||||||
|
S4 flag; //????
|
||||||
|
|
||||||
|
B4 use_c11_path = false;
|
||||||
|
if (use_c11_path) {
|
||||||
|
camera_look_at_c11(& smem.cam, & smem.cube.pos, & v3s4(0, -fp_one, 0));
|
||||||
|
}
|
||||||
|
if (use_c11_path == false)
|
||||||
|
{
|
||||||
|
tb.used = 0; tb_scope_run(& tb) {
|
||||||
|
camera_look_at(& tb, & smem.cam, & smem.cube.pos, & v3s4(0, -fp_one, 0));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Draw cube
|
||||||
|
if (1)
|
||||||
|
{
|
||||||
|
mt3s2s4_rotation (& smem.cube.rot, & smem.tform_world);
|
||||||
|
mt3s2s4_translation(& smem.tform_world, & smem.cube.pos);
|
||||||
|
mt3s2s4_scale (& smem.tform_world, & smem.cube.scale);
|
||||||
|
|
||||||
|
// Combine world and look_at matrix.
|
||||||
|
gte_comp_coord_m3s2(& smem.cam.look_at, & smem.tform_world, & smem.tform_view);
|
||||||
|
gte_matrix_set_rotation (& smem.tform_view);
|
||||||
|
gte_matrix_set_translation(& smem.tform_view);
|
||||||
|
|
||||||
|
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||||
|
U4 prim_cursor = prim_base + pa->used;
|
||||||
|
|
||||||
|
tb.used = 0; tb_scope(& tb) {
|
||||||
|
tb_emit(& tb, rbind_cube_g4_face);
|
||||||
|
tb_data(& tb, prim_cursor);
|
||||||
|
tb_data(& tb, u4_(smem.cube.faces));
|
||||||
|
tb_data(& tb, u4_(smem.cube.verts));
|
||||||
|
tb_data(& tb, u4_(ordering_buf));
|
||||||
|
|
||||||
|
for (U4 i = 0; i < Cube_num_faces; i++) {
|
||||||
|
// Two triangles per quad face: (x,y,z) and (x,z,w)
|
||||||
|
tb_emit(& tb, cube_g4_face);
|
||||||
|
}
|
||||||
|
|
||||||
|
tb_emit(& tb, sync_primitive_arena);
|
||||||
|
tb_data(& tb, u4_(& pa->used));
|
||||||
|
tb_data(& tb, prim_base);
|
||||||
|
}
|
||||||
|
tape_run(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
|
||||||
|
|
||||||
|
// smem.cube.rot.y += 30;
|
||||||
|
}
|
||||||
|
// Draw floor
|
||||||
|
if (1)
|
||||||
|
{
|
||||||
|
mt3s2s4_rotation (& smem.floor.rot, & smem.tform_world);
|
||||||
|
mt3s2s4_translation(& smem.tform_world, & smem.floor.pos);
|
||||||
|
mt3s2s4_scale (& smem.tform_world, & smem.floor.scale);
|
||||||
|
|
||||||
|
// Combine world and look_at matrix.
|
||||||
|
gte_comp_coord_m3s2(& smem.cam.look_at, & smem.tform_world, & smem.tform_view);
|
||||||
|
|
||||||
|
gte_matrix_set_rotation (& smem.tform_view);
|
||||||
|
gte_matrix_set_translation(& smem.tform_view);
|
||||||
|
|
||||||
|
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||||
|
U4 prim_cursor = prim_base + pa->used;
|
||||||
|
|
||||||
|
// TODO(Ed): We should do a bounds check beforehand to confirm pa can hold all tris?
|
||||||
|
// The tape atoms in-flight should not need to care.
|
||||||
|
|
||||||
|
// Prepare the tape. (Push protocol to tape)
|
||||||
|
tb.used = 0; tb_scope(& tb) {
|
||||||
|
// tb_emit(& tb, set_gte_mt3s2s4);
|
||||||
|
// tb_data(& tb, u4_(& smem.tform_view));
|
||||||
|
|
||||||
|
tb_emit(& tb, rbind_floor_f3_face);
|
||||||
|
// TODO(Ed): Just use a single context struct ref?
|
||||||
|
tb_data(& tb, prim_cursor);
|
||||||
|
tb_data(& tb, u4_(smem.floor.faces));
|
||||||
|
tb_data(& tb, u4_(smem.floor.verts));
|
||||||
|
tb_data(& tb, u4_(ordering_buf));
|
||||||
|
for (U4 i = 0; i < Floor_num_faces; i++) {
|
||||||
|
tb_emit(& tb, floor_f3_face);
|
||||||
|
}
|
||||||
|
// After floor_f3_face iterations complete, the primitive arena's used counter needs updating.
|
||||||
|
tb_emit(& tb, sync_primitive_arena);
|
||||||
|
tb_data(& tb, u4_(& pa->used));
|
||||||
|
tb_data(& tb, prim_base);
|
||||||
|
}
|
||||||
|
tape_run(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
|
||||||
|
|
||||||
|
// C-side state (pa->used) has already been updated by the tape!
|
||||||
|
// smem.floor.rot.y += 5;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void render(void) {
|
||||||
|
}
|
||||||
|
|
||||||
|
void gp_display_frame(DoubleBuffer* screen_buf, S4* active_buf_id, U4* ordering_buf, PrimitiveArena* pa) {
|
||||||
|
draw_sync(0);
|
||||||
|
vsync(0);
|
||||||
|
displayenv_put(& r_(screen_buf->display)[active_buf_id[0] ]);
|
||||||
|
drawenv_put (& r_(screen_buf->draw) [active_buf_id[0] ]);
|
||||||
|
{
|
||||||
|
draw_orderingtbl(ordering_buf + OrderingTbl_Len - 1);
|
||||||
|
pa->used = 0;
|
||||||
|
}
|
||||||
|
active_buf_id[0] = ! active_buf_id[0]; // Swap current buffer
|
||||||
|
}
|
||||||
|
|
||||||
|
int main(void)
|
||||||
|
{
|
||||||
|
smem = (SMemory){0};
|
||||||
|
// smem.primitives.used = 0;
|
||||||
|
// smem.active_buf_id = 0;
|
||||||
|
smem.cam.pos = v3s4(500, -1000, -1500);
|
||||||
|
/*Persistent Entity Setup*/{
|
||||||
|
ent_cube128_init(& smem.cube.verts, & smem.cube.faces); {
|
||||||
|
Ent_Cube* cube = & smem.cube;
|
||||||
|
cube->rot = v3s2(0, 0, 0);
|
||||||
|
cube->scale = v3s4_fp_one();
|
||||||
|
cube->accel = v3s4(0, 1, 0);
|
||||||
|
cube->pos = v3s4(0, -400, 1800);
|
||||||
|
}
|
||||||
|
ent_floor_init(& smem.floor.verts, & smem.floor.faces); {
|
||||||
|
Ent_Floor* floor = & smem.floor;
|
||||||
|
floor->rot = v3s2(0, 0, 0);
|
||||||
|
floor->pos = v3s4(0, 450, 1800);
|
||||||
|
floor->scale = v3s4_fp_one();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape)); {
|
||||||
|
reset_graph(0);
|
||||||
|
/* Direct BIOS: poll both ports during VBlank. */
|
||||||
|
pad_bios_init_start(& smem.pad_raw[0], & smem.pad_raw[1]);
|
||||||
|
|
||||||
|
compile_init_atoms();
|
||||||
|
compile_resolve_look_at();
|
||||||
|
|
||||||
|
/* Pinned registers for the GPU init atom. */
|
||||||
|
register U4* io_base_addr rgcc(R_IO_BaseAddr) = u4_r(IO_BASE_ADDR);
|
||||||
|
register DoubleBuffer* screen_buf rgcc(R_ScreenBuf) = & smem.screen_buf;
|
||||||
|
tb.used = 0; tb_scope_run(& tb) {
|
||||||
|
tb_emit(& tb, screen_env_init);
|
||||||
|
tb_emit(& tb, gp_screen_init);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
while (1) {
|
||||||
|
gknown S4* active_buf_id = & smem.active_buf_id;
|
||||||
|
gknown U4* ordering_buf = r_(smem.ordering_tbl)[active_buf_id[0]];
|
||||||
|
gknown PrimitiveArena* pa = & smem.primitives;
|
||||||
|
update(pa, ordering_buf);
|
||||||
|
render();
|
||||||
|
gp_display_frame(& smem.screen_buf, active_buf_id, ordering_buf, pa);
|
||||||
|
};
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
@@ -0,0 +1,102 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# pragma once
|
||||||
|
# include "duffle/dsl.h"
|
||||||
|
# include "duffle/math.h"
|
||||||
|
# include "duffle/gp.h"
|
||||||
|
# include "duffle/pad.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
enum {
|
||||||
|
// PrimitiveBuff_Len = 4096,
|
||||||
|
// OrderingTbl_Len = 2048,
|
||||||
|
PrimitiveBuff_Len = 131072,
|
||||||
|
OrderingTbl_Len = 8192,
|
||||||
|
};
|
||||||
|
|
||||||
|
enum {
|
||||||
|
ScreenRes_X = 320,
|
||||||
|
ScreenRes_Y = 240,
|
||||||
|
ScreenZ = 320,
|
||||||
|
ScreenRes_CenterX = (ScreenRes_X >> 1),
|
||||||
|
ScreenRes_CenterY = (ScreenRes_Y >> 1),
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
||||||
|
typedef Array_(OrderingTable_Buffer, 2);
|
||||||
|
|
||||||
|
typedef B1 PrimitiveBuffer[PrimitiveBuff_Len];
|
||||||
|
typedef Array_(PrimitiveBuffer, 2);
|
||||||
|
typedef Struct_(PrimitiveArena) {
|
||||||
|
A2_PrimitiveBuffer buf;
|
||||||
|
U4 used;
|
||||||
|
};
|
||||||
|
|
||||||
|
#define Cube_num_verts 8
|
||||||
|
typedef Array_(V3_S2, Cube_num_verts);
|
||||||
|
#define Cube_num_faces 6
|
||||||
|
typedef Array_(V4_S2, Cube_num_faces);
|
||||||
|
I_ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
||||||
|
LP_ A8_V3_S2 baked_verts = (A8_V3_S2) {
|
||||||
|
{ -128, -128, -128 },
|
||||||
|
{ 128, -128, -128 },
|
||||||
|
{ 128, -128, 128 },
|
||||||
|
{ -128, -128, 128 },
|
||||||
|
{ -128, 128, -128 },
|
||||||
|
{ 128, 128, -128 },
|
||||||
|
{ 128, 128, 128 },
|
||||||
|
{ -128, 128, 128 }
|
||||||
|
};
|
||||||
|
LP_ A6_V4_S2 baked_faces = (A6_V4_S2) {
|
||||||
|
{ 3, 2, 0, 1 },
|
||||||
|
{ 0, 1, 4, 5 },
|
||||||
|
{ 4, 5, 7, 6 },
|
||||||
|
{ 1, 2, 5, 6 },
|
||||||
|
{ 2, 3, 6, 7 },
|
||||||
|
{ 3, 0, 7, 4 },
|
||||||
|
};
|
||||||
|
mem_copy(u4_(verts), u4_(& baked_verts), S_(A8_V3_S2) );
|
||||||
|
mem_copy(u4_(faces), u4_(& baked_faces), S_(A6_V4_S2) );
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
typedef Struct_(Ent_Cube) {
|
||||||
|
V3_S4 accel;
|
||||||
|
V3_S4 vel;
|
||||||
|
V3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
|
||||||
|
V3_S4 scale;
|
||||||
|
V3_S2 rot;
|
||||||
|
A8_V3_S2 verts;
|
||||||
|
A6_V4_S2 faces;
|
||||||
|
};
|
||||||
|
|
||||||
|
#define Floor_num_verts 4
|
||||||
|
typedef Array_(V3_S2, Floor_num_verts);
|
||||||
|
#define Floor_num_faces 2
|
||||||
|
typedef Array_(V3_S2, Floor_num_faces);
|
||||||
|
I_ void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
|
||||||
|
LP_ A4_V3_S2 baked_verts = (A4_V3_S2) {
|
||||||
|
{ -900, 0, -900 },
|
||||||
|
{ -900, 0, 900 },
|
||||||
|
{ 900, 0, -900 },
|
||||||
|
{ 900, 0, 900 },
|
||||||
|
};
|
||||||
|
LP_ A2_V3_S2 baked_faces = (A2_V3_S2) {
|
||||||
|
{ 0, 1, 2 },
|
||||||
|
{ 1, 3, 2 },
|
||||||
|
};
|
||||||
|
mem_copy(u4_(verts), u4_(& baked_verts), S_(A4_V3_S2));
|
||||||
|
mem_copy(u4_(faces), u4_(& baked_faces), S_(A2_V3_S2));
|
||||||
|
};
|
||||||
|
typedef Struct_(Ent_Floor) {
|
||||||
|
V3_S4 accel;
|
||||||
|
V3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
|
||||||
|
V3_S4 scale;
|
||||||
|
V3_S2 rot;
|
||||||
|
A4_V3_S2 verts;
|
||||||
|
A2_V3_S2 faces;
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef Struct_(Camera) {
|
||||||
|
P3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
|
||||||
|
V3_S2 rot;
|
||||||
|
MT3_S2S4 look_at;
|
||||||
|
};
|
||||||
@@ -0,0 +1,29 @@
|
|||||||
|
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
|
||||||
|
// Source: C:\projects\Pikuma\ps1\code\gte_hello\hello_gte_tape.c
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#pragma region hello_gte_tape
|
||||||
|
|
||||||
|
|
||||||
|
// --- atom: cube_g4_face (77 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_cull_cube_g4_face_exit 42
|
||||||
|
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
|
||||||
|
atom_offset_bounds_chk_cube_g4_face_exit = _atom_offset_bounds_chk_cube_g4_face_exit,
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- atom: floor_f3_face (58 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_culling_floor_f3_face_exit 25
|
||||||
|
#define _atom_offset_bounds_chk_floor_f3_face_exit 16
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
|
||||||
|
atom_offset_bounds_chk_floor_f3_face_exit = _atom_offset_bounds_chk_floor_f3_face_exit,
|
||||||
|
};
|
||||||
|
|
||||||
|
#pragma endregion hello_gte_tape
|
||||||
|
|
||||||
@@ -0,0 +1,29 @@
|
|||||||
|
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
|
||||||
|
// Source: C:\projects\Pikuma\ps1\code\hello_gte\hello_gte.tape.c
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#pragma region hello_gte.tape
|
||||||
|
|
||||||
|
|
||||||
|
// --- atom: cube_g4_face (77 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_cull_cube_g4_face_exit 42
|
||||||
|
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
|
||||||
|
atom_offset_bounds_chk_cube_g4_face_exit = _atom_offset_bounds_chk_cube_g4_face_exit,
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- atom: floor_f3_face (58 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_culling_floor_f3_face_exit 25
|
||||||
|
#define _atom_offset_bounds_chk_floor_f3_face_exit 16
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
|
||||||
|
atom_offset_bounds_chk_floor_f3_face_exit = _atom_offset_bounds_chk_floor_f3_face_exit,
|
||||||
|
};
|
||||||
|
|
||||||
|
#pragma endregion hello_gte.tape
|
||||||
|
|
||||||
@@ -1,6 +1,6 @@
|
|||||||
#include "stdio.h"
|
#include <stdio.h>
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
#include "assert.h"
|
#include <assert.h>
|
||||||
// #include "libgpu.h"
|
// #include "libgpu.h"
|
||||||
// #include "libetc.h"
|
// #include "libetc.h"
|
||||||
// #include "libgte.h"
|
// #include "libgte.h"
|
||||||
@@ -8,19 +8,22 @@
|
|||||||
#include "duffle/dsl.h"
|
#include "duffle/dsl.h"
|
||||||
#include "duffle/memory.h"
|
#include "duffle/memory.h"
|
||||||
#include "duffle/math.h"
|
#include "duffle/math.h"
|
||||||
|
|
||||||
#include "duffle/gcc_asm.h"
|
#include "duffle/gcc_asm.h"
|
||||||
#include "duffle/mips.h"
|
#include "duffle/mips.h"
|
||||||
#include "duffle/gp.h"
|
#include "duffle/gp.h"
|
||||||
#include "duffle/gte.h"
|
#include "duffle/gte.h"
|
||||||
|
|
||||||
# include "duffle/gen/lottes_tape.offsets.h"
|
# include "duffle/gen/macs.h"
|
||||||
|
# include "duffle/gen/offsets.h"
|
||||||
#include "duffle/atom_dsl.h"
|
#include "duffle/atom_dsl.h"
|
||||||
#include "duffle/lottes_tape.h"
|
#include "duffle/lottes_tape.h"
|
||||||
|
#include "duffle/word_count.metadata.h"
|
||||||
|
|
||||||
# include "tape_atom.metadata.h"
|
# include "gen/offsets.h"
|
||||||
# include "gen/hello_gte_tape.offsets.h"
|
|
||||||
#include "hello_gte.h"
|
#include "hello_gte.h"
|
||||||
#include "hello_gte_tape.c"
|
|
||||||
|
#include "hello_gte.tape.c"
|
||||||
|
|
||||||
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
||||||
typedef Array_(OrderingTable_Buffer, 2);
|
typedef Array_(OrderingTable_Buffer, 2);
|
||||||
@@ -96,8 +99,13 @@ typedef Struct_(Ent_Floor) {
|
|||||||
A2_V3_S2 faces;
|
A2_V3_S2 faces;
|
||||||
};
|
};
|
||||||
|
|
||||||
enum { scratchpad_size = 1024, };
|
enum {
|
||||||
|
Scratchpad_Len = 1024,
|
||||||
|
MemTape_Len = 512,
|
||||||
|
};
|
||||||
typedef Struct_(SMemory) {
|
typedef Struct_(SMemory) {
|
||||||
|
U4 MemTape[MemTape_Len];
|
||||||
|
|
||||||
DoubleBuffer screen_buf;
|
DoubleBuffer screen_buf;
|
||||||
A2_OrderingTable_Buffer ordering_tbl;
|
A2_OrderingTable_Buffer ordering_tbl;
|
||||||
PrimitiveArena primitives;
|
PrimitiveArena primitives;
|
||||||
@@ -114,7 +122,7 @@ global SMemory smem;
|
|||||||
extern SMemory smem;
|
extern SMemory smem;
|
||||||
|
|
||||||
// TODO(Ed):
|
// TODO(Ed):
|
||||||
FI_ U4* spad_warm(MipsAtom atom) {
|
FI_ U4* spad_warm(Slice_MipsCode atom) {
|
||||||
return nullptr;
|
return nullptr;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -154,6 +162,10 @@ void gp_screen_init_c11(DoubleBuffer* screen_buf, S4* active_buf_id)
|
|||||||
|
|
||||||
// Initialize and setup the GTE geometry offsets
|
// Initialize and setup the GTE geometry offsets
|
||||||
geom_init();
|
geom_init();
|
||||||
|
// NOTE: geom_set_offset/geom_set_screen are kept as-is (the libgte versions
|
||||||
|
// are known to be broken in this PSYQ 4.7 build — see report 2026-07-09).
|
||||||
|
// The user's research wants the C-side non-tape reference to work as a
|
||||||
|
// known-good baseline for comparison against the tape.
|
||||||
geom_set_offset(ScreenRes_CenterX, ScreenRes_CenterY);
|
geom_set_offset(ScreenRes_CenterX, ScreenRes_CenterY);
|
||||||
geom_set_screen(ScreenZ);
|
geom_set_screen(ScreenZ);
|
||||||
|
|
||||||
@@ -175,6 +187,7 @@ void gp_display_frame(DoubleBuffer* screen_buf, S4* active_buf_id, U4* ordering_
|
|||||||
void render(void) {
|
void render(void) {
|
||||||
}
|
}
|
||||||
|
|
||||||
|
GCC_OPTIMIZATION_DISABLE
|
||||||
void update(PrimitiveArena* pa, U4* ordering_buf)
|
void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||||
{
|
{
|
||||||
orderingtbl_clear_reverse(ordering_buf, OrderingTbl_Len);
|
orderingtbl_clear_reverse(ordering_buf, OrderingTbl_Len);
|
||||||
@@ -200,13 +213,15 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
|||||||
A2_S2 p; //???
|
A2_S2 p; //???
|
||||||
S4 flag; //????
|
S4 flag; //????
|
||||||
|
|
||||||
|
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
|
||||||
|
|
||||||
// Draw Cube
|
// Draw Cube
|
||||||
if (1)
|
if (0)
|
||||||
{
|
{
|
||||||
m3s2_rotation (& smem.cube.rot, & smem.tform_world);
|
m3s2_rotation (& smem.cube.rot, & smem.tform_world);
|
||||||
m3s2_translation(& smem.tform_world, & smem.cube.pos);
|
m3s2_translation(& smem.tform_world, & smem.cube.pos);
|
||||||
m3s2_scale (& smem.tform_world, & smem.cube.scale);
|
m3s2_scale (& smem.tform_world, & smem.cube.scale);
|
||||||
gte_matrix_set_rotation (& smem.tform_world);
|
// gte_matrix_set_rotation (& smem.tform_world);
|
||||||
gte_matrix_set_translation(& smem.tform_world);
|
gte_matrix_set_translation(& smem.tform_world);
|
||||||
for (U4 face_id = 0; face_id < Cube_num_faces; face_id += 1)
|
for (U4 face_id = 0; face_id < Cube_num_faces; face_id += 1)
|
||||||
{
|
{
|
||||||
@@ -241,7 +256,7 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
|||||||
smem.cube.rot.y += 30;
|
smem.cube.rot.y += 30;
|
||||||
}
|
}
|
||||||
// Draw cube (tape method) - two triangles per face
|
// Draw cube (tape method) - two triangles per face
|
||||||
if (0)
|
if (1)
|
||||||
{
|
{
|
||||||
m3s2_rotation (& smem.cube.rot, & smem.tform_world);
|
m3s2_rotation (& smem.cube.rot, & smem.tform_world);
|
||||||
m3s2_translation(& smem.tform_world, & smem.cube.pos);
|
m3s2_translation(& smem.tform_world, & smem.cube.pos);
|
||||||
@@ -252,9 +267,8 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
|||||||
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||||
U4 prim_cursor = prim_base + pa->used;
|
U4 prim_cursor = prim_base + pa->used;
|
||||||
|
|
||||||
LP_ U4 mem_temp_tape[512]; FArena tape_arena; farena_init(& tape_arena, slice_ut_arr(mem_temp_tape));
|
tb.used = 0; tb_scope(& tb) {
|
||||||
TapeBuilder tb = tb_make_old(&tape_arena); tb_scope(& tb) {
|
tb_emit(& tb, rbind_cube_g4_face);
|
||||||
tb_emit(& tb, code_rbind_cube_tri);
|
|
||||||
tb_data(& tb, prim_cursor);
|
tb_data(& tb, prim_cursor);
|
||||||
tb_data(& tb, u4_(smem.cube.faces));
|
tb_data(& tb, u4_(smem.cube.faces));
|
||||||
tb_data(& tb, u4_(smem.cube.verts));
|
tb_data(& tb, u4_(smem.cube.verts));
|
||||||
@@ -262,10 +276,10 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
|||||||
|
|
||||||
for (U4 i = 0; i < Cube_num_faces; i++) {
|
for (U4 i = 0; i < Cube_num_faces; i++) {
|
||||||
// Two triangles per quad face: (x,y,z) and (x,z,w)
|
// Two triangles per quad face: (x,y,z) and (x,z,w)
|
||||||
tb_emit(& tb, code_cube_tri);
|
tb_emit(& tb, cube_g4_face);
|
||||||
}
|
}
|
||||||
|
|
||||||
tb_emit(& tb, code_sync_primitive_arena);
|
tb_emit(& tb, sync_primitive_arena);
|
||||||
tb_data(& tb, u4_(& pa->used));
|
tb_data(& tb, u4_(& pa->used));
|
||||||
tb_data(& tb, prim_base);
|
tb_data(& tb, prim_base);
|
||||||
}
|
}
|
||||||
@@ -333,67 +347,55 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
|||||||
m3s2_rotation (& smem.floor.rot, & smem.tform_world);
|
m3s2_rotation (& smem.floor.rot, & smem.tform_world);
|
||||||
m3s2_translation(& smem.tform_world, & smem.floor.pos);
|
m3s2_translation(& smem.tform_world, & smem.floor.pos);
|
||||||
m3s2_scale (& smem.tform_world, & smem.floor.scale);
|
m3s2_scale (& smem.tform_world, & smem.floor.scale);
|
||||||
// TODO(Ed): This can either be in the tape or here...
|
|
||||||
// gte_matrix_set_rotation (& smem.tform_world);
|
|
||||||
// gte_matrix_set_translation(& smem.tform_world);
|
|
||||||
|
|
||||||
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||||
U4 prim_cursor = prim_base + pa->used;
|
U4 prim_cursor = prim_base + pa->used;
|
||||||
|
|
||||||
// TODO(Ed): We should do a bounds check beforehand to confirm pa can hold all tris.
|
// TODO(Ed): We should do a bounds check beforehand to confirm pa can hold all tris?
|
||||||
// The tape atoms in-flight should not need to care.
|
// The tape atoms in-flight should not need to care.
|
||||||
|
|
||||||
// Prepare the tape. (Push protocol to tape)
|
// Prepare the tape. (Push protocol to tape)
|
||||||
LP_ U4 mem_temp_tape[512];
|
tb.used = 0; tb_scope(& tb) {
|
||||||
TapeBuilder tb = tb_make(slice_ut_arr(mem_temp_tape)); tb_scope(& tb) {
|
tb_emit(& tb, set_gte_world);
|
||||||
// TODO(Ed): This is bugged.
|
|
||||||
tb_emit(& tb, code_set_gte_world);
|
|
||||||
tb_data(& tb, u4_(& smem.tform_world));
|
tb_data(& tb, u4_(& smem.tform_world));
|
||||||
|
|
||||||
tb_emit(& tb, code_rbind_floor_tri);
|
tb_emit(& tb, rbind_floor_f3_face);
|
||||||
// TODO(Ed): Just use a single context struct ref
|
// TODO(Ed): Just use a single context struct ref
|
||||||
tb_data(& tb, prim_cursor);
|
tb_data(& tb, prim_cursor);
|
||||||
tb_data(& tb, u4_(smem.floor.faces));
|
tb_data(& tb, u4_(smem.floor.faces));
|
||||||
tb_data(& tb, u4_(smem.floor.verts));
|
tb_data(& tb, u4_(smem.floor.verts));
|
||||||
tb_data(& tb, u4_(ordering_buf));
|
tb_data(& tb, u4_(ordering_buf));
|
||||||
for (U4 i = 0; i < Floor_num_faces; i++) {
|
for (U4 i = 0; i < Floor_num_faces; i++) {
|
||||||
tb_emit(& tb, code_floor_tri);
|
tb_emit(& tb, floor_f3_face);
|
||||||
}
|
}
|
||||||
// After code_floor_tri iterations complete, the primitive arena's used counter needs updating.
|
// After floor_f3_face iterations complete, the primitive arena's used counter needs updating.
|
||||||
tb_emit(& tb, code_sync_primitive_arena);
|
tb_emit(& tb, sync_primitive_arena);
|
||||||
tb_data(& tb, u4_(& pa->used));
|
tb_data(& tb, u4_(& pa->used));
|
||||||
tb_data(& tb, prim_base);
|
tb_data(& tb, prim_base);
|
||||||
}
|
}
|
||||||
|
|
||||||
tape_run(tb_slice(tb));// Fire off the tape.
|
tape_run(tb_slice(tb));// Fire off the tape.
|
||||||
|
|
||||||
// C-side state (pa->used) has already been updated by the tape!
|
// C-side state (pa->used) has already been updated by the tape!
|
||||||
smem.floor.rot.y += 5;
|
smem.floor.rot.y += 5;
|
||||||
}
|
}
|
||||||
// --- TAPE DIAGNOSTICS ---
|
// --- TAPE DIAGNOSTICS ---
|
||||||
if (1)
|
if (0)
|
||||||
{
|
{
|
||||||
LP_ U4 mem_temp_tape[512]; FArena tape_arena; farena_init(& tape_arena, slice_ut_arr(mem_temp_tape));
|
LP_ U4 mem_temp_tape[512]; FArena tape_arena; farena_init(& tape_arena, slice_ut_arr(mem_temp_tape));
|
||||||
TapeBuilder tb = tb_make_old(& tape_arena); tb_scope(& tb) {
|
TapeBuilder tb = tb_make_old(& tape_arena); tb_scope(& tb) {
|
||||||
// Skip set_gte_world atom for diagnostics to isolate the triangle loop
|
// Skip set_gte_world atom for diagnostics to isolate the triangle loop
|
||||||
for (U4 i = 0; i < Floor_num_faces; i++) {
|
for (U4 i = 0; i < Floor_num_faces; i++) {
|
||||||
// =======================================================
|
|
||||||
// SWAP EMIT TO TEST DIFFERENT PARTS OF THE PIPELINE:
|
|
||||||
// =======================================================
|
|
||||||
// 1. code_diag_yield -> Tests Tape Engine jump logic
|
|
||||||
// 2. code_diag_color -> Tests OT and Prim Arena memory
|
|
||||||
// 3. code_diag_gte -> Tests Vertex arrays and GTE Math
|
|
||||||
// tb_emit(& tb, code_diag_yield);
|
// tb_emit(& tb, code_diag_yield);
|
||||||
tb_emit(& tb, code_diag_color);
|
// tb_emit(& tb, code_diag_color);
|
||||||
// tb_emit(& tb, code_diag_gte);
|
// tb_emit(& tb, code_diag_gte);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
B1* prim_cursor = (B1*)r_(pa->buf)[smem.active_buf_id] + pa->used;
|
B1* prim_cursor = (B1*)r_(pa->buf)[smem.active_buf_id] + pa->used;
|
||||||
tape_run(tb_slice(tb));
|
tape_run(tb_slice(tb));
|
||||||
pa->used = (U4)prim_cursor - (U4)r_(pa->buf)[smem.active_buf_id];
|
pa->used = (U4)prim_cursor - (U4)r_(pa->buf)[smem.active_buf_id];
|
||||||
smem.floor.rot.y += 5;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
GCC_OPTIMIZATION_ENABLE
|
||||||
|
|
||||||
int main(void)
|
int main(void)
|
||||||
{
|
{
|
||||||
@@ -63,14 +63,6 @@ U4 vsync(U4 mode) __asm__("VSync");
|
|||||||
|
|
||||||
void draw_orderingtbl(U4* buf) __asm__("DrawOTag");
|
void draw_orderingtbl(U4* buf) __asm__("DrawOTag");
|
||||||
|
|
||||||
/* Primitive Handling Macros
|
|
||||||
* All primitive types (PolyTag, Poly_F3, Poly_F4, Poly_G3, Poly_G4,
|
|
||||||
* Poly_FT3, Poly_FT4, Poly_GT3, Poly_GT4) and the set_poly_* setters,
|
|
||||||
* set_len / set_addr / get_len / get_addr macros, and the
|
|
||||||
* orderingtbl_add_primitive(s) helpers all live in `duffle/gp.h`
|
|
||||||
* now (per the Phase 3 gp.h overhaul). This file no longer duplicates
|
|
||||||
* those definitions. */
|
|
||||||
|
|
||||||
typedef Struct_(Tile) {
|
typedef Struct_(Tile) {
|
||||||
U4 tag;
|
U4 tag;
|
||||||
RGB8 color;
|
RGB8 color;
|
||||||
@@ -78,7 +70,6 @@ typedef Struct_(Tile) {
|
|||||||
Rect_S2 rect;
|
Rect_S2 rect;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|
||||||
/*
|
/*
|
||||||
Linear Algebra
|
Linear Algebra
|
||||||
*/
|
*/
|
||||||
@@ -0,0 +1,218 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# include "duffle/gen/macs.h"
|
||||||
|
# include "duffle/gen/offsets.h"
|
||||||
|
# include "duffle/atom_dsl.h"
|
||||||
|
# include "duffle/lottes_tape.h"
|
||||||
|
# include "duffle/word_count.metadata.h"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
# include "hello_gte.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#pragma region MACs (Mips Atom components)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
#pragma endregion MACs
|
||||||
|
|
||||||
|
#pragma region Baked Atoms
|
||||||
|
|
||||||
|
/* DIAGNOSTIC 1: Pure tape loop test */
|
||||||
|
internal MipsAtom_(diag_yield) { mac_yield() };
|
||||||
|
|
||||||
|
/* DIAGNOSTIC 2: Pure memory test (No GTE). Draws a fixed cyan triangle. */
|
||||||
|
internal MipsAtom_(diag_color) {
|
||||||
|
store_word( R_0, R_T7, 0),
|
||||||
|
load_upper_i(R_AT, gp0_cmd_poly_f3 << 8 | 0xFF), /* High: MipsCode Poly_F3(0x20) + Color B:FF */
|
||||||
|
or_i_self( R_AT, 0xFF00), /* Low: Color G:FF, R:00 (Cyan) */
|
||||||
|
store_word( R_AT, R_T7, 4),
|
||||||
|
|
||||||
|
/* Fake coordinates - Swapped winding order to prevent GPU culling! */
|
||||||
|
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 8), /* (16, 16) */
|
||||||
|
load_upper_i(R_AT, 0x0050), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 12), /* (80, 16) */
|
||||||
|
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0050), store_word(R_AT, R_T7, 16), /* (16, 80) */
|
||||||
|
|
||||||
|
add_ui( R_T1, R_0, 10),
|
||||||
|
shift_lleft_self(R_T1, S_(U4)/2),
|
||||||
|
add_u_self( R_T1, R_T6),
|
||||||
|
|
||||||
|
load_word( R_AT, R_T1, 0),
|
||||||
|
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits),
|
||||||
|
store_word( R_AT, R_T7, 0),
|
||||||
|
shift_lleft(R_AT, R_T7, S_(PolyTag_len_bits)), shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
|
||||||
|
or_u_self( R_AT, R_V0),
|
||||||
|
store_word( R_AT, R_T1, 0),
|
||||||
|
|
||||||
|
add_ui(R_T7, R_T7, 20),
|
||||||
|
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
/* DIAGNOSTIC 3: Pure GTE test (No Memory Writes) */
|
||||||
|
internal MipsAtom_(diag_gte) {
|
||||||
|
/* Load 3 indices */
|
||||||
|
load_half_u(R_T0, R_T4, 0),
|
||||||
|
load_half_u(R_T1, R_T4, 2),
|
||||||
|
load_half_u(R_T2, R_T4, 4),
|
||||||
|
|
||||||
|
/* Load Vertices into GTE */
|
||||||
|
shift_lleft( R_AT, R_T0, 3), add_u( R_AT, R_AT, R_T5),
|
||||||
|
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||||
|
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||||
|
|
||||||
|
shift_lleft( R_AT, R_T1, 3), add_u(R_AT, R_AT, R_T5),
|
||||||
|
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||||
|
gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
||||||
|
|
||||||
|
shift_lleft(R_AT, R_T2, 3), add_u(R_AT, R_AT, R_T5),
|
||||||
|
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||||
|
gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
||||||
|
|
||||||
|
/* Run Math */
|
||||||
|
nop2, gte_cmdw_rtpt,
|
||||||
|
nop2, gte_cmdw_nclip,
|
||||||
|
nop2,
|
||||||
|
|
||||||
|
/* Advance Face Cursor and Yield */
|
||||||
|
add_ui(R_T4, R_T4, 8),
|
||||||
|
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef Struct_(Binds_CubeTri) {
|
||||||
|
U4 PrimCursor;
|
||||||
|
V4_S2* FaceCursor;
|
||||||
|
V3_S2* VertBase;
|
||||||
|
U4* OtBase;
|
||||||
|
};
|
||||||
|
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)
|
||||||
|
, atom_reads(R_TapePtr)
|
||||||
|
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||||
|
){
|
||||||
|
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||||
|
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
|
||||||
|
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
|
||||||
|
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
|
||||||
|
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
|
||||||
|
internal
|
||||||
|
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||||
|
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
||||||
|
atom_writes(R_PrimCursor, R_FaceCursor)
|
||||||
|
){
|
||||||
|
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||||
|
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
||||||
|
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
||||||
|
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
||||||
|
|
||||||
|
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
|
||||||
|
nop2, gte_cmdw_rotate_translate_perspective_triple, // required cpu -> gte delay slot
|
||||||
|
gte_cmdw_nclip,
|
||||||
|
|
||||||
|
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
|
||||||
|
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)), nop,
|
||||||
|
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
||||||
|
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
||||||
|
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
|
||||||
|
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||||
|
|
||||||
|
mac_gte_store_g4_p012(),
|
||||||
|
gte_cmdw_rotate_translate_perspective_single,
|
||||||
|
mac_gte_store_g4_p3(),
|
||||||
|
|
||||||
|
gte_cmdw_avg_sort_z4,
|
||||||
|
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||||
|
|
||||||
|
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||||
|
set_lt_u( R_AT, R_T1, R_AT),
|
||||||
|
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
|
||||||
|
mac_insert_ot_tag_g4(),
|
||||||
|
mac_format_g4_color(
|
||||||
|
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
||||||
|
/* c1 yellow */ 0xFF, 0xFF, 0x00,
|
||||||
|
/* c2 cyan */ 0x00, 0xFF, 0xFF,
|
||||||
|
/* c3 green */ 0x00, 0xFF, 0x00),
|
||||||
|
// end: branch(bounds_chk)
|
||||||
|
// end: branch(cull)
|
||||||
|
|
||||||
|
atom_label(cube_g4_face_exit)
|
||||||
|
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
|
||||||
|
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef Struct_(Binds_FloorTri) {
|
||||||
|
U4 PrimCursor;
|
||||||
|
V3_S2* FaceCursor;
|
||||||
|
V3_S2* VertBase;
|
||||||
|
U4* OtBase;
|
||||||
|
};
|
||||||
|
internal
|
||||||
|
MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(floor_f3)
|
||||||
|
, atom_reads(R_TapePtr)
|
||||||
|
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||||
|
){
|
||||||
|
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||||
|
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
|
||||||
|
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
|
||||||
|
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
|
||||||
|
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
// atom_dbg_skip
|
||||||
|
internal
|
||||||
|
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
||||||
|
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||||
|
, atom_writes(R_PrimCursor, R_FaceCursor)
|
||||||
|
) {
|
||||||
|
mac_load_tri_indices( R_T0, R_T1, R_T2),
|
||||||
|
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
|
||||||
|
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
|
||||||
|
gte_cmdw_nclip,
|
||||||
|
|
||||||
|
/* Culling (Branch forward if Backface) */
|
||||||
|
gte_mv_from_data_r(R_T0, C2_MAC0),
|
||||||
|
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
|
||||||
|
/* Format Primitive */
|
||||||
|
mac_gte_store_f3(),
|
||||||
|
|
||||||
|
/* Calculate Depth */
|
||||||
|
gte_avg_sort_z3,
|
||||||
|
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||||
|
/* Bounds Check OTZ < OrderingTbl_Len (Branch forward to skip insertion) */
|
||||||
|
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||||
|
set_lt_u( R_AT, R_T1, R_AT),
|
||||||
|
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
|
||||||
|
mac_format_f3_color(0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
|
||||||
|
mac_insert_ot_tag_f3(), /* Insert into Ordering Table Linked List */
|
||||||
|
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
|
||||||
|
// Note(Ed): No bounds checking, should be checked before atom runs.
|
||||||
|
// end: branch(bounds_chk)
|
||||||
|
// end: branch(culling)
|
||||||
|
|
||||||
|
/* Advance Input Cursor & Yield (Both branch targets land here) */
|
||||||
|
atom_label(floor_f3_face_exit)
|
||||||
|
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
|
||||||
|
internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitiveArena)
|
||||||
|
, atom_reads( R_TapePtr, R_PrimCursor)
|
||||||
|
, atom_writes(R_TapePtr)
|
||||||
|
){
|
||||||
|
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
|
||||||
|
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
|
||||||
|
/* Calculate byte offset and store directly back to RAM */
|
||||||
|
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
|
||||||
|
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
#pragma endregion Baked Atoms
|
||||||
@@ -0,0 +1,41 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
#pragma once
|
||||||
|
#endif
|
||||||
|
// Auto-generated by ps1_meta.lua — DO NOT EDIT
|
||||||
|
// Directory: C:\projects\Pikuma\ps1\code\hello_joypad/
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.atom.c
|
||||||
|
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
|
||||||
|
|
||||||
|
#ifndef WORD_COUNT
|
||||||
|
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#define mac_put_disp_env(reg_transfer, reg_base, port) \
|
||||||
|
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port)
|
||||||
|
WORD_COUNT(mac_put_disp_env, 5)
|
||||||
|
|
||||||
|
#define mac_put_draw_env(reg_transfer, reg_base, port) \
|
||||||
|
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port) /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */ \
|
||||||
|
, mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port) /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */ \
|
||||||
|
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[1] TextureWindow (tw=(0,0)) */ \
|
||||||
|
, mac_gcmd_push(enc_gp0_draw_area_tl_word(0, ScreenRes_Y), reg_transfer, reg_base, port) /* code[2] DrawArea top-left (clip.x=0, clip.y=ScreenRes_Y=240) */ \
|
||||||
|
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) /* code[3] DrawArea bottom-right (clip.x+w=320, clip.y+h=480) */ \
|
||||||
|
, mac_gcmd_push(gp0_word_set_draw_offset(), reg_transfer, reg_base, port) /* code[4] DrawOffset (ofs=(0,0)) — bare-cmd word; the GPU uses the current state machine. */ \
|
||||||
|
, mac_gcmd_push(gp0_word_dr_env_mask(), reg_transfer, reg_base, port) /* code[5] Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit. */ \
|
||||||
|
, mac_gcmd_push(gp0_word_dr_env_bg_color_cmd(1, 7, 7, 7), reg_transfer, reg_base, port) /* code[6] Initial-bg-color + auto-clear (isbg=1, r=7, g=7, b=7). */ \
|
||||||
|
, mac_gcmd_push(gp0_word_dr_env_draw_mode(1), reg_transfer, reg_base, port) /* code[7] Re-assert DrawMode with isbg=1 (isbg-flag set; the 0xE1 cmd byte plus isbg only). */ /* code[8..10] Padding (NOP — GPU discards; the DR_ENV requires 16 words total). */ \
|
||||||
|
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) /* code[11..12] TextureWindow bottom-right (tw.x+tw.w=0, tw.y+tw.h=0) — libpsyx emits twice. */ \
|
||||||
|
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[13..14] Padding (NOP) — completes the 16-word packet. */ \
|
||||||
|
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port)
|
||||||
|
WORD_COUNT(mac_put_draw_env, 16)
|
||||||
|
|
||||||
@@ -0,0 +1,76 @@
|
|||||||
|
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
|
||||||
|
// Directory: C:\projects\Pikuma\ps1\code\hello_joypad\
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.atom.c
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#pragma region hello_joypad
|
||||||
|
|
||||||
|
|
||||||
|
// --- atom: cube_g4_face (76 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_cull_cube_g4_face_exit 41
|
||||||
|
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
|
||||||
|
atom_offset_bounds_chk_cube_g4_face_exit = _atom_offset_bounds_chk_cube_g4_face_exit,
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- atom: floor_f3_face (58 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_culling_floor_f3_face_exit 25
|
||||||
|
#define _atom_offset_bounds_chk_floor_f3_face_exit 16
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
|
||||||
|
atom_offset_bounds_chk_floor_f3_face_exit = _atom_offset_bounds_chk_floor_f3_face_exit,
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- atom: pad_bios_snapshot (78 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_snap_root_skip_disconnected 8
|
||||||
|
#define _atom_offset_disconnected_snap_end 61
|
||||||
|
#define _atom_offset_case_2_id_dispatch 8
|
||||||
|
#define _atom_offset_pending_snap_end 51
|
||||||
|
#define _atom_offset_id_dispatch_try_analog_stick 11
|
||||||
|
#define _atom_offset_id_dispatch_snap_end 38
|
||||||
|
#define _atom_offset_try_analog_stick_try_analog_pad 12
|
||||||
|
#define _atom_offset_analog_stick_snap_end 24
|
||||||
|
#define _atom_offset_try_analog_pad_try_unsupported 11
|
||||||
|
#define _atom_offset_analog_pad_snap_end 10
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_snap_root_skip_disconnected = _atom_offset_snap_root_skip_disconnected,
|
||||||
|
atom_offset_disconnected_snap_end = _atom_offset_disconnected_snap_end,
|
||||||
|
atom_offset_case_2_id_dispatch = _atom_offset_case_2_id_dispatch,
|
||||||
|
atom_offset_pending_snap_end = _atom_offset_pending_snap_end,
|
||||||
|
atom_offset_id_dispatch_try_analog_stick = _atom_offset_id_dispatch_try_analog_stick,
|
||||||
|
atom_offset_id_dispatch_snap_end = _atom_offset_id_dispatch_snap_end,
|
||||||
|
atom_offset_try_analog_stick_try_analog_pad = _atom_offset_try_analog_stick_try_analog_pad,
|
||||||
|
atom_offset_analog_stick_snap_end = _atom_offset_analog_stick_snap_end,
|
||||||
|
atom_offset_try_analog_pad_try_unsupported = _atom_offset_try_analog_pad_try_unsupported,
|
||||||
|
atom_offset_analog_pad_snap_end = _atom_offset_analog_pad_snap_end,
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- atom: pad_apply_input (60 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_dpad_left_exit_dpad_left 6
|
||||||
|
#define _atom_offset_dpad_right_exit_dpad_right 6
|
||||||
|
#define _atom_offset_dead_zone_low_check_dead_low_active 8
|
||||||
|
#define _atom_offset_dead_zone_high_check_dead_high_active 15
|
||||||
|
#define _atom_offset_dead_zone_skip_exit_stick 24
|
||||||
|
#define _atom_offset_end_low_exit_stick 12
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_dpad_left_exit_dpad_left = _atom_offset_dpad_left_exit_dpad_left,
|
||||||
|
atom_offset_dpad_right_exit_dpad_right = _atom_offset_dpad_right_exit_dpad_right,
|
||||||
|
atom_offset_dead_zone_low_check_dead_low_active = _atom_offset_dead_zone_low_check_dead_low_active,
|
||||||
|
atom_offset_dead_zone_high_check_dead_high_active = _atom_offset_dead_zone_high_check_dead_high_active,
|
||||||
|
atom_offset_dead_zone_skip_exit_stick = _atom_offset_dead_zone_skip_exit_stick,
|
||||||
|
atom_offset_end_low_exit_stick = _atom_offset_end_low_exit_stick,
|
||||||
|
};
|
||||||
|
|
||||||
|
#pragma endregion hello_joypad
|
||||||
|
|
||||||
@@ -0,0 +1,638 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# pragma once
|
||||||
|
# include "duffle/gen/macs.h"
|
||||||
|
# include "duffle/gen/offsets.h"
|
||||||
|
# include "duffle/dsl.atom.h"
|
||||||
|
# include "duffle/lottes_tape.h"
|
||||||
|
# include "duffle/mips.h"
|
||||||
|
# include "duffle/gte.h"
|
||||||
|
# include "duffle/gp.h"
|
||||||
|
# include "duffle/pad.h"
|
||||||
|
# include "duffle/word_count.metadata.h"
|
||||||
|
# include "duffle/psyq.h"
|
||||||
|
# include "duffle/math.atom.c"
|
||||||
|
# include "duffle/mips.atom.c"
|
||||||
|
# include "duffle/gte.atom.c"
|
||||||
|
# include "duffle/gp.atom.c"
|
||||||
|
# include "duffle/psyq.atom.c"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "hello_joypad.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c);
|
||||||
|
|
||||||
|
#pragma region MACs (Mips Atom components)
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_put_disp_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
||||||
|
MipsAtomComp_Proc_(ab, {
|
||||||
|
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
|
||||||
|
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
|
||||||
|
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_put_draw_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
||||||
|
MipsAtomComp_Proc_(ab, {
|
||||||
|
/*
|
||||||
|
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
|
||||||
|
* References:
|
||||||
|
* - libpsyx source: `toolchain/psyq-4_7/lib/libgpu.a` (binary, function `PutDrawEnv`)
|
||||||
|
* - PSX-SPX doc: https://problemkaputt.de/psx-spx.htm#gputdrawingcommands
|
||||||
|
* - PSYQ SDK: `setdrawenv` / `makelongdr_env` source
|
||||||
|
* - NOCASH PSX spec: §"GP0(E1h) Draw Mode setting" through §"DR_ENV"
|
||||||
|
*
|
||||||
|
* The 16-word format is documented in the PSYQ SDK manual and on NOCASH's PSX-spec.txt. The libpsyx reference is at:
|
||||||
|
* ./toolchain/psyq-4_7/lib/libgpu.a
|
||||||
|
* (binary; the PutDrawEnv implementation builds the 16-word DR_ENV from the user's DRAWENV struct and emits it via GP0 GPU commands.)
|
||||||
|
*
|
||||||
|
* Word indices (libpsyx PutDrawEnv / SetDrawEnv order):
|
||||||
|
* tag = (length << 24) | addr — 16-word packet (1 tag + 15 code)
|
||||||
|
* code[0] = DrawMode (dfe=1, dtd=0, tpage=0) — must come first per libpsyx
|
||||||
|
* code[1] = TextureWindow (tw=(0,0)) — bare-cmd word; GPU uses current state
|
||||||
|
* code[2] = DrawArea top-left (clip.x=0, clip.y=240)
|
||||||
|
* code[3] = DrawArea bottom-right (clip.x+w=320, clip.y+h=480)
|
||||||
|
* code[4] = DrawOffset (ofs=(0,0)) — bare-cmd word
|
||||||
|
* code[5] = Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit
|
||||||
|
* code[6] = Initial-bg-color (isbg=1, r=7, g=7, b=7)
|
||||||
|
* code[7] = DrawMode (isbg=1, tpage=0) — re-asserts DrawMode with isbg
|
||||||
|
* code[8..10] = padding (NOP) — 3 words to fill the packet
|
||||||
|
* code[11..12] = TextureWindow bottom-right — defaults to (0,0,0,0)
|
||||||
|
* code[13..14] = padding (NOP) — completes the 16-word packet
|
||||||
|
*/
|
||||||
|
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port), /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */
|
||||||
|
mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port), /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */
|
||||||
|
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port), /* code[1] TextureWindow (tw=(0,0)) */
|
||||||
|
mac_gcmd_push(enc_gp0_draw_area_tl_word(0, ScreenRes_Y), reg_transfer, reg_base, port), /* code[2] DrawArea top-left (clip.x=0, clip.y=ScreenRes_Y=240) */
|
||||||
|
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port), /* code[3] DrawArea bottom-right (clip.x+w=320, clip.y+h=480) */
|
||||||
|
|
||||||
|
mac_gcmd_push(gp0_word_set_draw_offset(), reg_transfer, reg_base, port), /* code[4] DrawOffset (ofs=(0,0)) — bare-cmd word; the GPU uses the current state machine. */
|
||||||
|
mac_gcmd_push(gp0_word_dr_env_mask(), reg_transfer, reg_base, port), /* code[5] Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit. */
|
||||||
|
mac_gcmd_push(gp0_word_dr_env_bg_color_cmd(1, 7, 7, 7), reg_transfer, reg_base, port), /* code[6] Initial-bg-color + auto-clear (isbg=1, r=7, g=7, b=7). */
|
||||||
|
mac_gcmd_push(gp0_word_dr_env_draw_mode(1), reg_transfer, reg_base, port), /* code[7] Re-assert DrawMode with isbg=1 (isbg-flag set; the 0xE1 cmd byte plus isbg only). */
|
||||||
|
|
||||||
|
/* code[8..10] Padding (NOP — GPU discards; the DR_ENV requires 16 words total). */
|
||||||
|
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||||
|
|
||||||
|
/* code[11..12] TextureWindow bottom-right (tw.x+tw.w=0, tw.y+tw.h=0) — libpsyx emits twice. */
|
||||||
|
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port),
|
||||||
|
|
||||||
|
/* code[13..14] Padding (NOP) — completes the 16-word packet. */
|
||||||
|
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||||
|
})
|
||||||
|
|
||||||
|
#pragma endregion MACs
|
||||||
|
|
||||||
|
#pragma region Baked Atoms
|
||||||
|
|
||||||
|
enum {
|
||||||
|
R_ScreenX = R_T5 atom_reg atom_type(U2),
|
||||||
|
R_ScreenY = R_T6 atom_reg atom_type(U2),
|
||||||
|
R_ScreenBuf = R_T7 atom_reg, /* Caller-pinned: & smem.screen_buf */
|
||||||
|
#define R_ScreenBuf_Code R_T7_Code
|
||||||
|
};
|
||||||
|
//screen_env_init. Mirrors the libpsyx's SetDefDispEnv + SetDefDrawEnv + the manual enable_auto_clear / initial_bg_color writes.
|
||||||
|
internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
|
||||||
|
, atom_reads(R_T0, R_ScreenX, R_ScreenY, R_ScreenBuf)
|
||||||
|
, atom_writes(R_T0, R_ScreenX, R_ScreenY)
|
||||||
|
) {
|
||||||
|
/* display[0] = (0, 0, 320, 240); rest of struct zeroed. */
|
||||||
|
add_ui(R_ScreenX, R_0, ScreenRes_X), add_ui(R_ScreenY, R_0, ScreenRes_Y),
|
||||||
|
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area.width) + OA_(DoubleBuffer,display,0)),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,display_area) + OA_(DoubleBuffer,display,0)),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + OA_(DoubleBuffer,display,0)),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,0)),
|
||||||
|
|
||||||
|
/* display[1] = (0, 240, 320, 240); rest of struct zeroed. */
|
||||||
|
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area) + OA_(DoubleBuffer,display,1)),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + OA_(DoubleBuffer,display,1)),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,1)),
|
||||||
|
|
||||||
|
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + OA_(DoubleBuffer,draw,0)), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
|
||||||
|
mac_store_v2s2(R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + OA_(DoubleBuffer,draw,0)), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
|
||||||
|
|
||||||
|
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
|
||||||
|
/* draw[0].texture_window = (0, 0, 0, 0); two word-zeroes cover the full 8-byte tw field. */
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + OA_(DoubleBuffer,draw,0)),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,0)),
|
||||||
|
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,drawing_offset[0].x) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
|
||||||
|
/* draw[0].texture_page = 10 (gp0_tpage_default). C11 SetDefDrawEnv at C11_only.elf:0x8001273C writes the same 0x0A. . */
|
||||||
|
add_ui(R_T0, R_0, gp0_tpage_default),
|
||||||
|
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,0)),
|
||||||
|
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
|
||||||
|
/* draw[0] control bytes: flag_dither=1, flag_draw_on_display=1 (the dfe bit per psx-spx; libpsyx sets it via `SetDefDrawEnv`'s conditional at C11_only.elf:0x80012728), enable_auto_clear=1. Each byte is named;
|
||||||
|
* the previous `store_word(R_0, ..., +20)` overwrote all four with zero. */
|
||||||
|
add_ui(R_T0, R_0, 1),
|
||||||
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + OA_(DoubleBuffer,draw,0)),
|
||||||
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + OA_(DoubleBuffer,draw,0)),
|
||||||
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,0)),
|
||||||
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
|
||||||
|
/* draw[0].initial_bg_color = (r=7, g=7, b=7). */
|
||||||
|
add_ui(R_T0, R_0, 7),
|
||||||
|
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,0)),
|
||||||
|
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
|
||||||
|
mac_yield(),
|
||||||
|
};
|
||||||
|
|
||||||
|
enum {
|
||||||
|
R_IO_BaseAddr = R_T4 atom_reg, /* Caller-pinned: IO_BASE_ADDR = 0x1F800000 */
|
||||||
|
#define R_IO_BaseAddr_Code R_T4_Code
|
||||||
|
};
|
||||||
|
internal MipsAtom_(gp_screen_init) atom_info(atom_phase(screen_init), atom_reads(R_IO_BaseAddr)) {
|
||||||
|
store_word(R_0, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(00h) Reset */
|
||||||
|
mac_gcmd_push(gp1_word_ResetCmdBuffer(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(01h) ClearFIFO */
|
||||||
|
mac_gcmd_push(gp1_word_AcknowledgeIRQ(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(02h) AckIRQ */
|
||||||
|
mac_gcmd_push(gp1_word_DisplayOn(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(03h) Display ON */
|
||||||
|
mac_gcmd_push(gp1_word_dma_to_gpu(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(04h) DMADirection=2 (CPU→GPU). libpsyx's per-frame PutDrawEnv/DrawOTag use DMA2; without this the DMA queue never drains. */
|
||||||
|
mac_gcmd_push(gp1_word_StartDisplayArea(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(05h) StartDisplayArea (X=0, Y=0) */
|
||||||
|
|
||||||
|
/* GP1: DisplayMode + Display Ranges */
|
||||||
|
mac_gcmd_push(gp1_word_display_mode_320x240_15bit_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||||
|
mac_gcmd_push(gp1_word_horizontal_range_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||||
|
mac_gcmd_push(gp1_word_vertical_range_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||||
|
|
||||||
|
/* GTE: SetGeomOffset (OFX, OFY) — ScreenRes_CenterX, ScreenRes_CenterY. */
|
||||||
|
load_upper_i(R_T5, ScreenRes_CenterX), gte_mv_to_ctrl_r(R_T5, gte_cr_OFX_Code),
|
||||||
|
load_upper_i(R_T5, ScreenRes_CenterY), gte_mv_to_ctrl_r(R_T5, gte_cr_OFY_Code),
|
||||||
|
|
||||||
|
/* GTE: SetGeomScreen (H) — CR26 (per PSX-SPX / libpsyx), value is the raw projection-plane distance, NOT shifted. */
|
||||||
|
add_ui(R_T5, R_0, ScreenZ), gte_mv_to_ctrl_r(R_T5, gte_cr_H_Code),
|
||||||
|
|
||||||
|
/* GP1: DisplayEnable — bit 0 = 0 (Display ON). */
|
||||||
|
mac_gcmd_push(gp1_word_DisplayOn(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||||
|
mac_yield(),
|
||||||
|
};
|
||||||
|
|
||||||
|
enum {
|
||||||
|
R_PrimCursor = R_T7 atom_reg atom_type(U4*), /* VRAM output cursor (primitive buffer) */
|
||||||
|
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2*), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
|
||||||
|
R_VertBase = R_T5 atom_reg atom_type(V3_S2*), /* Base address of the vertex array */
|
||||||
|
R_OtBase = R_T6 atom_reg atom_type(U4*), /* Base address of the Ordering Table */
|
||||||
|
#define R_PrimCursor_Code R_T7_Code
|
||||||
|
#define R_FaceCursor_Code R_T4_Code
|
||||||
|
#define R_VertBase_Code R_T5_Code
|
||||||
|
#define R_OtBase_Code R_T6_Code
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef Struct_(Binds_CubeTri) {
|
||||||
|
U4 PrimCursor;
|
||||||
|
V4_S2* FaceCursor;
|
||||||
|
V3_S2* VertBase;
|
||||||
|
U4* OtBase;
|
||||||
|
};
|
||||||
|
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)
|
||||||
|
, atom_reads(R_TapePtr)
|
||||||
|
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
|
||||||
|
){
|
||||||
|
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||||
|
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
|
||||||
|
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
|
||||||
|
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
|
||||||
|
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
|
||||||
|
internal
|
||||||
|
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||||
|
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
||||||
|
atom_writes(R_PrimCursor, R_FaceCursor)
|
||||||
|
){
|
||||||
|
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||||
|
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
||||||
|
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
||||||
|
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
||||||
|
|
||||||
|
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
||||||
|
nop2, gte_cmdw_rotate_translate_perspective_triple, // required cpu -> gte delay slot
|
||||||
|
gte_cmdw_nclip,
|
||||||
|
|
||||||
|
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
|
||||||
|
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
|
||||||
|
/* BD-slot: write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
|
||||||
|
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
|
||||||
|
* harmless because the OT entry that points to this prim is created later, only on the body path. */
|
||||||
|
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
||||||
|
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
||||||
|
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
|
||||||
|
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||||
|
|
||||||
|
mac_gte_store_g4_p012(R_PrimCursor),
|
||||||
|
gte_cmdw_rotate_translate_perspective_single,
|
||||||
|
mac_gte_store_g4_p3(R_PrimCursor),
|
||||||
|
|
||||||
|
gte_cmdw_avg_sort_z4,
|
||||||
|
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||||
|
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||||
|
set_lt_u( R_AT, R_T1, R_AT),
|
||||||
|
|
||||||
|
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
|
||||||
|
mac_insert_ot_tag_g4(R_OtBase, R_PrimCursor),
|
||||||
|
mac_format_g4_color(R_PrimCursor,
|
||||||
|
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
||||||
|
/* c1 yellow */ 0xFF, 0xFF, 0x00,
|
||||||
|
/* c2 cyan */ 0x00, 0xFF, 0xFF,
|
||||||
|
/* c3 green */ 0x00, 0xFF, 0x00),
|
||||||
|
// end: branch(bounds_chk)
|
||||||
|
// end: branch(cull)
|
||||||
|
|
||||||
|
atom_label(cube_g4_face_exit)
|
||||||
|
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
|
||||||
|
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef Struct_(Binds_FloorTri) {
|
||||||
|
U4 PrimCursor;
|
||||||
|
V3_S2* FaceCursor;
|
||||||
|
V3_S2* VertBase;
|
||||||
|
U4* OtBase;
|
||||||
|
};
|
||||||
|
internal
|
||||||
|
MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(floor_f3)
|
||||||
|
, atom_reads(R_TapePtr)
|
||||||
|
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
|
||||||
|
){
|
||||||
|
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||||
|
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
|
||||||
|
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
|
||||||
|
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
|
||||||
|
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
// atom_dbg_skip
|
||||||
|
internal
|
||||||
|
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
||||||
|
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||||
|
, atom_writes(R_PrimCursor, R_FaceCursor)
|
||||||
|
) {
|
||||||
|
mac_load_tri_indices(R_FaceCursor, R_T0, R_T1, R_T2),
|
||||||
|
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
||||||
|
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
|
||||||
|
gte_cmdw_nclip,
|
||||||
|
|
||||||
|
/* Culling (Branch forward if Backface) */
|
||||||
|
gte_mv_from_data_r(R_T0, C2_MAC0),
|
||||||
|
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
|
||||||
|
/* Format Primitive */
|
||||||
|
mac_gte_store_f3(R_PrimCursor),
|
||||||
|
|
||||||
|
/* Calculate Depth */
|
||||||
|
gte_avg_sort_z3,
|
||||||
|
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||||
|
/* Bounds Check OTZ < 2048 (Branch forward to skip insertion) */
|
||||||
|
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||||
|
set_lt_u( R_AT, R_T1, R_AT),
|
||||||
|
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
|
||||||
|
mac_format_f3_color(R_PrimCursor, 0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
|
||||||
|
mac_insert_ot_tag_f3(R_OtBase, R_PrimCursor), /* Insert into Ordering Table Linked List */
|
||||||
|
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
|
||||||
|
// Note(Ed): No bounds checking, should be checked before atom runs.
|
||||||
|
// end: branch(bounds_chk)
|
||||||
|
// end: branch(culling)
|
||||||
|
|
||||||
|
/* Advance Input Cursor & Yield (Both branch targets land here) */
|
||||||
|
atom_label(floor_f3_face_exit)
|
||||||
|
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
|
||||||
|
internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitiveArena)
|
||||||
|
, atom_reads( R_TapePtr, R_PrimCursor)
|
||||||
|
, atom_writes(R_TapePtr)
|
||||||
|
){
|
||||||
|
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
|
||||||
|
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
|
||||||
|
/* Calculate byte offset and store directly back to RAM */
|
||||||
|
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
|
||||||
|
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
/* ----- pad_bios_snapshot -----
|
||||||
|
* Per-frame snapshot of one BIOS pad buffer into PadState.
|
||||||
|
* Decoder (branch ladder on raw[0] status + raw[1] id):
|
||||||
|
* 1. raw[0] == 0xFF -> Disconnected (buttons=0, axes=0x80)
|
||||||
|
* 2. raw[0]==0 && raw[1]==0 -> Pending (buttons=0, axes=0x80)
|
||||||
|
* 3. raw[1] == 0x41 -> Digital (buttons normalized; axes=0x80)
|
||||||
|
* 4. raw[1] == 0x53 -> AnalogStick (buttons normalized; axes from raw[4..7])
|
||||||
|
* 5. raw[1] in 0x7x -> AnalogPad (buttons normalized; axes from raw[4..7])
|
||||||
|
* 6. else -> Unsupported (buttons=0, axes=0x80)
|
||||||
|
*
|
||||||
|
* Buttons normalization: byte_swap16((~raw_buttons) & 0xFFFF).
|
||||||
|
* raw_buttons = load_half_u(raw, 2) = raw[2] | (raw[3] << 8).
|
||||||
|
* byte_swap16(x) = (x >> 8) | (x << 8); nor(x, R_0) = ~x. store_half truncates to 16 bits so the upper-16 mask is implicit in the store.
|
||||||
|
*
|
||||||
|
* Register use (atom-local; no wave-context touched):
|
||||||
|
* R_T0 = raw base (kept throughout; axes loads read raw[4..7] from R_T0)
|
||||||
|
* R_T1 = state base (kept throughout; all stores go through R_T1)
|
||||||
|
* R_T2 = raw[0] status (alive across the disc/pending/id dispatch, then dead)
|
||||||
|
* R_T3 = raw[1] id (alive across the id dispatch, then dead)
|
||||||
|
* R_T4 = scratch (shifts, compares, immediate loads, store values)
|
||||||
|
* R_T5 = scratch (parallel lui+ori for the 0x80808080 axes constant + byte-swap target)
|
||||||
|
*/
|
||||||
|
enum {
|
||||||
|
R_PadRaw = R_T0 atom_reg atom_type(U1),
|
||||||
|
R_PadState = R_T1 atom_reg,
|
||||||
|
R_RawStatus = R_T2 atom_reg,
|
||||||
|
R_RawId = R_T3 atom_reg,
|
||||||
|
};
|
||||||
|
typedef Struct_(Binds_PadBiosSnapshot) {
|
||||||
|
PadBiosRaw* raw;
|
||||||
|
PadState* state;
|
||||||
|
};
|
||||||
|
internal MipsAtom_(pad_bios_snapshot) atom_info(atom_bind(Binds_PadBiosSnapshot)
|
||||||
|
, atom_reads( R_PadRaw, R_PadState, R_RawStatus, R_RawId, R_T4, R_T5, R_TapePtr)
|
||||||
|
, atom_writes(R_PadRaw, R_PadState, R_RawStatus, R_RawId, R_T4, R_T5, R_TapePtr)
|
||||||
|
) {
|
||||||
|
/* === Bind consumption: T0 = raw, T1 = state, advance R_TapePtr by 8. */
|
||||||
|
load_word(R_PadRaw, R_TapePtr, O_(Binds_PadBiosSnapshot,raw)),
|
||||||
|
load_word(R_PadState, R_TapePtr, O_(Binds_PadBiosSnapshot,state)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_PadBiosSnapshot)),
|
||||||
|
|
||||||
|
/* === Read raw[0] (status) + raw[1] (id) */
|
||||||
|
load_byte_u(R_RawStatus, R_PadRaw, 0),
|
||||||
|
load_byte_u(R_RawId, R_PadRaw, 1),
|
||||||
|
|
||||||
|
atom_label(snap_root) /* === Case 1: Disconnected (status == 0xFF). */
|
||||||
|
add_ui(R_T4, R_0, 0xFF), branch_ne(R_RawStatus, R_T4, atom_offset(snap_root, skip_disconnected)),
|
||||||
|
/* BD-slot: pre-compute PadStatus_Disconnected. Branch reads R_T4=0xFF in EX before this WB completes.
|
||||||
|
* If branch NOT taken (fall through to pending/id_dispatch), R_T4 is overwritten by the next case body's add_ui — harmless. */
|
||||||
|
|
||||||
|
atom_label(disconnected) /* === Disconnected body. */
|
||||||
|
/* R_T4 = PadStatus_Disconnected from snap_root BD-slot. */
|
||||||
|
store_word(R_T4, R_PadState, O_(PadState,status)),
|
||||||
|
store_half(R_0, R_PadState, O_(PadState,buttons)),
|
||||||
|
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||||
|
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
|
||||||
|
store_word( R_T4, R_PadState, O_(PadState,left_x)),
|
||||||
|
store_byte( R_RawId, R_PadState, O_(PadState,id)),
|
||||||
|
jump_rel(atom_offset(disconnected, snap_end)),
|
||||||
|
/* BD-slot: load next atom's entry point (replaces the nop).
|
||||||
|
* The unconditional branch always jumps to snap_end, where mac_yield_tail()
|
||||||
|
* transfers control to R_AtomJmp without re-loading it. */
|
||||||
|
mac_yield_load(),
|
||||||
|
atom_label(skip_disconnected)
|
||||||
|
|
||||||
|
/* === Case 2: Pending (status == 0 && id == 0)
|
||||||
|
* Combined check: if (status | id) != 0 then skip to id_dispatch.
|
||||||
|
* Falls through to the Pending case only when both are zero. */
|
||||||
|
or_u_self(R_RawStatus, R_RawId), branch_ne(R_RawStatus, R_0, atom_offset(case_2, id_dispatch)),
|
||||||
|
/* BD-slot: pre-compute PadStatus_Pending. Branch reads R_RawStatus in EX before this WB completes.
|
||||||
|
* If branch NOT taken (fall through to id_dispatch), R_T4 is overwritten by the digital/analog body add_ui — harmless. */
|
||||||
|
|
||||||
|
atom_label(pending) /* === Pending body */
|
||||||
|
/* R_T4 = PadStatus_Pending from case_2 BD-slot. */
|
||||||
|
store_word(R_T4, R_PadState, O_(PadState,status)),
|
||||||
|
store_half(R_0, R_PadState, O_(PadState,buttons)),
|
||||||
|
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||||
|
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
|
||||||
|
store_word( R_T4, R_PadState, O_(PadState,left_x)),
|
||||||
|
store_byte( R_RawId, R_PadState, O_(PadState,id)),
|
||||||
|
jump_rel(atom_offset(pending, snap_end)),
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(id_dispatch) /* === Case 3-6: ID dispatch */
|
||||||
|
add_ui(R_T4, R_0, 0x41), branch_ne(R_RawId, R_T4, atom_offset(id_dispatch, try_analog_stick)),
|
||||||
|
/* BD-slot: pre-compute PadStatus_Digital. Branch reads R_RawId in EX before this WB completes.
|
||||||
|
* If branch NOT taken (fall through to try_analog_stick), R_T4 is overwritten by the analog body add_ui. */
|
||||||
|
|
||||||
|
/* === Digital body (status, buttons normalize, axes=0x80, id, branch. */
|
||||||
|
/* R_T4 = PadStatus_Digital from id_dispatch BD-slot. */
|
||||||
|
store_word( R_T4, R_PadState, O_(PadState,status)),
|
||||||
|
load_half_u(R_T4, R_PadRaw, 2 * S_(U1)),
|
||||||
|
/* Fill R_T4's load-delay slot with the 0x80808080 axes constant into R_T5
|
||||||
|
* (R_T5 is dead on this path; it's only consumed at the analog_pad range check). */
|
||||||
|
load_upper_i(R_T5, 0x8080), or_i_self(R_T5, 0x8080),
|
||||||
|
nor_u( R_T4, R_T4, R_0), /* raw_buttons is already in host bit order; no swap needed */
|
||||||
|
store_half( R_T4, R_PadState, O_(PadState,buttons)),
|
||||||
|
|
||||||
|
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||||
|
store_word( R_T5, R_PadState, O_(PadState,left_x)),
|
||||||
|
add_ui( R_T4, R_0, 0x41),
|
||||||
|
store_byte( R_T4, R_PadState, O_(PadState,id)),
|
||||||
|
|
||||||
|
jump_rel(atom_offset(id_dispatch, snap_end)),
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(try_analog_stick) /* === Case 4: AnalogStick (id == 0x53)*/
|
||||||
|
add_ui(R_T4, R_0, 0x53), branch_ne(R_RawId, R_T4, atom_offset(try_analog_stick, try_analog_pad)),
|
||||||
|
/* BD-slot: pre-compute PadStatus_AnalogStick. Branch reads R_RawId in EX before this WB completes.
|
||||||
|
* If branch NOT taken (fall through to try_analog_pad), R_T4 is overwritten by the analog_pad body add_ui. */
|
||||||
|
|
||||||
|
atom_label(analog_stick) /* === AnalogStick body
|
||||||
|
* Axes are loaded as two halfwords: raw[6..7] → left_xy (sh at offset 8), raw[4..5] → right_xy (sh at offset 10).
|
||||||
|
* R_T5 holds left_xy / id-value in turn (it's dead on this path — only consumed at the analog_pad range check). */
|
||||||
|
/* R_T4 = PadStatus_AnalogStick from try_analog_stick BD-slot. */
|
||||||
|
store_word( R_T4, R_PadState, O_(PadState,status)),
|
||||||
|
load_half_u( R_T4, R_PadRaw, 2 * S_(U1)), /* R_T4 = raw_buttons */
|
||||||
|
load_half_u( R_T5, R_PadRaw, 6 * S_(U1)), /* R_T5 = left_xy; fills R_T4's load-delay slot (doesn't read R_T4) */
|
||||||
|
nor_u( R_T4, R_T4, R_0), /* R_T4 = ~raw_buttons */
|
||||||
|
store_half( R_T4, R_PadState, O_(PadState,buttons)),
|
||||||
|
load_half_u( R_T4, R_PadRaw, 4 * S_(U1)), /* R_T4 = right_xy; fills R_T5's load-delay slot */
|
||||||
|
store_half( R_T5, R_PadState, O_(PadState,left_x)), /* R_T5 settled, store left_xy */
|
||||||
|
store_half( R_T4, R_PadState, O_(PadState,right_x)),
|
||||||
|
add_ui( R_T5, R_0, 0x53), /* R_T5 = id value (clobbers left_xy, already stored) */
|
||||||
|
store_byte( R_T5, R_PadState, O_(PadState,id)),
|
||||||
|
jump_rel(atom_offset(analog_stick, snap_end)),
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(try_analog_pad) /* === Case 5-6: AnalogPad (id & 0xF0 == 0x70) */
|
||||||
|
and_i( R_T4, R_RawId, 0xF0),
|
||||||
|
add_ui( R_T5, R_0, 0x70),
|
||||||
|
branch_ne(R_T4, R_T5, atom_offset(try_analog_pad, try_unsupported)),
|
||||||
|
/* BD-slot: pre-compute PadStatus_AnalogPad. Branch reads R_T4 in EX before this WB completes.
|
||||||
|
* If branch NOT taken (fall through to try_unsupported), R_T4 is overwritten by the unsupported body add_ui. */
|
||||||
|
|
||||||
|
atom_label(analog_pad) /* === AnalogPad body
|
||||||
|
* Same shape as AnalogStick with AnalogPad status. R_T5 holds left_xy (it's dead on this path). */
|
||||||
|
/* R_T4 = PadStatus_AnalogPad from try_analog_pad BD-slot. */
|
||||||
|
store_word( R_T4, R_PadState, O_(PadState,status)),
|
||||||
|
load_half_u(R_T4, R_PadRaw, 2 * S_(U1)), /* R_T4 = raw_buttons */
|
||||||
|
load_half_u(R_T5, R_PadRaw, 6 * S_(U1)), /* R_T5 = left_xy; fills R_T4's load-delay slot */
|
||||||
|
nor_u( R_T4, R_T4, R_0), /* R_T4 = ~raw_buttons */
|
||||||
|
store_half( R_T4, R_PadState, O_(PadState,buttons)),
|
||||||
|
load_half_u(R_T4, R_PadRaw, 4 * S_(U1)), /* R_T4 = right_xy; fills R_T5's load-delay slot */
|
||||||
|
store_half( R_T5, R_PadState, O_(PadState,left_x)), /* R_T5 settled, store left_xy */
|
||||||
|
store_half( R_T4, R_PadState, O_(PadState,right_x)),
|
||||||
|
store_byte( R_RawId, R_PadState, O_(PadState,id)),
|
||||||
|
|
||||||
|
jump_rel(atom_offset(analog_pad, snap_end)),
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(try_unsupported) /* === Case 7: Unsupported — fall through from the AnalogPad range-check miss. */
|
||||||
|
add_ui( R_T4, R_0, PadStatus_Unsupported),
|
||||||
|
store_word(R_T4, R_PadState, O_(PadState,status)),
|
||||||
|
store_half(R_0, R_PadState, O_(PadState,buttons)),
|
||||||
|
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||||
|
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
|
||||||
|
store_word( R_T4, R_PadState, O_(PadState,left_x)),
|
||||||
|
add_ui( R_T4, R_0, 0xFF), /* 0xFF sentinel: "unknown id" */
|
||||||
|
store_byte( R_T4, R_PadState, O_(PadState,id)),
|
||||||
|
/* Fall through to snap_end. */
|
||||||
|
|
||||||
|
atom_label(no_jump_fallthrough)
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(snap_end)
|
||||||
|
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the case-exit branch. */
|
||||||
|
mac_yield_tail(),
|
||||||
|
};
|
||||||
|
|
||||||
|
/* ----- pad_apply_input -----
|
||||||
|
* Reads pad[0].buttons + pad[0].left_x;
|
||||||
|
* Applies the input-semantics deltas to cube_rot.y + floor_rot.y:
|
||||||
|
* - D-pad Left: cube_rot.y += 30, floor_rot.y += 5
|
||||||
|
* - D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5
|
||||||
|
* - Analog stick X (dead zone 0x70..0x90):
|
||||||
|
* cube delta = (0x80 - left_x) >> 2 (range approx -32..+32)
|
||||||
|
* floor delta = (0x80 - left_x) >> 5 (range approx -4..+4)
|
||||||
|
* - D-pad + analog deltas add when used together.
|
||||||
|
*
|
||||||
|
* Convention:
|
||||||
|
* pad_state = 0 means no buttons active.
|
||||||
|
* The fail-safe zero-button value flows through unchanged, so a disconnected/fresh pad produces no rotation.
|
||||||
|
* The branch_le_zero pattern below matches the existing pad_input_demo convention (atom body lines 248/257).
|
||||||
|
*
|
||||||
|
* Signed-delta trick:
|
||||||
|
* load_byte_u zero-extends left_x to 32 bits; sub_u from 0x80 wraps to a SIGNED two's-complement value in the negative range;
|
||||||
|
* shift_aright (sra) then correctly sign-extends the shift for both positive (left_x < 0x80) and negative (left_x > 0x80) cases.
|
||||||
|
* Digital pads publish left_x = 0x80 → delta = 0 → no rotation, so the analog step is naturally a no-op for digital controllers.
|
||||||
|
*/
|
||||||
|
typedef Struct_(Binds_PadApplyInput) {
|
||||||
|
PadState* state;
|
||||||
|
V3_S2* cube_rot;
|
||||||
|
V3_S2* floor_rot;
|
||||||
|
};
|
||||||
|
enum {
|
||||||
|
R_PadStateT5 = R_T5 atom_reg,
|
||||||
|
R_CubeRot = R_T1 atom_reg,
|
||||||
|
R_FloorRot = R_T2 atom_reg,
|
||||||
|
};
|
||||||
|
internal MipsAtom_(pad_apply_input) atom_info(atom_bind(Binds_PadApplyInput)
|
||||||
|
, atom_reads(R_T0, R_CubeRot, R_FloorRot, R_T3, R_T4, R_PadStateT5, R_TapePtr)
|
||||||
|
, atom_writes( R_CubeRot, R_FloorRot)
|
||||||
|
) {
|
||||||
|
/* Pop Binds from tape (state, cube_rot, floor_rot) */
|
||||||
|
load_word(R_PadStateT5, R_TapePtr, O_(Binds_PadApplyInput,state)),
|
||||||
|
load_word(R_CubeRot, R_TapePtr, O_(Binds_PadApplyInput,cube_rot)),
|
||||||
|
load_word(R_FloorRot, R_TapePtr, O_(Binds_PadApplyInput,floor_rot)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_PadApplyInput)),
|
||||||
|
|
||||||
|
/* Load pad[0].buttons into R_T0. */
|
||||||
|
load_word(R_T0, R_PadStateT5, O_(PadState,buttons)), nop,
|
||||||
|
// Note(Ed): Potential op with delay slot?
|
||||||
|
|
||||||
|
/* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */
|
||||||
|
and_i(R_T3, R_T0, pad0_(Pad_Left)), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)),
|
||||||
|
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
|
||||||
|
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
add_si( R_T4, R_T4, 30),
|
||||||
|
add_si( R_T3, R_T3, 5),
|
||||||
|
store_half(R_T4, R_CubeRot, O_(V3_S2,y)),
|
||||||
|
store_half(R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
atom_label(exit_dpad_left)
|
||||||
|
|
||||||
|
/* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */
|
||||||
|
and_i(R_T3, R_T0, pad0_(Pad_Right)), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)),
|
||||||
|
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
|
||||||
|
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
add_si( R_T4, R_T4, -30),
|
||||||
|
add_si( R_T3, R_T3, -5),
|
||||||
|
store_half(R_T4, R_CubeRot, O_(V3_S2,y)),
|
||||||
|
store_half(R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
atom_label(exit_dpad_right)
|
||||||
|
|
||||||
|
/* Analog left-stick X: dead zone 0x70..0x90.
|
||||||
|
* Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */
|
||||||
|
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left_x)),
|
||||||
|
|
||||||
|
/* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly).
|
||||||
|
* set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */
|
||||||
|
add_ui(R_T4, R_0, 0x70), set_lt_u(R_T4, R_T3, R_T4), branch_ne(R_T4, R_0, atom_offset(dead_zone_low_check, dead_low_active)),
|
||||||
|
add_ui(R_T4, R_0, 0x80), /* BD-slot: pre-load 0x80 for dead_low_active */
|
||||||
|
|
||||||
|
atom_label(dead_check_upper)
|
||||||
|
/* left_x >= 0x70 → check upper bound. */
|
||||||
|
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left_x)), /* reload */
|
||||||
|
add_ui( R_T4, R_0, 0x90),
|
||||||
|
|
||||||
|
/* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */
|
||||||
|
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)),
|
||||||
|
add_ui( R_T4, R_0, 0x80), /* BD-slot: pre-load 0x80 for dead_high_active */
|
||||||
|
jump_rel(atom_offset(dead_zone_skip, exit_stick)),
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(dead_low_active)
|
||||||
|
/* R_T3 = left_x (from line 632 lbu; not clobbered between dead_zone_low_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
||||||
|
* The earlier `load_byte_u(R_T3, ...)` reload was redundant and introduced a load-use hazard on the next `sub_u`.
|
||||||
|
* R_T4 = 0x80 from the BD-slot of `dead_zone_low_check`'s branch_ne. */
|
||||||
|
sub_u( R_T3, R_T4, R_T3), /* R_T3 = 0x80 - left_x */
|
||||||
|
/* delta = 0x80 - left_x (positive). */
|
||||||
|
|
||||||
|
/* R_T4 = cube_delta */
|
||||||
|
shift_aright(R_T4, R_T3, 2),
|
||||||
|
load_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||||
|
nop,
|
||||||
|
add_u( R_T0, R_T0, R_T4),
|
||||||
|
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||||
|
/* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap;
|
||||||
|
* doesn't read R_T0; R_T4 settles by the subsequent add_u). */
|
||||||
|
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
shift_aright(R_T4, R_T3, 5),
|
||||||
|
add_u( R_T0, R_T0, R_T4),
|
||||||
|
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
|
||||||
|
jump_rel(atom_offset(end_low, exit_stick)),
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(dead_high_active)
|
||||||
|
/* R_T3 = left_x (from line 641 lbu in dead_check_upper; not clobbered between dead_zone_high_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
||||||
|
* The earlier `load_byte_u(R_T3, ...)` reload was redundant and introduced a load-use hazard on the next `sub_u`.
|
||||||
|
* R_T4 = 0x80 from the BD-slot of `dead_zone_high_check`'s branch_ne. */
|
||||||
|
sub_u( R_T3, R_T4, R_T3),
|
||||||
|
/* delta = 0x80 - left_x (signed negative). */
|
||||||
|
|
||||||
|
shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */
|
||||||
|
load_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||||
|
nop,
|
||||||
|
add_u( R_T0, R_T0, R_T4),
|
||||||
|
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||||
|
|
||||||
|
/* R_T4 = floor_delta (signed) — moved into the load-delay slot of the floor load below. */
|
||||||
|
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
shift_aright(R_T4, R_T3, 5),
|
||||||
|
add_u( R_T0, R_T0, R_T4),
|
||||||
|
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
|
||||||
|
atom_label(no_jump_fallthrough)
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(exit_stick)
|
||||||
|
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the dead-zone/exit branch. */
|
||||||
|
mac_yield_tail(),
|
||||||
|
};
|
||||||
|
|
||||||
|
#pragma endregion Baked Atoms
|
||||||
@@ -0,0 +1,441 @@
|
|||||||
|
#pragma region Vendors
|
||||||
|
#include <stdio.h>
|
||||||
|
#include <stdlib.h>
|
||||||
|
#include <assert.h>
|
||||||
|
// #include "libgpu.h"
|
||||||
|
// #include "libetc.h"
|
||||||
|
// #include "libgte.h"
|
||||||
|
#pragma endregion Vendors
|
||||||
|
|
||||||
|
#pragma region Duffle Headers
|
||||||
|
# include "duffle/gen/macs.h"
|
||||||
|
# include "duffle/gen/offsets.h"
|
||||||
|
|
||||||
|
#include "duffle/word_count.metadata.h"
|
||||||
|
|
||||||
|
#include "duffle/dsl.h"
|
||||||
|
#include "duffle/memory.h"
|
||||||
|
#include "duffle/math.h"
|
||||||
|
|
||||||
|
#include "duffle/gcc_asm.h"
|
||||||
|
#include "duffle/mips.h"
|
||||||
|
#include "duffle/gp.h"
|
||||||
|
#include "duffle/gte.h"
|
||||||
|
#include "duffle/pad.h"
|
||||||
|
|
||||||
|
#include "duffle/dsl.atom.h"
|
||||||
|
#include "duffle/lottes_tape.h"
|
||||||
|
|
||||||
|
#include "duffle/psyq.h"
|
||||||
|
#pragma endregion Duffle Headers
|
||||||
|
|
||||||
|
#pragma region Duffle TUs
|
||||||
|
#include "duffle/math.atom.c"
|
||||||
|
#include "duffle/mips.atom.c"
|
||||||
|
#include "duffle/gte.atom.c"
|
||||||
|
#include "duffle/gp.atom.c"
|
||||||
|
#include "duffle/psyq.atom.c"
|
||||||
|
#pragma endregion Duffle TUs
|
||||||
|
|
||||||
|
#pragma region Joypade Headers
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
|
||||||
|
#include "hello_joypad.h"
|
||||||
|
#pragma region Joypad Headers
|
||||||
|
|
||||||
|
#pragma region Hello Joypad TUs
|
||||||
|
#include "hello_joypad.atom.c"
|
||||||
|
#pragma endregion Hello Joypad TUs
|
||||||
|
|
||||||
|
enum {
|
||||||
|
Scratchpad_Len = 1024,
|
||||||
|
MemTape_Len = 512,
|
||||||
|
};
|
||||||
|
typedef Struct_(SMemory) {
|
||||||
|
PrimitiveArena primitives;
|
||||||
|
A2_OrderingTable_Buffer ordering_tbl;
|
||||||
|
DoubleBuffer screen_buf;
|
||||||
|
S4 active_buf_id;
|
||||||
|
|
||||||
|
U4 MemTape[MemTape_Len];
|
||||||
|
|
||||||
|
M3_S2 tform_world;
|
||||||
|
|
||||||
|
Ent_Cube cube;
|
||||||
|
Ent_Floor floor;
|
||||||
|
|
||||||
|
PadBiosRaw pad_raw[2];
|
||||||
|
PadState pad[2];
|
||||||
|
|
||||||
|
U4_V scratchpad; // d-cache
|
||||||
|
};
|
||||||
|
global SMemory smem;
|
||||||
|
extern SMemory smem;
|
||||||
|
|
||||||
|
I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
|
||||||
|
gknown PrimitiveArena* pa = & smem.primitives;
|
||||||
|
gknown B1* buf = (B1*) r_(smem.primitives.buf)[smem.active_buf_id];
|
||||||
|
assert(pa->used + type_width < PrimitiveBuff_Len);
|
||||||
|
B1* next = buf + pa->used;
|
||||||
|
pa->used += type_width;
|
||||||
|
return next;
|
||||||
|
}
|
||||||
|
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
|
||||||
|
|
||||||
|
/* Uses ONE 8-byte frame allocated via the compiler's standard prologue.
|
||||||
|
* The 4 wasted-arg words for B(12h) InitPAD2 live at [SP+0..15] but are not explicitly allocated.
|
||||||
|
* The compiler handles the MIPS O32 "wasted stack" convention for us by treating the B-call as a 4-arg call.
|
||||||
|
*
|
||||||
|
* The buffer pointers are passed as arguments so the compiler keeps them in callee-saved registers;
|
||||||
|
* The B(12h) asm volatile block does NOT clobber those registers (it clobbers only the volatile GPRs + the B-table arg registers explicitly).
|
||||||
|
* The C-level writes after the call re-load the pointers from their callee-saved homes.
|
||||||
|
*
|
||||||
|
* The clobber list for both B-calls names the full BIOS destroy set documented in kernelbios.md:167-174 (R1..R15, R24..R25, R31, HI/LO).
|
||||||
|
* The kernel-ABI "volatile GPRs" subset is clb_system; the rest of the destroy set is enumerated explicitly here. */
|
||||||
|
NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
|
||||||
|
{
|
||||||
|
/* Pin raw0 + raw1 to $a0 + $a1 via rgcc; the B(12h) call uses these directly.
|
||||||
|
* The `(void)` casts mark them as unread after the call so the compiler doesn't need to move them back. */
|
||||||
|
register PadBiosRaw* p0 rgcc(R_A0) = raw0;
|
||||||
|
register PadBiosRaw* p1 rgcc(R_A1) = raw1;
|
||||||
|
(void)p0; (void)p1;
|
||||||
|
|
||||||
|
// TODO(Ed): Properly annotate the raw values in the inline asm instructions.
|
||||||
|
// Use enums.
|
||||||
|
|
||||||
|
/* B(12h) InitPAD2(raw0, 0x22, raw1, 0x22)
|
||||||
|
* $a0 = raw0 (rgcc-bound; survives the sequence below)
|
||||||
|
* $a1 = raw1 (preserved into $a2 before $a1 is overwritten)
|
||||||
|
* $a2 = raw1 (moved from $a1; survives $a1's overwrite)
|
||||||
|
* $a3 = 0x22 (immediate)
|
||||||
|
* $t1 = 0x12 (function number)
|
||||||
|
* $t2 = 0xB0 (BIOS B-table address) */
|
||||||
|
asm volatile(
|
||||||
|
asm_words(
|
||||||
|
or_u( rarg_2, rarg_1, rdiscard), /* $a2 = $a1 = raw1 */
|
||||||
|
add_ui( rarg_1, rdiscard, 0x22), /* $a1 = 0x22 */
|
||||||
|
add_ui( rarg_3, rdiscard, 0x22), /* $a3 = 0x22 */
|
||||||
|
add_ui( rtmp_1, rdiscard, 0x12), /* $t1 = 0x12 */
|
||||||
|
add_ui( rtmp_2, rdiscard, 0xB0), /* $t2 = 0xB0 */
|
||||||
|
call_reg(rtmp_2), /* jalr $t2, $ra */
|
||||||
|
nop /* BD slot */
|
||||||
|
)
|
||||||
|
asm_rpins, r_use(p0), r_use(p1)
|
||||||
|
asm_clobber:
|
||||||
|
rlit(R_AT),
|
||||||
|
rlit(R_V0), rlit(R_V1),
|
||||||
|
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||||
|
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
||||||
|
rlit(R_RA),
|
||||||
|
clb_mem_drain
|
||||||
|
);
|
||||||
|
|
||||||
|
/* The C-level writes re-load the pointers via the parameter names and write 0xFF to each
|
||||||
|
* buffer's status byte to mark the initial-state hazard documented in kernelbios.md:1621-1624. */
|
||||||
|
u1_v(raw0)[0] = 0xFF;
|
||||||
|
u1_v(raw1)[0] = 0xFF;
|
||||||
|
|
||||||
|
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */
|
||||||
|
asm volatile(
|
||||||
|
asm_words(
|
||||||
|
add_ui( rtmp_1, rdiscard, 0x13), /* $t1 = 0x13 */
|
||||||
|
add_ui( rtmp_2, rdiscard, 0xB0), /* $t2 = 0xB0 (re-load) */
|
||||||
|
call_reg(rtmp_2), /* jalr $t2, $ra */
|
||||||
|
nop /* BD slot */
|
||||||
|
)
|
||||||
|
asm_clobber:
|
||||||
|
rlit(R_AT),
|
||||||
|
rlit(R_V0), rlit(R_V1),
|
||||||
|
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||||
|
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
||||||
|
rlit(R_RA),
|
||||||
|
clb_mem_drain
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
GCC_OPTIMIZATION_DISABLE
|
||||||
|
void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||||
|
{
|
||||||
|
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
|
||||||
|
|
||||||
|
if (0) // Pad Input (dead — kept for the source-as-written record; references the deleted `pad_state` field)
|
||||||
|
{
|
||||||
|
(void)Pad_Left; (void)Pad_Right; /* suppress unused-token warnings */
|
||||||
|
if (false) {
|
||||||
|
smem.cube.rot.y += 30;
|
||||||
|
smem.floor.rot.y += 5;
|
||||||
|
}
|
||||||
|
if (false) {
|
||||||
|
smem.cube.rot.y -= 30;
|
||||||
|
smem.floor.rot.y -= 5;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (1) // Pad Input (Tape version)
|
||||||
|
{
|
||||||
|
tb.used = 0; tb_scope_run(& tb) {
|
||||||
|
/* BIOS-owned polling: per-frame snapshot of both ports. */
|
||||||
|
tb_emit_(pad_bios_snapshot);
|
||||||
|
tb_data_(raw, & smem.pad_raw[0]);
|
||||||
|
tb_data_(state, & smem.pad[0]);
|
||||||
|
tb_emit_(pad_bios_snapshot);
|
||||||
|
tb_data_(raw, & smem.pad_raw[1]);
|
||||||
|
tb_data_(state, & smem.pad[1]);
|
||||||
|
/* Per-frame rotation apply: consume pad[0].buttons + pad[0].left_x */
|
||||||
|
tb_emit_(pad_apply_input);
|
||||||
|
tb_data_(state, & smem.pad[0]);
|
||||||
|
tb_data_(cube_rot, & smem.cube.rot);
|
||||||
|
tb_data_(floor_rot, & smem.floor.rot);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
orderingtbl_clear_reverse(ordering_buf, OrderingTbl_Len);
|
||||||
|
|
||||||
|
// Update the position based on acceleration and velocity
|
||||||
|
gknown V3_S4_R pos = & smem.cube.pos;
|
||||||
|
gknown V3_S4_R vel = & smem.cube.vel;
|
||||||
|
gknown V3_S4_R acc = & smem.cube.accel;
|
||||||
|
add_v3s4(vel, acc[0]);
|
||||||
|
add_v3s4_fp(pos, vel[0]);
|
||||||
|
// vel->x += acc->x;
|
||||||
|
// vel->y += acc->y;
|
||||||
|
// vel->z += acc->z;
|
||||||
|
// pos->x += vel->x;
|
||||||
|
// pos->y += vel->y;
|
||||||
|
// pos->z += vel->z;
|
||||||
|
|
||||||
|
if (pos->y + 150 > smem.floor.pos.y) vel->y *= -1;
|
||||||
|
|
||||||
|
// Prep
|
||||||
|
S4 nclip = 0;
|
||||||
|
S4 orderingtbl_z = 0;
|
||||||
|
A2_S2 p; //???
|
||||||
|
S4 flag; //????
|
||||||
|
|
||||||
|
|
||||||
|
// Draw Cube
|
||||||
|
if (0)
|
||||||
|
{
|
||||||
|
m3s2_rotation (& smem.cube.rot, & smem.tform_world);
|
||||||
|
m3s2_translation(& smem.tform_world, & smem.cube.pos);
|
||||||
|
m3s2_scale (& smem.tform_world, & smem.cube.scale);
|
||||||
|
// gte_matrix_set_rotation (& smem.tform_world);
|
||||||
|
gte_matrix_set_translation(& smem.tform_world);
|
||||||
|
for (U4 face_id = 0; face_id < Cube_num_faces; face_id += 1)
|
||||||
|
{
|
||||||
|
Poly_G4* quad = prim_alloc(Poly_G4); set_poly_g4(quad);
|
||||||
|
quad->c0 = rgb8(255, 0, 255);
|
||||||
|
quad->c1 = rgb8(255, 255, 0);
|
||||||
|
quad->c2 = rgb8( 0, 255, 255);
|
||||||
|
quad->c3 = rgb8( 0, 255, 0);
|
||||||
|
|
||||||
|
V4_S2* face = & smem.cube.faces[face_id];
|
||||||
|
V3_S2* p0 = & smem.cube.verts[face->x];
|
||||||
|
V3_S2* p1 = & smem.cube.verts[face->y];
|
||||||
|
V3_S2* p2 = & smem.cube.verts[face->z];
|
||||||
|
V3_S2* p3 = & smem.cube.verts[face->w];
|
||||||
|
|
||||||
|
nclip = rtp_avg_nclip_a4_v3s2(
|
||||||
|
p0, p1, p2, p3,
|
||||||
|
& quad->p0, & quad->p1, & quad->p2, & quad->p3,
|
||||||
|
& p, & orderingtbl_z, & flag
|
||||||
|
);
|
||||||
|
if (nclip <= 0) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
if ((orderingtbl_z > 0) && (orderingtbl_z < OrderingTbl_Len)) {
|
||||||
|
orderingtbl_add_primitive(ordering_buf[orderingtbl_z], quad);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// smem.cube.rot.x += 6;
|
||||||
|
// smem.cube.rot.y += 8;
|
||||||
|
// smem.cube.rot.z += 12;
|
||||||
|
smem.cube.rot.y += 30;
|
||||||
|
}
|
||||||
|
// Draw cube (tape method) - two triangles per face
|
||||||
|
if (1)
|
||||||
|
{
|
||||||
|
m3s2_rotation (& smem.cube.rot, & smem.tform_world);
|
||||||
|
m3s2_translation(& smem.tform_world, & smem.cube.pos);
|
||||||
|
m3s2_scale (& smem.tform_world, & smem.cube.scale);
|
||||||
|
gte_matrix_set_rotation (& smem.tform_world);
|
||||||
|
gte_matrix_set_translation(& smem.tform_world);
|
||||||
|
|
||||||
|
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||||
|
U4 prim_cursor = prim_base + pa->used;
|
||||||
|
|
||||||
|
tb.used = 0; tb_scope(& tb) {
|
||||||
|
tb_emit(& tb, rbind_cube_g4_face);
|
||||||
|
tb_data(& tb, prim_cursor);
|
||||||
|
tb_data(& tb, u4_(smem.cube.faces));
|
||||||
|
tb_data(& tb, u4_(smem.cube.verts));
|
||||||
|
tb_data(& tb, u4_(ordering_buf));
|
||||||
|
|
||||||
|
for (U4 i = 0; i < Cube_num_faces; i++) {
|
||||||
|
// Two triangles per quad face: (x,y,z) and (x,z,w)
|
||||||
|
tb_emit(& tb, cube_g4_face);
|
||||||
|
}
|
||||||
|
|
||||||
|
tb_emit(& tb, sync_primitive_arena);
|
||||||
|
tb_data(& tb, u4_(& pa->used));
|
||||||
|
tb_data(& tb, prim_base);
|
||||||
|
}
|
||||||
|
tape_run(tb_slice(tb));
|
||||||
|
|
||||||
|
// smem.cube.rot.y += 30;
|
||||||
|
}
|
||||||
|
// Draw Floor
|
||||||
|
if (0)
|
||||||
|
{
|
||||||
|
m3s2_rotation (& smem.floor.rot, & smem.tform_world);
|
||||||
|
m3s2_translation(& smem.tform_world, & smem.floor.pos);
|
||||||
|
m3s2_scale (& smem.tform_world, & smem.floor.scale);
|
||||||
|
gte_matrix_set_rotation (& smem.tform_world);
|
||||||
|
gte_matrix_set_translation(& smem.tform_world);
|
||||||
|
for (U4 face_id = 0; face_id < Floor_num_faces; face_id += 1)
|
||||||
|
{
|
||||||
|
Poly_F3* tri = prim_alloc(Poly_F3); set_poly_f3(tri);
|
||||||
|
tri->color = rgb8(255, 255, 255);
|
||||||
|
|
||||||
|
V3_S2* face = & smem.floor.faces[face_id];
|
||||||
|
register V3_S2* p0 rgcc(R_T4) = & smem.floor.verts[face->x];
|
||||||
|
register V3_S2* p1 rgcc(R_T5) = & smem.floor.verts[face->y];
|
||||||
|
register V3_S2* p2 rgcc(R_T6) = & smem.floor.verts[face->z];
|
||||||
|
|
||||||
|
gte_load_v0(p0, R_T4);
|
||||||
|
/*
|
||||||
|
asm volatile( ".word " "%0" ", %1" : :
|
||||||
|
"i"(((op_lwc2 & OPCODE_MASK) << OPCODE_SHIFT) | ((R_T4 & REG_MASK) << RS_SHIFT) | ((gte_in_v0_xy & REG_MASK) << RT_SHIFT) | (0 & IMM_MASK)),
|
||||||
|
"i"(((op_lwc2 & OPCODE_MASK) << OPCODE_SHIFT) | ((R_T4 & REG_MASK) << RS_SHIFT) | ((gte_in_v0_z & REG_MASK) << RT_SHIFT) | (GTE_Z_Offset & IMM_MASK)),
|
||||||
|
"r"(p0) :
|
||||||
|
"$2", "$8", "$9", "$31", "memory"
|
||||||
|
);
|
||||||
|
*/
|
||||||
|
gte_load_v1(p1, R_T5);
|
||||||
|
gte_load_v2(p2, R_T6);
|
||||||
|
|
||||||
|
gte_rtpt();
|
||||||
|
gte_nclip();
|
||||||
|
gte_stotz(& nclip);
|
||||||
|
|
||||||
|
// nclip = rtp_avg_nclip_a3_v3s2(p0, p1, p2
|
||||||
|
// , & tri->p0, & tri->p1, & tri->p2
|
||||||
|
// , & p, & orderingtbl_z, & flag
|
||||||
|
// );
|
||||||
|
// if (nclip <= 0) {
|
||||||
|
// continue;
|
||||||
|
// }
|
||||||
|
|
||||||
|
if (nclip > 0 ) {
|
||||||
|
gte_stsxy3(& tri->p0, & tri->p1, & tri->p2);
|
||||||
|
gte_avsz3();
|
||||||
|
gte_stotz(& orderingtbl_z);
|
||||||
|
|
||||||
|
if ((orderingtbl_z > 0) && (orderingtbl_z < OrderingTbl_Len)) {
|
||||||
|
orderingtbl_add_primitive(ordering_buf[orderingtbl_z], tri);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
smem.floor.rot.y += 5;
|
||||||
|
}
|
||||||
|
// Draw floor tape method
|
||||||
|
if (1)
|
||||||
|
{
|
||||||
|
m3s2_rotation (& smem.floor.rot, & smem.tform_world);
|
||||||
|
m3s2_translation(& smem.tform_world, & smem.floor.pos);
|
||||||
|
m3s2_scale (& smem.tform_world, & smem.floor.scale);
|
||||||
|
|
||||||
|
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||||
|
U4 prim_cursor = prim_base + pa->used;
|
||||||
|
|
||||||
|
// TODO(Ed): We should do a bounds check beforehand to confirm pa can hold all tris?
|
||||||
|
// The tape atoms in-flight should not need to care.
|
||||||
|
|
||||||
|
// Prepare the tape. (Push protocol to tape)
|
||||||
|
tb.used = 0; tb_scope(& tb) {
|
||||||
|
tb_emit(& tb, set_gte_world);
|
||||||
|
tb_data(& tb, u4_(& smem.tform_world));
|
||||||
|
|
||||||
|
tb_emit(& tb, rbind_floor_f3_face);
|
||||||
|
// TODO(Ed): Just use a single context struct ref
|
||||||
|
tb_data(& tb, prim_cursor);
|
||||||
|
tb_data(& tb, u4_(smem.floor.faces));
|
||||||
|
tb_data(& tb, u4_(smem.floor.verts));
|
||||||
|
tb_data(& tb, u4_(ordering_buf));
|
||||||
|
for (U4 i = 0; i < Floor_num_faces; i++) {
|
||||||
|
tb_emit(& tb, floor_f3_face);
|
||||||
|
}
|
||||||
|
// After floor_f3_face iterations complete, the primitive arena's used counter needs updating.
|
||||||
|
tb_emit(& tb, sync_primitive_arena);
|
||||||
|
tb_data(& tb, u4_(& pa->used));
|
||||||
|
tb_data(& tb, prim_base);
|
||||||
|
}
|
||||||
|
tape_run(tb_slice(tb));// Fire off the tape.
|
||||||
|
|
||||||
|
// C-side state (pa->used) has already been updated by the tape!
|
||||||
|
// smem.floor.rot.y += 5;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
GCC_OPTIMIZATION_ENABLE
|
||||||
|
|
||||||
|
void render(void) {
|
||||||
|
}
|
||||||
|
|
||||||
|
void gp_display_frame(DoubleBuffer* screen_buf, S4* active_buf_id, U4* ordering_buf, PrimitiveArena* pa) {
|
||||||
|
draw_sync(0);
|
||||||
|
vsync(0);
|
||||||
|
displayenv_put(& r_(screen_buf->display)[active_buf_id[0] ]);
|
||||||
|
drawenv_put (& r_(screen_buf->draw) [active_buf_id[0] ]);
|
||||||
|
{
|
||||||
|
draw_orderingtbl(ordering_buf + OrderingTbl_Len - 1);
|
||||||
|
pa->used = 0;
|
||||||
|
}
|
||||||
|
active_buf_id[0] = ! active_buf_id[0]; // Swap current buffer
|
||||||
|
}
|
||||||
|
|
||||||
|
GCC_OPTIMIZATION_DISABLE
|
||||||
|
int main(void)
|
||||||
|
{
|
||||||
|
smem = (SMemory){0};
|
||||||
|
smem.scratchpad = C_(U4_V, 0x1F800000);
|
||||||
|
// smem.primitives.used = 0;
|
||||||
|
// smem.active_buf_id = 0;
|
||||||
|
/*Persistent Entity Setup*/{
|
||||||
|
ent_cube128_init(& smem.cube.verts, & smem.cube.faces); {
|
||||||
|
Ent_Cube* cube = & smem.cube;
|
||||||
|
cube->rot = v3s2(0, 0, 0);
|
||||||
|
cube->scale = v3s4_fp_one();
|
||||||
|
cube->accel = v3s4(0, 1, 0);
|
||||||
|
cube->pos = v3s4(0, -400, 1800);
|
||||||
|
}
|
||||||
|
ent_floor_init(& smem.floor.verts, & smem.floor.faces); {
|
||||||
|
Ent_Floor* floor = & smem.floor;
|
||||||
|
floor->rot = v3s2(0, 0, 0);
|
||||||
|
floor->pos = v3s4(0, 450, 1800);
|
||||||
|
floor->scale = v3s4_fp_one();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape)); {
|
||||||
|
reset_graph(0);
|
||||||
|
/* Direct BIOS: poll both ports during VBlank. */
|
||||||
|
pad_bios_init_start(& smem.pad_raw[0], & smem.pad_raw[1]);
|
||||||
|
/* Pinned registers for the GPU init atom. */
|
||||||
|
register U4* io_base_addr rgcc(R_IO_BaseAddr) = u4_r(IO_BASE_ADDR);
|
||||||
|
register DoubleBuffer* screen_buf rgcc(R_ScreenBuf) = & smem.screen_buf;
|
||||||
|
tb.used = 0; tb_scope_run(& tb) {
|
||||||
|
tb_emit(& tb, screen_env_init);
|
||||||
|
tb_emit(& tb, gp_screen_init);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
while (1) {
|
||||||
|
gknown S4* active_buf_id = & smem.active_buf_id;
|
||||||
|
gknown U4* ordering_buf = r_(smem.ordering_tbl)[active_buf_id[0]];
|
||||||
|
gknown PrimitiveArena* pa = & smem.primitives;
|
||||||
|
update(pa, ordering_buf);
|
||||||
|
render();
|
||||||
|
gp_display_frame(& smem.screen_buf, active_buf_id, ordering_buf, pa);
|
||||||
|
};
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
GCC_OPTIMIZATION_ENABLE
|
||||||
@@ -0,0 +1,102 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# pragma once
|
||||||
|
# include "duffle/dsl.h"
|
||||||
|
# include "duffle/math.h"
|
||||||
|
# include "duffle/gp.h"
|
||||||
|
# include "duffle/pad.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
enum {
|
||||||
|
// PrimitiveBuff_Len = 4096,
|
||||||
|
// OrderingTbl_Len = 2048,
|
||||||
|
PrimitiveBuff_Len = 131072,
|
||||||
|
OrderingTbl_Len = 8192,
|
||||||
|
};
|
||||||
|
|
||||||
|
enum {
|
||||||
|
ScreenRes_X = 320,
|
||||||
|
ScreenRes_Y = 240,
|
||||||
|
ScreenZ = 320,
|
||||||
|
ScreenRes_CenterX = (ScreenRes_X >> 1),
|
||||||
|
ScreenRes_CenterY = (ScreenRes_Y >> 1),
|
||||||
|
};
|
||||||
|
|
||||||
|
enum {
|
||||||
|
fp_one = (1 << 12),
|
||||||
|
};
|
||||||
|
|
||||||
|
#define v3s4_fp_one() v3s4(fp_one, fp_one, fp_one)
|
||||||
|
|
||||||
|
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
||||||
|
typedef Array_(OrderingTable_Buffer, 2);
|
||||||
|
|
||||||
|
typedef B1 PrimitiveBuffer[PrimitiveBuff_Len];
|
||||||
|
typedef Array_(PrimitiveBuffer, 2);
|
||||||
|
typedef Struct_(PrimitiveArena) {
|
||||||
|
A2_PrimitiveBuffer buf;
|
||||||
|
U4 used;
|
||||||
|
};
|
||||||
|
|
||||||
|
#define Cube_num_verts 8
|
||||||
|
typedef Array_(V3_S2, Cube_num_verts);
|
||||||
|
#define Cube_num_faces 6
|
||||||
|
typedef Array_(V4_S2, Cube_num_faces);
|
||||||
|
I_ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
||||||
|
LP_ A8_V3_S2 baked_verts = (A8_V3_S2) {
|
||||||
|
{ -128, -128, -128 },
|
||||||
|
{ 128, -128, -128 },
|
||||||
|
{ 128, -128, 128 },
|
||||||
|
{ -128, -128, 128 },
|
||||||
|
{ -128, 128, -128 },
|
||||||
|
{ 128, 128, -128 },
|
||||||
|
{ 128, 128, 128 },
|
||||||
|
{ -128, 128, 128 }
|
||||||
|
};
|
||||||
|
LP_ A6_V4_S2 baked_faces = (A6_V4_S2) {
|
||||||
|
{ 3, 2, 0, 1 },
|
||||||
|
{ 0, 1, 4, 5 },
|
||||||
|
{ 4, 5, 7, 6 },
|
||||||
|
{ 1, 2, 5, 6 },
|
||||||
|
{ 2, 3, 6, 7 },
|
||||||
|
{ 3, 0, 7, 4 },
|
||||||
|
};
|
||||||
|
mem_copy(u4_(verts), u4_(& baked_verts), S_(A8_V3_S2) );
|
||||||
|
mem_copy(u4_(faces), u4_(& baked_faces), S_(A6_V4_S2) );
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
typedef Struct_(Ent_Cube) {
|
||||||
|
V3_S4 accel;
|
||||||
|
V3_S4 vel;
|
||||||
|
V3_S4 pos;
|
||||||
|
V3_S4 scale;
|
||||||
|
V3_S2 rot;
|
||||||
|
A8_V3_S2 verts;
|
||||||
|
A6_V4_S2 faces;
|
||||||
|
};
|
||||||
|
|
||||||
|
#define Floor_num_verts 4
|
||||||
|
typedef Array_(V3_S2, Floor_num_verts);
|
||||||
|
#define Floor_num_faces 2
|
||||||
|
typedef Array_(V3_S2, Floor_num_faces);
|
||||||
|
I_ void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
|
||||||
|
LP_ A4_V3_S2 baked_verts = (A4_V3_S2) {
|
||||||
|
{ -900, 0, -900 },
|
||||||
|
{ -900, 0, 900 },
|
||||||
|
{ 900, 0, -900 },
|
||||||
|
{ 900, 0, 900 },
|
||||||
|
};
|
||||||
|
LP_ A2_V3_S2 baked_faces = (A2_V3_S2) {
|
||||||
|
{ 0, 1, 2 },
|
||||||
|
{ 1, 3, 2 },
|
||||||
|
};
|
||||||
|
mem_copy(u4_(verts), u4_(& baked_verts), S_(A4_V3_S2));
|
||||||
|
mem_copy(u4_(faces), u4_(& baked_faces), S_(A2_V3_S2));
|
||||||
|
};
|
||||||
|
typedef Struct_(Ent_Floor) {
|
||||||
|
V3_S4 accel;
|
||||||
|
V3_S4 pos;
|
||||||
|
V3_S4 scale;
|
||||||
|
V3_S2 rot;
|
||||||
|
A4_V3_S2 verts;
|
||||||
|
A2_V3_S2 faces;
|
||||||
|
};
|
||||||
@@ -0,0 +1,659 @@
|
|||||||
|
|
||||||
|
#if 0 /* ac_pad_sio_write_pad_state — superseded by pad_bios_snapshot */
|
||||||
|
|
||||||
|
/* ============================================================
|
||||||
|
* raw_sio_pad_poll_20260802 — superseded by bios_pad_buffer_snapshot_20260803.
|
||||||
|
* The doomed raw-SIO production atoms (ac_pad_sio_write_pad_state,
|
||||||
|
* pad_sio_init, pad_sio_step, pad_sio_diag_pin, pad_sio_diag_byte_exchange)
|
||||||
|
* reference symbols that were removed from code/duffle/pad.h during
|
||||||
|
* Phase 1. Each is wrapped in a narrow `#if 0` so the C compile skips
|
||||||
|
* the body while the source-as-written text stays in place for the
|
||||||
|
* Phase 5.1 deletion pass. The wrap is removed (and the bodies are
|
||||||
|
* deleted) by Phase 5.1 of this track.
|
||||||
|
* ============================================================ */
|
||||||
|
|
||||||
|
* Writes the per-port PadState in 5 instructions plus 4 store_word calls (status,
|
||||||
|
* buttons, left_x/y/right_x/right_y packed, attempt). The provisional decode publishes
|
||||||
|
* 0x0000FFFF buttons + centered axes on every path until response-byte decode lands.
|
||||||
|
*
|
||||||
|
* Args:
|
||||||
|
* status_val - the PadSioStatus enum value to publish
|
||||||
|
* state_ptr_reg - the PadState* base (R_PadState at the call site)
|
||||||
|
* scratch_reg - scratch register for the value being stored (e.g., R_T0)
|
||||||
|
*
|
||||||
|
* Emits 9 instructions (status/buttons/axes/attempt stores plus the
|
||||||
|
* two-instruction zero-extended buttons load).
|
||||||
|
*/
|
||||||
|
FI_ Slice_MipsCode ac_pad_sio_write_pad_state(MipsAtomBuilder_R ab, U4 status_val, U4 state_ptr_reg, U4 scratch_reg)
|
||||||
|
MipsAtomComp_Proc_(ac_pad_sio_write_pad_state, ab, {
|
||||||
|
add_ui(scratch_reg, R_0, status_val),
|
||||||
|
store_word(scratch_reg, state_ptr_reg, O_(PadState,status)),
|
||||||
|
/* FIX 2026-08-02: buttons = 0x0000FFFF = "no buttons pressed" in
|
||||||
|
* libetc convention. Build it with LUI + ORI so addiu does not
|
||||||
|
* sign-extend 0xFFFF to 0xFFFFFFFF. */
|
||||||
|
load_upper_i(scratch_reg, 0x0000),
|
||||||
|
or_i(scratch_reg, scratch_reg, 0xFFFF),
|
||||||
|
store_word(scratch_reg, state_ptr_reg, O_(PadState,buttons)),
|
||||||
|
add_ui(scratch_reg, R_0, 0x80808080),
|
||||||
|
store_word(scratch_reg, state_ptr_reg, O_(PadState,left_x)),
|
||||||
|
add_ui(scratch_reg, R_0, 0),
|
||||||
|
store_word(scratch_reg, state_ptr_reg, O_(PadState,attempt))
|
||||||
|
})
|
||||||
|
#endif /* end ac_pad_sio_write_pad_state wrap */
|
||||||
|
|
||||||
|
/* ----- pad_sio_init -----
|
||||||
|
* Boot-time SIO0 init. Caller pins R_T6 = sio_base_addr0.
|
||||||
|
* Issues SIO CTRL=0x0040 (reset), MODE=0x000D, BAUD=0x0088.
|
||||||
|
* (Phase 2 fills the body.)
|
||||||
|
*/
|
||||||
|
#if 0 /* pad_sio_init — superseded by pad_bios_init_start (Phase 1.3) */
|
||||||
|
internal MipsAtom_(pad_sio_init) atom_info(atom_phase(pad_init)
|
||||||
|
, atom_reads(R_T5, R_T6)
|
||||||
|
, atom_writes(R_T5, R_T6)
|
||||||
|
) {
|
||||||
|
/* FIX 2026-08-02: explicitly load the KSEG1 base into R_T6 at the top of
|
||||||
|
* the atom body. The rgcc(R_PadSioBase) binding in main() pins R_T6 = base
|
||||||
|
* when main() runs, but $12 is caller-saved per the O32 ABI — when tape_run
|
||||||
|
* is invoked, R_T6 is fair game. The atom body cannot rely on the value. */
|
||||||
|
load_upper_i(R_T6, pad_IO_KSEG1_BASE >> 16), /* R_T6 high 16 = 0xBF80 */
|
||||||
|
or_i(R_T6, R_T6, pad_IO_KSEG1_BASE & 0xFFFF), /* R_T6 = 0xBF800000 */
|
||||||
|
|
||||||
|
/* SIO CTRL = 0x0040 (reset) */
|
||||||
|
add_ui(R_T5, R_0, pad_SIO_CTRL_RESET),
|
||||||
|
store_half(R_T5, R_T6, pad_SIO_CTRL_OFFSET),
|
||||||
|
/* SIO MODE = 0x000D (MUL1, 8-bit, no parity, idle-high) */
|
||||||
|
add_ui(R_T5, R_0, pad_SIO_MODE_INIT),
|
||||||
|
store_half(R_T5, R_T6, pad_SIO_MODE_OFFSET),
|
||||||
|
/* SIO BAUD = 0x0088 (~250 kHz) */
|
||||||
|
add_ui(R_T5, R_0, pad_SIO_BAUD_INIT),
|
||||||
|
store_half(R_T5, R_T6, pad_SIO_BAUD_OFFSET),
|
||||||
|
mac_yield(),
|
||||||
|
};
|
||||||
|
#endif /* end pad_sio_init wrap */
|
||||||
|
|
||||||
|
/* ----- pad_sio_step -----
|
||||||
|
* Per-frame bounded raw-SIO transaction. Reads PadState pointers + SIO
|
||||||
|
* base addresses from Binds_PadSioStep; writes per-port status +
|
||||||
|
* buttons + axes into smem.pad[0..1].
|
||||||
|
* Body shape (per spec §"Transaction model (per port, per pad_sio_step)"):
|
||||||
|
* port 0: CTRL=CLEANUP → settle → CTRL=port-select → settle → exchange 5
|
||||||
|
* bytes (addr + 0x42 0x00 0x00 0x00) → decode → write PadState[0]
|
||||||
|
* → CTRL=CLEANUP.
|
||||||
|
* port 1: swap scratch regs (sio_base_addr1 → R_PadSioBase, state1 →
|
||||||
|
* R_PadState) → mirror port 0 sequence.
|
||||||
|
*
|
||||||
|
* Bounded-loop semantics: every countdown is wrapped in
|
||||||
|
* add_ui_self(R_T1, -1) + branch_ne(R_T1, R_0, ...)
|
||||||
|
* with a known maximum (pad_SIO_SETTLE_BEFORE_TX=1000, pad_SIO_SETTLE_AFTER_TX=2000,
|
||||||
|
* pad_SIO_WAIT_BUDGET=4096). The static-analysis pass currently reports
|
||||||
|
* has_loops = true; the follow-up metaprogram track that learns modeled-bounded
|
||||||
|
* loops is out of scope here (per spec §"Risks").
|
||||||
|
*
|
||||||
|
* Scratch register strategy:
|
||||||
|
* R_PadStatus = R_T4 — RESERVED for port-1 swap (holds state1)
|
||||||
|
* R_PadCountdown = R_T5 — RESERVED for port-1 swap (holds sio_base_addr1)
|
||||||
|
* R_T0 — byte-exchange value + STAT read (clobbered freely)
|
||||||
|
* R_T1 — countdown budget (clobbered freely)
|
||||||
|
* R_PadState = R_T7 — PadState* (preserved for PadState writes)
|
||||||
|
* R_PadSioBase = R_T6 — SIO base (preserved through the port)
|
||||||
|
*
|
||||||
|
* Response decode (Task 3.1 teaching scope):
|
||||||
|
* - status = PadSioStatus_Digital (hardcoded)
|
||||||
|
* - buttons = 0xFFFF (no buttons pressed in the provisional libetc
|
||||||
|
* convention; full response-byte decode is follow-up)
|
||||||
|
* - axes = 0x80808080 (centered: left_x=0x80, left_y=0x80,
|
||||||
|
* right_x=0x80, right_y=0x80)
|
||||||
|
* - attempt = 0
|
||||||
|
* - DualShock handshake (0x43 0x01 → 0x44 0x01 0x03 → 0x43 0x00) is
|
||||||
|
* follow-up scope; the hardcoded digital decode is a placeholder.
|
||||||
|
*
|
||||||
|
* Both ports raise /CS (CTRL = pad_SIO_CTRL_CLEANUP) before exit. Both ports
|
||||||
|
* treat response timeout as PadSioStatus_Disconnected per the spec §"Failure
|
||||||
|
* handling" + the canonical per-port timeout semantics.
|
||||||
|
*/
|
||||||
|
#if 0 /* pad_sio_step — superseded by pad_bios_snapshot (Phase 2.1) */
|
||||||
|
internal MipsAtom_(pad_sio_step) atom_info(atom_bind(Binds_PadSioStep)
|
||||||
|
, atom_reads(R_TapePtr, R_PadSioBase, R_PadState, R_PadStatus, R_PadCountdown)
|
||||||
|
, atom_writes(R_PadStatus, R_PadCountdown)
|
||||||
|
) {
|
||||||
|
/* FIX 2026-08-02: explicitly load KSEG1 base into R_PadSioBase (R_T6) at the
|
||||||
|
* top. The rgcc() binding in main() does NOT survive the tape_run call
|
||||||
|
* because R_T6 is caller-saved per the O32 ABI. The pad_sio_init atom
|
||||||
|
* (also in the per-frame tape) reloads R_T6 separately. */
|
||||||
|
load_upper_i(R_PadSioBase, pad_IO_KSEG1_BASE >> 16),
|
||||||
|
or_i(R_PadSioBase, R_PadSioBase, pad_IO_KSEG1_BASE & 0xFFFF),
|
||||||
|
|
||||||
|
/* Pop Binds from tape (in Binds_PadSioStep declaration order) */
|
||||||
|
load_word(R_PadState, R_TapePtr, O_(Binds_PadSioStep,state0)),
|
||||||
|
load_word(R_PadStatus, R_TapePtr, O_(Binds_PadSioStep,state1)), /* reserved for port-1 swap */
|
||||||
|
load_word(R_PadSioBase, R_TapePtr, O_(Binds_PadSioStep,sio_base_addr0)),
|
||||||
|
load_word(R_PadCountdown, R_TapePtr, O_(Binds_PadSioStep,sio_base_addr1)), /* reserved for port-1 swap */
|
||||||
|
add_ui_self(R_TapePtr, S_(Binds_PadSioStep)),
|
||||||
|
|
||||||
|
/* ============== PORT 0 TRANSACTION ============== */
|
||||||
|
/* Use R_T0 (byte value / STAT read) + R_T1 (countdown) as scratch.
|
||||||
|
* R_PadStatus (state1) + R_PadCountdown (sio_base_addr1) are preserved
|
||||||
|
* through the port-0 body and swapped into R_PadSioBase + R_PadState
|
||||||
|
* at atom_offset(port1_start, ...) below. */
|
||||||
|
|
||||||
|
/* 1. Cleanup: CTRL = 0x0010 (raise /CS, clear stale status) */
|
||||||
|
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
|
||||||
|
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||||
|
/* Bounded by pad_SIO_SETTLE_BEFORE_TX = 1000 iterations. */
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_SETTLE_BEFORE_TX),
|
||||||
|
atom_label(settle_pre_port0)
|
||||||
|
nop, /* BD slot */
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(settle_pre_port0, settle_pre_port0)),
|
||||||
|
|
||||||
|
/* 2. Port-select: CTRL = 0x0003 (TX enable + DTR /CS) for port 0 */
|
||||||
|
add_ui(R_T0, R_0, pad_SIO_CTRL_TX_ENABLE),
|
||||||
|
or_i(R_T0, R_T0, pad_SIO_CTRL_DTR_CS), /* set /CS line low */
|
||||||
|
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||||
|
/* Bounded by pad_SIO_SETTLE_AFTER_TX = 2000 iterations. */
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_SETTLE_AFTER_TX),
|
||||||
|
atom_label(settle_post_port0)
|
||||||
|
nop,
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(settle_post_port0, settle_post_port0)),
|
||||||
|
|
||||||
|
/* 3. Address byte (0x01) — send + RX-ready wait + read response + RX-drain confirmation */
|
||||||
|
add_ui(R_T0, R_0, pad_PROTO_ADDR),
|
||||||
|
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
|
||||||
|
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(wait_ack0_port0)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_ne(R_T0, R_0, atom_offset(wait_ack0_port0, ack0_received_port0)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
atom_label(continue_wait_ack0_port0)
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack0_port0, wait_ack0_port0)),
|
||||||
|
/* RX timeout → mark disconnected; skip to port 1 */
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||||
|
atom_label(skip_port0_from_ack0)
|
||||||
|
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack0, port1_start)),
|
||||||
|
|
||||||
|
atom_label(ack0_received_port0)
|
||||||
|
/* Read open-bus response byte 0 — discard per docs/psx-spx §controllersandmemorycards.md */
|
||||||
|
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
|
||||||
|
/* Confirm RX FIFO drained before sending byte 1. Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(wait_ackrel0_port0)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_equal(R_T0, R_0, atom_offset(wait_ackrel0_port0, ack_released_port0)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
atom_label(continue_wait_ackrel0_port0)
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel0_port0, wait_ackrel0_port0)),
|
||||||
|
/* RX-drain timeout → disconnected; skip to port 1 */
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||||
|
atom_label(skip_port0_from_ackrel0)
|
||||||
|
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel0, port1_start)),
|
||||||
|
|
||||||
|
atom_label(ack_released_port0)
|
||||||
|
|
||||||
|
/* === Byte 1 (port 0): send 0x42 (cmd read) + RX-ready wait + read response + RX-drain confirmation === */
|
||||||
|
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||||
|
add_ui(R_T0, R_0, pad_PROTO_CMD_READ),
|
||||||
|
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(wait_ack1_port0)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_ne(R_T0, R_0, atom_offset(wait_ack1_port0, ack1_received_port0)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
atom_label(continue_wait_ack1_port0)
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack1_port0, wait_ack1_port0)),
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||||
|
atom_label(skip_port0_from_ack1)
|
||||||
|
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack1, port1_start)),
|
||||||
|
|
||||||
|
atom_label(ack1_received_port0)
|
||||||
|
/* Read response ID byte — discarded for teaching scope (decode hardcoded). */
|
||||||
|
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
|
||||||
|
/* RX FIFO drain wait. Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(wait_ackrel1_port0)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_equal(R_T0, R_0, atom_offset(wait_ackrel1_port0, ack_released1_port0)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
atom_label(continue_wait_ackrel1_port0)
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel1_port0, wait_ackrel1_port0)),
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||||
|
atom_label(skip_port0_from_ackrel1)
|
||||||
|
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel1, port1_start)),
|
||||||
|
|
||||||
|
atom_label(ack_released1_port0)
|
||||||
|
|
||||||
|
/* === Byte 2 (port 0): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
|
||||||
|
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||||
|
add_ui(R_T0, R_0, 0x00),
|
||||||
|
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(wait_ack2_port0)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_ne(R_T0, R_0, atom_offset(wait_ack2_port0, ack2_received_port0)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
atom_label(continue_wait_ack2_port0)
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack2_port0, wait_ack2_port0)),
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||||
|
atom_label(skip_port0_from_ack2)
|
||||||
|
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack2, port1_start)),
|
||||||
|
|
||||||
|
atom_label(ack2_received_port0)
|
||||||
|
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
|
||||||
|
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(wait_ackrel2_port0)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_equal(R_T0, R_0, atom_offset(wait_ackrel2_port0, ack_released2_port0)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
atom_label(continue_wait_ackrel2_port0)
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel2_port0, wait_ackrel2_port0)),
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||||
|
atom_label(skip_port0_from_ackrel2)
|
||||||
|
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel2, port1_start)),
|
||||||
|
|
||||||
|
atom_label(ack_released2_port0)
|
||||||
|
|
||||||
|
/* === Byte 3 (port 0): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
|
||||||
|
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||||
|
add_ui(R_T0, R_0, 0x00),
|
||||||
|
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(wait_ack3_port0)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_ne(R_T0, R_0, atom_offset(wait_ack3_port0, ack3_received_port0)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
atom_label(continue_wait_ack3_port0)
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack3_port0, wait_ack3_port0)),
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||||
|
atom_label(skip_port0_from_ack3)
|
||||||
|
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack3, port1_start)),
|
||||||
|
|
||||||
|
atom_label(ack3_received_port0)
|
||||||
|
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
|
||||||
|
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(wait_ackrel3_port0)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_equal(R_T0, R_0, atom_offset(wait_ackrel3_port0, ack_released3_port0)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
atom_label(continue_wait_ackrel3_port0)
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel3_port0, wait_ackrel3_port0)),
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||||
|
atom_label(skip_port0_from_ackrel3)
|
||||||
|
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel3, port1_start)),
|
||||||
|
|
||||||
|
atom_label(ack_released3_port0)
|
||||||
|
|
||||||
|
/* === Byte 4 (FINAL, port 0): send 0x00 + RX-not-empty wait + read final byte === */
|
||||||
|
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||||
|
add_ui(R_T0, R_0, 0x00),
|
||||||
|
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(wait_rx4_port0)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_ne(R_T0, R_0, atom_offset(wait_rx4_port0, rx4_received_port0)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
atom_label(continue_wait_rx4_port0)
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(continue_wait_rx4_port0, wait_rx4_port0)),
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||||
|
atom_label(skip_port0_from_rx4)
|
||||||
|
branch_equal(R_0, R_0, atom_offset(skip_port0_from_rx4, port1_start)),
|
||||||
|
|
||||||
|
atom_label(rx4_received_port0)
|
||||||
|
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET), /* discard final byte */
|
||||||
|
|
||||||
|
/* === RESPONSE DECODE (hardcoded for teaching scope) ===
|
||||||
|
* Per the plan §"Phase 3 task 3.1" + spec §"Architecture":
|
||||||
|
* - Full decode (buttons/axes from response bytes) is follow-up scope.
|
||||||
|
* - Teaching scope: hardcode digital poll response.
|
||||||
|
* status = PadSioStatus_Digital
|
||||||
|
* buttons = 0x0000FFFF (no buttons pressed — placeholder)
|
||||||
|
* axes = 0x80808080 (left_x=0x80, left_y=0x80, right_x=0x80, right_y=0x80)
|
||||||
|
* attempt = 0
|
||||||
|
*/
|
||||||
|
atom_label(decode_port0)
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Digital, R_PadState, R_T0),
|
||||||
|
|
||||||
|
/* /CS cleanup: raise /CS, clear stale status before exiting port 0. */
|
||||||
|
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
|
||||||
|
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||||
|
|
||||||
|
/* ============== PORT 1 SETUP ============== */
|
||||||
|
/* Swap: R_PadCountdown holds sio_base_addr1; R_PadStatus holds state1. */
|
||||||
|
atom_label(port1_start)
|
||||||
|
add_u(R_PadSioBase, R_0, R_PadCountdown), /* sio_base_addr1 → R_PadSioBase */
|
||||||
|
add_u(R_PadState, R_0, R_PadStatus), /* state1 → R_PadState */
|
||||||
|
|
||||||
|
/* ============== PORT 1 TRANSACTION (mirror of port 0) ============== */
|
||||||
|
/* R_PadStatus + R_PadCountdown are no longer reserved (port 1 is the
|
||||||
|
* last transaction); we still use R_T0/R_T1 as scratch to match port 0. */
|
||||||
|
|
||||||
|
/* 1. Cleanup: CTRL = 0x0010 (raise /CS, clear stale status) */
|
||||||
|
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
|
||||||
|
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||||
|
/* Bounded by pad_SIO_SETTLE_BEFORE_TX = 1000 iterations. */
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_SETTLE_BEFORE_TX),
|
||||||
|
atom_label(settle_pre_port1)
|
||||||
|
nop,
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(settle_pre_port1, settle_pre_port1)),
|
||||||
|
|
||||||
|
/* 2. Port-select: CTRL = 0x0003 | (1 << 13) (port 1 select) */
|
||||||
|
add_ui(R_T0, R_0, pad_SIO_CTRL_TX_ENABLE),
|
||||||
|
or_i(R_T0, R_T0, pad_SIO_CTRL_DTR_CS),
|
||||||
|
or_i(R_T0, R_T0, 1 << 13), /* port 1 select bit (CTRL bit 13 = port select) */
|
||||||
|
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||||
|
/* Bounded by pad_SIO_SETTLE_AFTER_TX = 2000 iterations. */
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_SETTLE_AFTER_TX),
|
||||||
|
atom_label(settle_post_port1)
|
||||||
|
nop,
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(settle_post_port1, settle_post_port1)),
|
||||||
|
|
||||||
|
/* 3. Address byte (0x01) — send + RX-ready wait + read response + RX-drain confirmation */
|
||||||
|
add_ui(R_T0, R_0, pad_PROTO_ADDR),
|
||||||
|
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
|
||||||
|
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(wait_ack0_port1)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_ne(R_T0, R_0, atom_offset(wait_ack0_port1, ack0_received_port1)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
atom_label(continue_wait_ack0_port1)
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack0_port1, wait_ack0_port1)),
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||||
|
atom_label(skip_port1_from_ack0)
|
||||||
|
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack0, end_atom)),
|
||||||
|
|
||||||
|
atom_label(ack0_received_port1)
|
||||||
|
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
|
||||||
|
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(wait_ackrel0_port1)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_equal(R_T0, R_0, atom_offset(wait_ackrel0_port1, ack_released_port1)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
atom_label(continue_wait_ackrel0_port1)
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel0_port1, wait_ackrel0_port1)),
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||||
|
atom_label(skip_port1_from_ackrel0)
|
||||||
|
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel0, end_atom)),
|
||||||
|
|
||||||
|
atom_label(ack_released_port1)
|
||||||
|
|
||||||
|
/* === Byte 1 (port 1): send 0x42 (cmd read) + RX-ready wait + read response + RX-drain confirmation === */
|
||||||
|
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||||
|
add_ui(R_T0, R_0, pad_PROTO_CMD_READ),
|
||||||
|
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(wait_ack1_port1)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_ne(R_T0, R_0, atom_offset(wait_ack1_port1, ack1_received_port1)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
atom_label(continue_wait_ack1_port1)
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack1_port1, wait_ack1_port1)),
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||||
|
atom_label(skip_port1_from_ack1)
|
||||||
|
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack1, end_atom)),
|
||||||
|
|
||||||
|
atom_label(ack1_received_port1)
|
||||||
|
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
|
||||||
|
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(wait_ackrel1_port1)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_equal(R_T0, R_0, atom_offset(wait_ackrel1_port1, ack_released1_port1)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
atom_label(continue_wait_ackrel1_port1)
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel1_port1, wait_ackrel1_port1)),
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||||
|
atom_label(skip_port1_from_ackrel1)
|
||||||
|
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel1, end_atom)),
|
||||||
|
|
||||||
|
atom_label(ack_released1_port1)
|
||||||
|
|
||||||
|
/* === Byte 2 (port 1): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
|
||||||
|
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||||
|
add_ui(R_T0, R_0, 0x00),
|
||||||
|
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(wait_ack2_port1)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_ne(R_T0, R_0, atom_offset(wait_ack2_port1, ack2_received_port1)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
atom_label(continue_wait_ack2_port1)
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack2_port1, wait_ack2_port1)),
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||||
|
atom_label(skip_port1_from_ack2)
|
||||||
|
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack2, end_atom)),
|
||||||
|
|
||||||
|
atom_label(ack2_received_port1)
|
||||||
|
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
|
||||||
|
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(wait_ackrel2_port1)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_equal(R_T0, R_0, atom_offset(wait_ackrel2_port1, ack_released2_port1)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
atom_label(continue_wait_ackrel2_port1)
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel2_port1, wait_ackrel2_port1)),
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||||
|
atom_label(skip_port1_from_ackrel2)
|
||||||
|
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel2, end_atom)),
|
||||||
|
|
||||||
|
atom_label(ack_released2_port1)
|
||||||
|
|
||||||
|
/* === Byte 3 (port 1): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
|
||||||
|
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||||
|
add_ui(R_T0, R_0, 0x00),
|
||||||
|
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(wait_ack3_port1)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_ne(R_T0, R_0, atom_offset(wait_ack3_port1, ack3_received_port1)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
atom_label(continue_wait_ack3_port1)
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack3_port1, wait_ack3_port1)),
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||||
|
atom_label(skip_port1_from_ack3)
|
||||||
|
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack3, end_atom)),
|
||||||
|
|
||||||
|
atom_label(ack3_received_port1)
|
||||||
|
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
|
||||||
|
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(wait_ackrel3_port1)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_equal(R_T0, R_0, atom_offset(wait_ackrel3_port1, ack_released3_port1)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
atom_label(continue_wait_ackrel3_port1)
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel3_port1, wait_ackrel3_port1)),
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||||
|
atom_label(skip_port1_from_ackrel3)
|
||||||
|
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel3, end_atom)),
|
||||||
|
|
||||||
|
atom_label(ack_released3_port1)
|
||||||
|
|
||||||
|
/* === Byte 4 (FINAL, port 1): send 0x00 + RX-not-empty wait + read final byte === */
|
||||||
|
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||||
|
add_ui(R_T0, R_0, 0x00),
|
||||||
|
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(wait_rx4_port1)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_ne(R_T0, R_0, atom_offset(wait_rx4_port1, rx4_received_port1)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
atom_label(continue_wait_rx4_port1)
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(continue_wait_rx4_port1, wait_rx4_port1)),
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||||
|
atom_label(skip_port1_from_rx4)
|
||||||
|
branch_equal(R_0, R_0, atom_offset(skip_port1_from_rx4, end_atom)),
|
||||||
|
|
||||||
|
atom_label(rx4_received_port1)
|
||||||
|
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET), /* discard final byte */
|
||||||
|
|
||||||
|
/* === RESPONSE DECODE (port 1) === */
|
||||||
|
atom_label(decode_port1)
|
||||||
|
mac_pad_sio_write_pad_state(PadSioStatus_Digital, R_PadState, R_T0),
|
||||||
|
|
||||||
|
/* /CS cleanup: raise /CS, clear stale status before exiting port 1. */
|
||||||
|
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
|
||||||
|
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||||
|
|
||||||
|
atom_label(end_atom)
|
||||||
|
mac_yield(),
|
||||||
|
};
|
||||||
|
#endif /* end pad_sio_step wrap */
|
||||||
|
|
||||||
|
/* ----- pad_sio_diag_pin -----
|
||||||
|
* Per-frame diagnostic counter. The caller binds R_DiagPinScratch to
|
||||||
|
* scratch_for_atom_diag_pin for temporary gdb verification.
|
||||||
|
*/
|
||||||
|
#if 0 /* pad_sio_diag_pin — superseded (raw-SIO phase removed) */
|
||||||
|
internal MipsAtom_(pad_sio_diag_pin) atom_info(atom_phase(pad_init)
|
||||||
|
, atom_reads(R_T0, R_T1, R_DiagPinScratch)
|
||||||
|
, atom_writes(R_T0, R_T1, R_DiagPinScratch)
|
||||||
|
) {
|
||||||
|
/* FIX 2026-08-02: explicitly reload R_DiagPinScratch (R_T3 = $t3). Caller-saved
|
||||||
|
* per O32 ABI; the rgcc binding in main() does not survive tape_run. */
|
||||||
|
load_upper_i(R_DiagPinScratch, 0x8001),
|
||||||
|
or_i(R_DiagPinScratch, R_DiagPinScratch, 0xC800),
|
||||||
|
|
||||||
|
/* High half = 0xD1A6; low half increments once per atom invocation. */
|
||||||
|
load_word(R_T1, R_DiagPinScratch, 0),
|
||||||
|
nop,
|
||||||
|
add_ui(R_T1, R_T1, 1),
|
||||||
|
and_i(R_T0, R_T1, 0xFFFF),
|
||||||
|
load_upper_i(R_T1, 0xD1A6),
|
||||||
|
or_i(R_T1, R_T1, 0),
|
||||||
|
or_u(R_T1, R_T1, R_T0),
|
||||||
|
store_word(R_T1, R_DiagPinScratch, 0),
|
||||||
|
mac_yield(),
|
||||||
|
};
|
||||||
|
#endif /* end pad_sio_diag_pin wrap */
|
||||||
|
|
||||||
|
/* ----- pad_sio_diag_byte_exchange -----
|
||||||
|
* Temporary two-byte wire probe: sends 0x01 and 0x42, then stores the
|
||||||
|
* open-bus byte and response ID in scratch_for_atom_diag_pin.
|
||||||
|
*/
|
||||||
|
#if 0 /* pad_sio_diag_byte_exchange — superseded (raw-SIO phase removed) */
|
||||||
|
internal MipsAtom_(pad_sio_diag_byte_exchange) atom_info(atom_phase(pad_init)
|
||||||
|
, atom_reads(R_T0, R_T1, R_T2, R_PadSioBase, R_DiagPinScratch)
|
||||||
|
, atom_writes(R_T0, R_T1, R_T2, R_PadSioBase, R_DiagPinScratch)
|
||||||
|
) {
|
||||||
|
/* FIX 2026-08-02: explicitly reload R_DiagPinScratch (R_T3 = $t3). Caller-saved
|
||||||
|
* per O32 ABI; the rgcc binding in main() does not survive tape_run. */
|
||||||
|
load_upper_i(R_DiagPinScratch, 0x8001),
|
||||||
|
or_i(R_DiagPinScratch, R_DiagPinScratch, 0xC800),
|
||||||
|
|
||||||
|
/* FIX 2026-08-02: explicitly load KSEG1 base into R_PadSioBase (R_T6) at the
|
||||||
|
* top. The rgcc() binding in main() does NOT survive the tape_run call
|
||||||
|
* because R_T6 is caller-saved per the O32 ABI. */
|
||||||
|
load_upper_i(R_PadSioBase, pad_IO_KSEG1_BASE >> 16),
|
||||||
|
or_i(R_PadSioBase, R_PadSioBase, pad_IO_KSEG1_BASE & 0xFFFF),
|
||||||
|
|
||||||
|
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
|
||||||
|
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||||
|
add_ui(R_T0, R_0, pad_SIO_CTRL_TX_ENABLE),
|
||||||
|
or_i(R_T0, R_T0, pad_SIO_CTRL_DTR_CS),
|
||||||
|
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||||
|
|
||||||
|
add_ui(R_T0, R_0, pad_PROTO_ADDR),
|
||||||
|
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(diag_wait_ack0)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_ne(R_T0, R_0, atom_offset(diag_wait_ack0, diag_ack0_done)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(diag_wait_ack0, diag_wait_ack0)),
|
||||||
|
add_ui(R_T0, R_0, 0xDEADAC01),
|
||||||
|
store_word(R_T0, R_DiagPinScratch, 0),
|
||||||
|
branch_equal(R_0, R_0, atom_offset(diag_timeout_ack0, diag_timeout)),
|
||||||
|
atom_label(diag_ack0_done)
|
||||||
|
load_byte_u(R_T2, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
add_ui(R_T0, R_0, pad_PROTO_CMD_READ),
|
||||||
|
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||||
|
atom_label(diag_wait_ack1)
|
||||||
|
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||||
|
nop,
|
||||||
|
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||||
|
branch_ne(R_T0, R_0, atom_offset(diag_wait_ack1, diag_ack1_done)),
|
||||||
|
add_ui_self(R_T1, -1),
|
||||||
|
branch_ne(R_T1, R_0, atom_offset(diag_wait_ack1, diag_wait_ack1)),
|
||||||
|
add_ui(R_T0, R_0, 0xDEADAC02),
|
||||||
|
store_word(R_T0, R_DiagPinScratch, 0),
|
||||||
|
branch_equal(R_0, R_0, atom_offset(diag_timeout_ack1, diag_timeout)),
|
||||||
|
atom_label(diag_ack1_done)
|
||||||
|
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||||
|
nop,
|
||||||
|
shift_lleft(R_T0, R_T0, 8),
|
||||||
|
or_u(R_T2, R_T2, R_T0),
|
||||||
|
store_word(R_T2, R_DiagPinScratch, 0),
|
||||||
|
atom_label(diag_success)
|
||||||
|
branch_equal(R_0, R_0, atom_offset(diag_success, diag_done)),
|
||||||
|
nop,
|
||||||
|
atom_label(diag_timeout_ack0)
|
||||||
|
add_ui(R_T0, R_0, 0xDEADAC01),
|
||||||
|
store_word(R_T0, R_DiagPinScratch, 0),
|
||||||
|
atom_label(diag_timeout_ack1)
|
||||||
|
add_ui(R_T0, R_0, 0xDEADAC02),
|
||||||
|
store_word(R_T0, R_DiagPinScratch, 0),
|
||||||
|
atom_label(diag_timeout)
|
||||||
|
add_ui(R_T0, R_0, 0xDEADACFF),
|
||||||
|
store_word(R_T0, R_DiagPinScratch, 0),
|
||||||
|
atom_label(diag_done)
|
||||||
|
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
|
||||||
|
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||||
|
mac_yield(),
|
||||||
|
};
|
||||||
|
#endif /* end pad_sio_diag_byte_exchange wrap */
|
||||||
+10625
File diff suppressed because one or more lines are too long
Binary file not shown.
|
After Width: | Height: | Size: 220 KiB |
@@ -6,18 +6,33 @@ A rest from the usual.
|
|||||||
|
|
||||||
## Dependencies
|
## Dependencies
|
||||||
|
|
||||||
I will be programming from a Windows 11 machine:
|
I will be programming from a Windows 11 machine (may eventually try this on the Steam Deck...):
|
||||||

|
|
||||||
|
|
||||||
```ps1
|
|
||||||
# not really used yet for scripts (may never)
|
|
||||||
scoop install lua
|
|
||||||
```
|
|
||||||
|
|
||||||
[armips](https://github.com/Kingcom/armips)
|
[armips](https://github.com/Kingcom/armips)
|
||||||
|
|
||||||
* Supports doing bare-metal assembly for the ps1
|
* Supports doing bare-metal assembly for the ps1
|
||||||
* `scoop install armips` or just clone and build..
|
* `scoop install armips` or just clone and build..
|
||||||
|
* Was used early in the course. Now I just use an macro asm dsl in C11.
|
||||||
|
|
||||||
|
[luajit-2.1](https://github.com/LuaJIT/LuaJIT.git)
|
||||||
|
|
||||||
|
```
|
||||||
|
scoop install luajit
|
||||||
|
```
|
||||||
|
|
||||||
|
* Used for lua scripts
|
||||||
|
* Particularly, ps1_meta.lua which is a staged metaprogram pass for the custom C11 Assembly DSL used in this codebase.
|
||||||
|
|
||||||
|
[lpeg](https://github.com/roberto-ieru/LPeg.git)
|
||||||
|
|
||||||
|
* Lua is slow (even jitted) so this helps.
|
||||||
|
|
||||||
|
[lfs (LuaFileSystem)](https://github.com/lunarmodules/luafilesystem)
|
||||||
|
|
||||||
|
* Native directory enumeration + `mkdir` for the build scripts.
|
||||||
|
* Used by `passes/word_count_eval.lua :: scan_dir` (native walk vs. `dir /b /s` subprocess,
|
||||||
|
~2ms vs. ~56ms) and by `duffle.lua :: ensure_dir` + `to_absolute_path` (avoids
|
||||||
|
`cmd.exe mkdir` + `cd` shell spawns, ~50ms each).
|
||||||
|
|
||||||
[pscx-redux](https://github.com/grumpycoders/pcsx-redux/): A collection of tools, research, hardware design, and libraries aiming at development and reverse engineering on the PlayStation 1.
|
[pscx-redux](https://github.com/grumpycoders/pcsx-redux/): A collection of tools, research, hardware design, and libraries aiming at development and reverse engineering on the PlayStation 1.
|
||||||
|
|
||||||
@@ -57,3 +72,10 @@ scoop install lua
|
|||||||

|

|
||||||

|

|
||||||

|

|
||||||
|

|
||||||
|
|
||||||
|
Win 11 machine:
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
Still haven't gotten around to trying this on linux...
|
||||||
|
|||||||
@@ -0,0 +1,30 @@
|
|||||||
|
-- gte_debug.lua — defensive version + prints error context.
|
||||||
|
local ok, err = pcall(function()
|
||||||
|
print("[debug] PCSX exists:", PCSX ~= nil)
|
||||||
|
print("[debug] PCSX.WebServer exists:", PCSX and PCSX.WebServer ~= nil)
|
||||||
|
print("[debug] PCSX.WebServer.Handlers exists:", PCSX and PCSX.WebServer and PCSX.WebServer.Handlers ~= nil)
|
||||||
|
if not PCSX.WebServer then
|
||||||
|
print("[debug] creating PCSX.WebServer...")
|
||||||
|
PCSX.WebServer = {}
|
||||||
|
end
|
||||||
|
if not PCSX.WebServer.Handlers then
|
||||||
|
print("[debug] creating PCSX.WebServer.Handlers...")
|
||||||
|
PCSX.WebServer.Handlers = {}
|
||||||
|
end
|
||||||
|
print("[debug] type of Handlers:", type(PCSX.WebServer.Handlers))
|
||||||
|
|
||||||
|
PCSX.WebServer.Handlers.gte = function(req)
|
||||||
|
local r = PCSX.getRegisters()
|
||||||
|
local out = { "pc=0x" .. string.format("%x", r.pc) }
|
||||||
|
for i = 0, 31 do
|
||||||
|
out[#out + 1] = string.format("D[%d]=0x%08x C[%d]=0x%08x",
|
||||||
|
i, r.CP2D.r[i], i, r.CP2C.r[i])
|
||||||
|
end
|
||||||
|
return table.concat(out, "\n")
|
||||||
|
end
|
||||||
|
print("[debug] handler registered")
|
||||||
|
end)
|
||||||
|
|
||||||
|
if not ok then
|
||||||
|
print("[debug] ERROR: " .. tostring(err))
|
||||||
|
end
|
||||||
@@ -0,0 +1,217 @@
|
|||||||
|
--- audit_lua_nesting.lua — Walk Lua source files and flag any block nesting deeper than 5 levels.
|
||||||
|
---
|
||||||
|
--- Usage:
|
||||||
|
--- luajit scripts/audit_lua_nesting.lua scripts/duffle.lua scripts/ps1_meta.lua
|
||||||
|
--- luajit scripts/audit_lua_nesting.lua scripts/passes/
|
||||||
|
---
|
||||||
|
--- Output: for each file, a list of {line, depth} entries where depth > 5.
|
||||||
|
--- Returns exit code 1 if any violations found, 0 if clean.
|
||||||
|
---
|
||||||
|
--- **Implementation**: a hand-rolled depth tracker that counts:
|
||||||
|
--- - `do`, `function`, `if`, `for`, `while`, `repeat` -> depth +1
|
||||||
|
--- - `end`, `until` -> depth -1
|
||||||
|
--- - `else`, `elseif` -> depth unchanged
|
||||||
|
---
|
||||||
|
--- **Caveats**: doesn't fully handle string/comment state (will miscount braces inside multi-line strings or block comments).
|
||||||
|
--- For our metaprogram files (no embedded code generation), this is acceptable.
|
||||||
|
|
||||||
|
local M = {}
|
||||||
|
|
||||||
|
local BLOCK_OPEN = {
|
||||||
|
["do"] = true,
|
||||||
|
["function"] = true,
|
||||||
|
["if"] = true,
|
||||||
|
["for"] = true,
|
||||||
|
["while"] = true,
|
||||||
|
["repeat"] = true,
|
||||||
|
}
|
||||||
|
|
||||||
|
local function is_block_close(token) return token == "end" or token == "until" end
|
||||||
|
|
||||||
|
-- (internal) Walk one source file and return a list of
|
||||||
|
-- {line, depth, token} entries where depth > max_nesting.
|
||||||
|
local function audit_file(path, max_nesting)
|
||||||
|
local f = io.open(path, "r")
|
||||||
|
if not f then error("Cannot open " .. path) end
|
||||||
|
local content = f:read("*a")
|
||||||
|
f:close()
|
||||||
|
|
||||||
|
local violations = {}
|
||||||
|
local depth = 0
|
||||||
|
local line = 1
|
||||||
|
local pos = 1
|
||||||
|
local src_len = #content
|
||||||
|
local token_idx = 0
|
||||||
|
|
||||||
|
local function read_ident_at(start_pos)
|
||||||
|
local ident_start = start_pos
|
||||||
|
if ident_start > src_len then return nil end
|
||||||
|
local first_ch = content:sub(ident_start, ident_start)
|
||||||
|
if not (first_ch:match("[%a_]")) then return nil end
|
||||||
|
local scan = start_pos + 1
|
||||||
|
while scan <= src_len do
|
||||||
|
local ch = content:sub(scan, scan)
|
||||||
|
if not (ch:match("[%w_]")) then break end
|
||||||
|
scan = scan + 1
|
||||||
|
end
|
||||||
|
return content:sub(ident_start, scan - 1), scan
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Skip past a string literal or comment starting at `start_pos`.
|
||||||
|
-- Returns the position just past the construct, or nil if `start_pos`
|
||||||
|
-- is not the start of a string/comment.
|
||||||
|
local function skip_string_or_comment(start_pos)
|
||||||
|
local ch = content:sub(start_pos, start_pos)
|
||||||
|
if ch == '"' or ch == "'" then
|
||||||
|
local scan = start_pos + 1
|
||||||
|
while scan <= src_len do
|
||||||
|
local c = content:sub(scan, scan)
|
||||||
|
if c == "\\" then scan = scan + 2
|
||||||
|
elseif c == ch then return scan + 1
|
||||||
|
else scan = scan + 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return src_len + 1
|
||||||
|
elseif ch == "-" and content:sub(start_pos + 1, start_pos + 1) == "-" then
|
||||||
|
local scan = start_pos + 2
|
||||||
|
if content:sub(scan, scan + 1) == "[[" and content:sub(scan + 2, scan + 3) == "[" then
|
||||||
|
-- Long bracket comment [==[ ... ]==]
|
||||||
|
scan = scan + 2
|
||||||
|
local eq = ""
|
||||||
|
while content:sub(scan, scan) == "=" do
|
||||||
|
eq = eq .. "="
|
||||||
|
scan = scan + 1
|
||||||
|
end
|
||||||
|
local close_marker = "]" .. eq .. "]"
|
||||||
|
local close_pos = content:find(close_marker, scan, true)
|
||||||
|
if close_pos then
|
||||||
|
return close_pos + #close_marker
|
||||||
|
else
|
||||||
|
return src_len + 1
|
||||||
|
end
|
||||||
|
else
|
||||||
|
while scan <= src_len and content:sub(scan, scan) ~= "\n" do scan = scan + 1 end
|
||||||
|
return scan + 1
|
||||||
|
end
|
||||||
|
elseif ch == "[" and content:sub(start_pos + 1, start_pos + 1) == "[" then
|
||||||
|
local scan = start_pos + 2
|
||||||
|
local eq = ""
|
||||||
|
while content:sub(scan, scan) == "=" do
|
||||||
|
eq = eq .. "="
|
||||||
|
scan = scan + 1
|
||||||
|
end
|
||||||
|
local close_marker = "]" .. eq .. "]"
|
||||||
|
local close_pos = content:find(close_marker, scan, true)
|
||||||
|
if close_pos then
|
||||||
|
return close_pos + #close_marker
|
||||||
|
else
|
||||||
|
return src_len + 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return nil
|
||||||
|
end
|
||||||
|
|
||||||
|
while pos <= src_len do
|
||||||
|
local ch = content:sub(pos, pos)
|
||||||
|
if ch == "\n" then line = line + 1 end
|
||||||
|
|
||||||
|
local skip_to = skip_string_or_comment(pos)
|
||||||
|
if skip_to then
|
||||||
|
for scan = pos, skip_to - 1 do
|
||||||
|
if content:sub(scan, scan) == "\n" then line = line + 1 end
|
||||||
|
end
|
||||||
|
pos = skip_to
|
||||||
|
elseif ch:match("[%a_]") then
|
||||||
|
local tok, next_pos = read_ident_at(pos)
|
||||||
|
token_idx = token_idx + 1
|
||||||
|
if BLOCK_OPEN[tok] then
|
||||||
|
depth = depth + 1
|
||||||
|
if depth > max_nesting then
|
||||||
|
violations[#violations + 1] = {
|
||||||
|
line = line,
|
||||||
|
depth = depth,
|
||||||
|
token = tok,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
elseif is_block_close(tok) then
|
||||||
|
depth = depth - 1
|
||||||
|
end
|
||||||
|
pos = next_pos
|
||||||
|
else
|
||||||
|
pos = pos + 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return violations
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Audit one file. Returns nil if clean, else a list of violations.
|
||||||
|
--- @param path string
|
||||||
|
--- @param max_nesting integer -- default 5
|
||||||
|
--- @return table|nil
|
||||||
|
function M.audit(path, max_nesting)
|
||||||
|
local violations = audit_file(path, max_nesting or 5)
|
||||||
|
if #violations == 0 then return nil end
|
||||||
|
return violations
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Module CLI.
|
||||||
|
if arg and arg[1] then
|
||||||
|
local max_nesting = 5
|
||||||
|
local files = {}
|
||||||
|
for arg_idx = 1, #arg do
|
||||||
|
if arg[arg_idx] == "--max" and arg[arg_idx + 1] then
|
||||||
|
max_nesting = tonumber(arg[arg_idx + 1]) or 5
|
||||||
|
else
|
||||||
|
files[#files + 1] = arg[arg_idx]
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Accept either a directory or a file path. Directory args are
|
||||||
|
-- expanded via lfs.dir (native, no subprocess).
|
||||||
|
local lfs = require("lfs")
|
||||||
|
local function is_dir(p)
|
||||||
|
return lfs.attributes(p, "mode") == "directory"
|
||||||
|
end
|
||||||
|
local function list_lua(dir)
|
||||||
|
local out = {}
|
||||||
|
if not is_dir(dir) then return out end
|
||||||
|
for entry in lfs.dir(dir) do
|
||||||
|
if entry:match("%.lua$") then
|
||||||
|
out[#out + 1] = dir .. "/" .. entry
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return out
|
||||||
|
end
|
||||||
|
|
||||||
|
local to_check = {}
|
||||||
|
for _, f in ipairs(files) do
|
||||||
|
if is_dir(f) then
|
||||||
|
for _, sub in ipairs(list_lua(f)) do to_check[#to_check + 1] = sub end
|
||||||
|
else
|
||||||
|
to_check[#to_check + 1] = f
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
local total_violations = 0
|
||||||
|
for _, f in ipairs(to_check) do
|
||||||
|
local v = M.audit(f, max_nesting)
|
||||||
|
if v then
|
||||||
|
io.write(string.format("\n%s\n", f))
|
||||||
|
for _, x in ipairs(v) do
|
||||||
|
io.write(string.format(" line %d: depth %d (after '%s')\n", x.line, x.depth, x.token))
|
||||||
|
end
|
||||||
|
total_violations = total_violations + #v
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
if total_violations == 0 then
|
||||||
|
io.write("OK: no files exceed max nesting of " .. max_nesting .. "\n")
|
||||||
|
os.exit(0)
|
||||||
|
else
|
||||||
|
io.write(string.format("\n%d nesting violation(s) found.\n", total_violations))
|
||||||
|
os.exit(1)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return M
|
||||||
+239
-148
@@ -81,19 +81,6 @@ $path_psyq = join-path $path_toolchain 'psyq-4_7'
|
|||||||
$path_psyq_iwyu = join-path $path_toolchain 'psyq_iwyu'
|
$path_psyq_iwyu = join-path $path_toolchain 'psyq_iwyu'
|
||||||
$path_psyq_imyu_inc = join-path $path_psyq_iwyu 'include'
|
$path_psyq_imyu_inc = join-path $path_psyq_iwyu 'include'
|
||||||
|
|
||||||
function Get-SourceFiles { param([Parameter(Mandatory=$true)] [string[]]$paths, [Parameter(Mandatory=$true)] [string[]]$extensions)
|
|
||||||
$files = @()
|
|
||||||
foreach ($p in $paths) {
|
|
||||||
if (-not (test-path $p)) { continue }
|
|
||||||
foreach ($ext in $extensions) {
|
|
||||||
Get-ChildItem -Path $p -File -Recurse -Filter "*$ext" -ErrorAction SilentlyContinue | ForEach-Object {
|
|
||||||
$files += $_.FullName
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return ($files | Sort-Object -Unique)
|
|
||||||
}
|
|
||||||
|
|
||||||
function assemble-unit { param(
|
function assemble-unit { param(
|
||||||
[string] $unit,
|
[string] $unit,
|
||||||
[string] $link_module,
|
[string] $link_module,
|
||||||
@@ -153,7 +140,7 @@ function compile-unit { param(
|
|||||||
$f_arch_no_shared,
|
$f_arch_no_shared,
|
||||||
$f_arch_no_stack_prot
|
$f_arch_no_stack_prot
|
||||||
)
|
)
|
||||||
# $compile_args += $f_std_c23
|
$compile_args += $f_std_c11
|
||||||
$compile_args += ($f_include + $path_psyq_imyu_inc)
|
$compile_args += ($f_include + $path_psyq_imyu_inc)
|
||||||
$compile_args += ($f_include + $path_nugget)
|
$compile_args += ($f_include + $path_nugget)
|
||||||
|
|
||||||
@@ -193,29 +180,15 @@ function link-modules { param([string[]]$link_modules, [string] $elf, [string[]
|
|||||||
$link_args += ($f_link_pass_through_prefix + $f_link_mapfile + $map)
|
$link_args += ($f_link_pass_through_prefix + $f_link_mapfile + $map)
|
||||||
|
|
||||||
$link_args += ($f_link_pass_through_prefix + $f_link_start_group)
|
$link_args += ($f_link_pass_through_prefix + $f_link_start_group)
|
||||||
|
# 16 removed entries (c2, card, cd, comb, ds, gs, gun, hmd, math, mcrd, mcx, press, sio, snd, spu, tap)
|
||||||
|
# had LOAD lines in the map but ZERO .o files pulled in — they were unused.
|
||||||
|
# 5 kept libraries (api, c, etc, gpu, gte) are required by the C-side calls in hello_joypad.c (reset_graph, draw_sync, vsync, etc.).
|
||||||
$libraries = @(
|
$libraries = @(
|
||||||
"api",
|
"api",
|
||||||
"c",
|
"c",
|
||||||
"c2",
|
|
||||||
"card",
|
|
||||||
"cd",
|
|
||||||
"comb",
|
|
||||||
"ds",
|
|
||||||
"etc",
|
"etc",
|
||||||
"gpu",
|
"gpu",
|
||||||
"gs",
|
"gte"
|
||||||
"gte",
|
|
||||||
"gun",
|
|
||||||
"hmd",
|
|
||||||
"math",
|
|
||||||
"mcrd",
|
|
||||||
"mcx",
|
|
||||||
"pad",
|
|
||||||
"press",
|
|
||||||
"sio",
|
|
||||||
"snd",
|
|
||||||
"spu",
|
|
||||||
"tap"
|
|
||||||
)
|
)
|
||||||
foreach ($lib in $libraries) {
|
foreach ($lib in $libraries) {
|
||||||
$link_args += ($f_link_lib + $lib)
|
$link_args += ($f_link_lib + $lib)
|
||||||
@@ -243,6 +216,117 @@ function make-binary { param([string]$elf, [string]$exe)
|
|||||||
if ($LASTEXITCODE -ne 0) { Write-Error "Objcopy failed. Aborting."; exit 1 }
|
if ($LASTEXITCODE -ne 0) { Write-Error "Objcopy failed. Aborting."; exit 1 }
|
||||||
}
|
}
|
||||||
|
|
||||||
|
function ps1-meta { param(
|
||||||
|
[string]$unity_root,
|
||||||
|
[string[]]$sources,
|
||||||
|
[Parameter(Mandatory=$true)][string]$metadata,
|
||||||
|
[string]$out_root = (join-path $path_build 'gen'),
|
||||||
|
[string[]]$passes = @('--pre-link'),
|
||||||
|
[string[]]$extra_args = @()
|
||||||
|
)
|
||||||
|
# `--unity-root` and `--source` are mutually exclusive. Exactly one of `$unity_root` / `$sources` must be supplied; the other must be absent.
|
||||||
|
if ($null -ne $unity_root -and $unity_root -ne '')
|
||||||
|
{
|
||||||
|
if ($null -ne $sources -and $sources.Count -gt 0) {
|
||||||
|
write-error 'ps1-meta: -unity_root and -sources are mutually exclusive'
|
||||||
|
exit 2
|
||||||
|
}
|
||||||
|
}
|
||||||
|
elseif ($null -eq $sources -or $sources.Count -eq 0) {
|
||||||
|
write-error 'ps1-meta: either -unity_root <file> or -sources <file...> is required'
|
||||||
|
exit 2
|
||||||
|
}
|
||||||
|
|
||||||
|
$script = join-path $path_scripts 'ps1_meta.lua'
|
||||||
|
$input_summary = if ($null -ne $unity_root -and $unity_root -ne '') {
|
||||||
|
"unity=$unity_root"
|
||||||
|
}
|
||||||
|
else {
|
||||||
|
"$($sources.Count) source(s)"
|
||||||
|
}
|
||||||
|
write-host "ps1-meta $input_summary, passes=$($passes -join ',')" ` -ForegroundColor Magenta
|
||||||
|
|
||||||
|
$arg_list = @($passes) + @('--metadata', $metadata) + @('--out-root', $out_root) + @($extra_args)
|
||||||
|
if ($null -ne $unity_root -and $unity_root -ne '') {
|
||||||
|
$arg_list += @('--unity-root', $unity_root)
|
||||||
|
}
|
||||||
|
else {
|
||||||
|
foreach ($s in $sources) { $arg_list += @('--source', $s) }
|
||||||
|
}
|
||||||
|
& luajit $script @arg_list
|
||||||
|
if ($LASTEXITCODE -ne 0) {
|
||||||
|
write-error "ps1-meta failed (exit $LASTEXITCODE). Aborting."
|
||||||
|
exit $LASTEXITCODE
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
function inject-dwarf { param(
|
||||||
|
[string]$elf,
|
||||||
|
[string]$path_gen
|
||||||
|
)
|
||||||
|
$base_name = [System.IO.Path]::GetFileNameWithoutExtension($elf)
|
||||||
|
$path_dwarf_line_bin = join-path $path_gen "$base_name.dwarf_line.bin"
|
||||||
|
$path_dwarf_aranges_bin = join-path $path_gen "$base_name.dwarf_aranges.bin"
|
||||||
|
$path_dwarf_rnglists_bin = join-path $path_gen "$base_name.dwarf_rnglists.bin"
|
||||||
|
$path_dwarf_info_bin = join-path $path_gen "$base_name.dwarf_info.bin"
|
||||||
|
$path_dwarf_abbrev_bin = join-path $path_gen "$base_name.dwarf_abbrev.bin"
|
||||||
|
$path_dwarf_str_bin = join-path $path_gen "$base_name.dwarf_str.bin"
|
||||||
|
$path_dwarf_loc_bin = join-path $path_gen "$base_name.dwarf_loc.bin"
|
||||||
|
$path_dwarf_loclists_bin = join-path $path_gen "$base_name.dwarf_loclists.bin"
|
||||||
|
$path_inject_elf = join-path $path_build "$base_name.dwarf-injected.elf"
|
||||||
|
|
||||||
|
if (-not (Test-Path $path_dwarf_line_bin)) { return }
|
||||||
|
if (-not (Test-Path $path_dwarf_aranges_bin)) { return }
|
||||||
|
if (-not (Test-Path $path_dwarf_rnglists_bin)) { return }
|
||||||
|
|
||||||
|
Write-Host "[build] DWARF-injecting $elf -> $path_inject_elf"
|
||||||
|
Copy-Item -LiteralPath $elf -Destination $path_inject_elf -Force
|
||||||
|
|
||||||
|
# Objcopy call 1: 3x --update-section for the PC-mapping tables (line, aranges, rnglists).
|
||||||
|
$objcopy_args_dwarf_pc = @(
|
||||||
|
"--update-section=.debug_line=$path_dwarf_line_bin",
|
||||||
|
"--update-section=.debug_aranges=$path_dwarf_aranges_bin",
|
||||||
|
"--update-section=.debug_rnglists=$path_dwarf_rnglists_bin"
|
||||||
|
)
|
||||||
|
& $Objcopy @objcopy_args_dwarf_pc $path_inject_elf 2>&1 | Out-Null
|
||||||
|
if ($LASTEXITCODE -ne 0) {
|
||||||
|
Write-Warning "[build] objcopy dwarf-pc splice failed (exit $LASTEXITCODE); removing $path_inject_elf"
|
||||||
|
Remove-Item -LiteralPath $path_inject_elf -ErrorAction SilentlyContinue
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
# Objcopy call 2: 3x --update-section + 2x --add-section for the debug-data tables (info, abbrev, str, loc, loclists).
|
||||||
|
$objcopy_args_dwarf_info = @(
|
||||||
|
"--update-section=.debug_info=$path_dwarf_info_bin",
|
||||||
|
"--update-section=.debug_abbrev=$path_dwarf_abbrev_bin",
|
||||||
|
"--update-section=.debug_str=$path_dwarf_str_bin",
|
||||||
|
"--add-section=.debug_loc=$path_dwarf_loc_bin",
|
||||||
|
"--add-section=.debug_loclists=$path_dwarf_loclists_bin"
|
||||||
|
)
|
||||||
|
& $Objcopy @objcopy_args_dwarf_info $path_inject_elf 2>&1 | Out-Null
|
||||||
|
if ($LASTEXITCODE -ne 0) {
|
||||||
|
Write-Warning "[build] objcopy dwarf-info splice failed (exit $LASTEXITCODE); removing $path_inject_elf"
|
||||||
|
Remove-Item -LiteralPath $path_inject_elf -ErrorAction SilentlyContinue
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
# Baked atoms execute from RAM but are emitted as C data arrays, so their ELF sections lack SHF_EXECINSTR.
|
||||||
|
# GDB discards line rows for non-code sections. Mark only the debug-copy sections executable.
|
||||||
|
# The original ELF and PS-EXE remain byte/flag unchanged.
|
||||||
|
& $Objcopy `
|
||||||
|
--set-section-flags ".rodata=alloc,load,readonly,code,contents" `
|
||||||
|
--set-section-flags ".data=alloc,load,data,code,contents" `
|
||||||
|
$path_inject_elf 2>&1 | Out-Null
|
||||||
|
if ($LASTEXITCODE -ne 0) {
|
||||||
|
Write-Warning "[build] atom-section flag update failed (exit $LASTEXITCODE); removing $path_inject_elf"
|
||||||
|
Remove-Item -LiteralPath $path_inject_elf -ErrorAction SilentlyContinue
|
||||||
|
}
|
||||||
|
else {
|
||||||
|
Write-Host "[build] DWARF-injected ELF: $path_inject_elf"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
# inject-dwarf
|
||||||
|
|
||||||
function build-hello_psyqo {
|
function build-hello_psyqo {
|
||||||
$includes += @()
|
$includes += @()
|
||||||
|
|
||||||
@@ -287,7 +371,7 @@ function build-graphis_hello {
|
|||||||
|
|
||||||
$src_asm_crt = join-path $path_nugget_common 'crt0/crt0.s'
|
$src_asm_crt = join-path $path_nugget_common 'crt0/crt0.s'
|
||||||
$module_asm_crt = join-path $path_build 'crt0.o'
|
$module_asm_crt = join-path $path_build 'crt0.o'
|
||||||
# assemble-unit $src_asm_crt $module_asm_crt $includes $assemble_args
|
assemble-unit $src_asm_crt $module_asm_crt $includes $assemble_args
|
||||||
|
|
||||||
$src_asm = join-path $path_module 'hello_gpu.s'
|
$src_asm = join-path $path_module 'hello_gpu.s'
|
||||||
$module_asm = join-path $path_build 'hello_gpu.o'
|
$module_asm = join-path $path_build 'hello_gpu.o'
|
||||||
@@ -317,114 +401,16 @@ function build-graphis_hello {
|
|||||||
}
|
}
|
||||||
# build-graphis_hello
|
# build-graphis_hello
|
||||||
|
|
||||||
function generate-TapeAtomOffsets {param([Parameter(Mandatory=$true)] [string[]]$sources, [Parameter(Mandatory=$true)] [string]$metadata)
|
function build-hello_gte {
|
||||||
$gen_atom_offsets_script = join-path $path_scripts 'tape_atom.offset_gen.meta.lua'
|
|
||||||
|
|
||||||
$any_stale = $false
|
|
||||||
foreach ($src in $sources) {
|
|
||||||
$basename = [System.IO.Path]::GetFileNameWithoutExtension($src)
|
|
||||||
$dir = split-path -Path $src -Parent
|
|
||||||
$gen_dir = join-path $dir 'gen'
|
|
||||||
$out = join-path $gen_dir "$basename.offsets.h"
|
|
||||||
|
|
||||||
if (-not (test-path $out)) { $any_stale = $true; break }
|
|
||||||
$src_mtime = (get-item $src).LastWriteTimeUtc
|
|
||||||
$out_mtime = (get-item $out).LastWriteTimeUtc
|
|
||||||
$meta_mtime = (get-item $metadata).LastWriteTimeUtc
|
|
||||||
if (($src_mtime -gt $out_mtime) -or ($meta_mtime -gt $out_mtime)) {
|
|
||||||
$any_stale = $true
|
|
||||||
break
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if (-not $any_stale) {
|
|
||||||
write-host "AtomOffsets all $($sources.Count) source(s) up-to-date" -ForegroundColor DarkGray
|
|
||||||
return
|
|
||||||
}
|
|
||||||
|
|
||||||
write-host "AtomOffsets $($sources.Count) source(s)" -ForegroundColor Magenta
|
|
||||||
& lua $gen_atom_offsets_script $metadata @sources
|
|
||||||
if ($LASTEXITCODE -ne 0) {
|
|
||||||
write-error "Atom offset generation failed. Aborting."
|
|
||||||
exit 1
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
function generate-TapeAtomAnnotations {param([Parameter(Mandatory=$true)] [string[]]$sources, [Parameter(Mandatory=$true)] [string]$metadata)
|
|
||||||
# Sibling to generate-TapeAtomOffsets. Validates TAPE_ATOM_* / TAPE_WORDS
|
|
||||||
# annotations against the metadata manifest. Emits gen/<basename>.errors.h
|
|
||||||
# containing #error directives for the C build to fail on annotation drift.
|
|
||||||
$gen_atom_annot_script = join-path $path_scripts 'tape_atom_annotation_pass.lua'
|
|
||||||
|
|
||||||
$any_stale = $false
|
|
||||||
foreach ($src in $sources) {
|
|
||||||
$basename = [System.IO.Path]::GetFileNameWithoutExtension($src)
|
|
||||||
$dir = split-path -Path $src -Parent
|
|
||||||
$gen_dir = join-path $dir 'gen'
|
|
||||||
$out_txt = join-path $gen_dir "$basename.annotations.txt"
|
|
||||||
$out_err = join-path $gen_dir "$basename.errors.h"
|
|
||||||
|
|
||||||
if (-not (test-path $out_txt) -or -not (test-path $out_err)) { $any_stale = $true; break }
|
|
||||||
$src_mtime = (get-item $src).LastWriteTimeUtc
|
|
||||||
$out_txt_mtime = (get-item $out_txt).LastWriteTimeUtc
|
|
||||||
$out_err_mtime = (get-item $out_err).LastWriteTimeUtc
|
|
||||||
$out_mtime = if ($out_txt_mtime -gt $out_err_mtime) { $out_txt_mtime } else { $out_err_mtime }
|
|
||||||
$meta_mtime = (get-item $metadata).LastWriteTimeUtc
|
|
||||||
if (($src_mtime -gt $out_mtime) -or ($meta_mtime -gt $out_mtime)) {
|
|
||||||
$any_stale = $true
|
|
||||||
break
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if (-not $any_stale) {
|
|
||||||
write-host "AtomAnnotations all $($sources.Count) source(s) up-to-date" -ForegroundColor DarkGray
|
|
||||||
return
|
|
||||||
}
|
|
||||||
|
|
||||||
write-host "AtomAnnotations $($sources.Count) source(s)" -ForegroundColor Magenta
|
|
||||||
& lua $gen_atom_annot_script $metadata @sources
|
|
||||||
if ($LASTEXITCODE -ne 0) {
|
|
||||||
write-error "Atom annotation generation failed. Aborting."
|
|
||||||
exit 1
|
|
||||||
}
|
|
||||||
|
|
||||||
# If any source produced annotation errors, surface them now and halt the
|
|
||||||
# build. The errors.h files are also #include'd via -include below, so
|
|
||||||
# the C build would fail at preprocessing time anyway — failing here gives
|
|
||||||
# a more readable error in the build log.
|
|
||||||
$err_count = 0
|
|
||||||
foreach ($src in $sources) {
|
|
||||||
$basename = [System.IO.Path]::GetFileNameWithoutExtension($src)
|
|
||||||
$dir = split-path -Path $src -Parent
|
|
||||||
$gen_dir = join-path $dir 'gen'
|
|
||||||
$ann_txt = join-path $gen_dir "$basename.annotations.txt"
|
|
||||||
$err_h = join-path $gen_dir "$basename.errors.h"
|
|
||||||
if ((test-path $ann_txt) -and (test-path $err_h)) {
|
|
||||||
$txt = get-content $ann_txt -raw
|
|
||||||
if ($txt -match 'Errors:\s+([1-9]\d*)') {
|
|
||||||
$err_count += [int]$Matches[1]
|
|
||||||
write-warning "Annotation errors in $src — see $ann_txt"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if ($err_count -gt 0) {
|
|
||||||
write-error "Annotation pass failed: $err_count error(s) across $($sources.Count) source(s). Aborting."
|
|
||||||
exit 1
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
function build-gte_hello {
|
|
||||||
$includes += @()
|
$includes += @()
|
||||||
|
|
||||||
$path_module = join-path $path_code 'gte_hello'
|
$path_module = join-path $path_code 'hello_gte'
|
||||||
$path_duffle = join-path $path_code 'duffle'
|
$path_duffle = join-path $path_code 'duffle'
|
||||||
$path_atom_metadata = join-path $path_module 'tape_atom.metadata.h'
|
$path_atom_metadata = join-path $path_duffle 'word_count.metadata.h'
|
||||||
|
$path_build_gen = join-path $path_build 'gen'
|
||||||
|
|
||||||
$source_dirs = @($path_duffle, $path_module)
|
$src_c = join-path $path_module 'hello_gte.c'
|
||||||
$atom_sources = Get-SourceFiles -paths $source_dirs -extensions @('.h', '.c')
|
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen
|
||||||
|
|
||||||
generate-TapeAtomAnnotations -sources $atom_sources -metadata $path_atom_metadata
|
|
||||||
generate-TapeAtomOffsets -sources $atom_sources -metadata $path_atom_metadata
|
|
||||||
|
|
||||||
$assemble_args = @()
|
$assemble_args = @()
|
||||||
$assemble_args += $f_debug
|
$assemble_args += $f_debug
|
||||||
@@ -433,14 +419,13 @@ function build-gte_hello {
|
|||||||
|
|
||||||
$src_asm_crt = join-path $path_nugget_common 'crt0/crt0.s'
|
$src_asm_crt = join-path $path_nugget_common 'crt0/crt0.s'
|
||||||
$module_asm_crt = join-path $path_build 'crt0.o'
|
$module_asm_crt = join-path $path_build 'crt0.o'
|
||||||
# assemble-unit $src_asm_crt $module_asm_crt $includes $assemble_args
|
assemble-unit $src_asm_crt $module_asm_crt $includes $assemble_args
|
||||||
|
|
||||||
# $src_asm = join-path $path_module 'hello_gte.s'
|
# $src_asm = join-path $path_module 'hello_gte.s'
|
||||||
# $module_asm = join-path $path_build 'hello_gte.o'
|
# $module_asm = join-path $path_build 'hello_gte.o'
|
||||||
|
|
||||||
# assemble-unit $src_asm $module_asm $includes $assemble_args
|
# assemble-unit $src_asm $module_asm $includes $assemble_args
|
||||||
|
|
||||||
$src_c = join-path $path_module 'hello_gte.c'
|
|
||||||
$module_c = join-path $path_build 'hello_gte_c.o'
|
$module_c = join-path $path_build 'hello_gte_c.o'
|
||||||
|
|
||||||
$compile_args = @()
|
$compile_args = @()
|
||||||
@@ -458,25 +443,131 @@ function build-gte_hello {
|
|||||||
$link_args = @()
|
$link_args = @()
|
||||||
$link_args += $f_debug
|
$link_args += $f_debug
|
||||||
# $link_args += $f_optimize_size
|
# $link_args += $f_optimize_size
|
||||||
link-modules @($module_asm_crt, $module_c) $elf $link_args
|
$link_modules = @(
|
||||||
|
$module_asm_crt,
|
||||||
|
$module_c
|
||||||
|
)
|
||||||
|
link-modules $link_modules $elf $link_args
|
||||||
make-binary $elf $exe
|
make-binary $elf $exe
|
||||||
}
|
|
||||||
build-gte_hello
|
|
||||||
|
|
||||||
|
# Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start).
|
||||||
|
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
|
||||||
|
|
||||||
|
inject-dwarf $elf $path_build_gen
|
||||||
|
}
|
||||||
|
# build-hello_gte
|
||||||
|
|
||||||
|
function build-hello_joypad {
|
||||||
|
$includes += @()
|
||||||
|
|
||||||
|
$path_module = join-path $path_code 'hello_joypad'
|
||||||
|
$path_duffle = join-path $path_code 'duffle'
|
||||||
|
$path_atom_metadata = join-path $path_duffle 'word_count.metadata.h'
|
||||||
|
$path_build_gen = join-path $path_build 'gen'
|
||||||
|
|
||||||
|
$src_c = join-path $path_module 'hello_joypad.c'
|
||||||
|
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen
|
||||||
|
|
||||||
|
$assemble_args = @()
|
||||||
|
$assemble_args += $f_debug
|
||||||
|
$assemble_args += $f_optimize_none
|
||||||
|
$assemble_args += ($f_include + $path_code)
|
||||||
|
|
||||||
|
$src_asm_crt = join-path $path_nugget_common 'crt0/crt0.s'
|
||||||
|
$module_asm_crt = join-path $path_build 'crt0.o'
|
||||||
|
assemble-unit $src_asm_crt $module_asm_crt $includes $assemble_args
|
||||||
|
|
||||||
|
$module_c = join-path $path_build 'hello_joypad_c.o'
|
||||||
|
|
||||||
|
$compile_args = @()
|
||||||
|
$compile_args += $f_debug
|
||||||
|
$compile_args += $f_optimize_none
|
||||||
|
# $compile_args += $f_optimize_intrinsics
|
||||||
|
# $compile_args += $f_optimize_size
|
||||||
|
# $compile_args += $f_optimize_debug
|
||||||
|
$compile_args += ($f_include + $path_code)
|
||||||
|
compile-unit $src_c $module_c $includes $compile_args
|
||||||
|
|
||||||
|
$elf = join-path $path_build 'hello_joypad.elf'
|
||||||
|
$exe = join-path $path_build 'hello_joypad.ps-exe'
|
||||||
|
|
||||||
|
$link_args = @()
|
||||||
|
$link_args += $f_debug
|
||||||
|
# $link_args += $f_optimize_size
|
||||||
|
$link_modules = @(
|
||||||
|
$module_asm_crt,
|
||||||
|
$module_c
|
||||||
|
)
|
||||||
|
link-modules $link_modules $elf $link_args
|
||||||
|
make-binary $elf $exe
|
||||||
|
|
||||||
|
# Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start).
|
||||||
|
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
|
||||||
|
|
||||||
|
inject-dwarf $elf $path_build_gen
|
||||||
|
}
|
||||||
|
# build-hello_joypad
|
||||||
|
|
||||||
|
function build-hello_camera {
|
||||||
|
$includes += @()
|
||||||
|
|
||||||
|
$path_module = join-path $path_code 'hello_camera'
|
||||||
|
$path_duffle = join-path $path_code 'duffle'
|
||||||
|
$path_atom_metadata = join-path $path_duffle 'word_count.metadata.h'
|
||||||
|
$path_build_gen = join-path $path_build 'gen'
|
||||||
|
|
||||||
|
$src_c = join-path $path_module 'hello_camera.c'
|
||||||
|
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--pre-link')
|
||||||
|
|
||||||
|
$assemble_args = @()
|
||||||
|
$assemble_args += $f_debug
|
||||||
|
$assemble_args += $f_optimize_none
|
||||||
|
$assemble_args += ($f_include + $path_code)
|
||||||
|
|
||||||
|
$src_asm_crt = join-path $path_nugget_common 'crt0/crt0.s'
|
||||||
|
$module_asm_crt = join-path $path_build 'crt0.o'
|
||||||
|
assemble-unit $src_asm_crt $module_asm_crt $includes $assemble_args
|
||||||
|
|
||||||
|
$module_c = join-path $path_build 'hello_camera_c.o'
|
||||||
|
|
||||||
|
$compile_args = @()
|
||||||
|
$compile_args += $f_debug
|
||||||
|
$compile_args += ($f_define + 'BUILD_DEBUG')
|
||||||
|
$compile_args += $f_optimize_none
|
||||||
|
# $compile_args += $f_optimize_intrinsics
|
||||||
|
# $compile_args += $f_optimize_size
|
||||||
|
# $compile_args += $f_optimize_debug
|
||||||
|
$compile_args += ($f_include + $path_code)
|
||||||
|
compile-unit $src_c $module_c $includes $compile_args
|
||||||
|
|
||||||
|
$elf = join-path $path_build 'hello_camera.elf'
|
||||||
|
$exe = join-path $path_build 'hello_camera.ps-exe'
|
||||||
|
|
||||||
|
$link_args = @()
|
||||||
|
$link_args += $f_debug
|
||||||
|
# $link_args += $f_optimize_size
|
||||||
|
$link_modules = @(
|
||||||
|
$module_asm_crt,
|
||||||
|
$module_c
|
||||||
|
)
|
||||||
|
link-modules $link_modules $elf $link_args
|
||||||
|
make-binary $elf $exe
|
||||||
|
|
||||||
|
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
|
||||||
|
|
||||||
|
inject-dwarf $elf $path_build_gen
|
||||||
|
}
|
||||||
|
build-hello_camera
|
||||||
|
|
||||||
# NO idea if this works yet...
|
# NO idea if this works yet...
|
||||||
function Send-ToEmulator { param(
|
function Send-ToEmulator { param( [string]$exePath )
|
||||||
[string]$exePath
|
|
||||||
)
|
|
||||||
$uri = "http://localhost:8080/api/v1/load-exec"
|
$uri = "http://localhost:8080/api/v1/load-exec"
|
||||||
|
|
||||||
# Absolute path is safest for the emulator web server
|
# Absolute path is safest for the emulator web server
|
||||||
$absolutePath = [System.IO.Path]::GetFullPath($exePath)
|
$absolutePath = [System.IO.Path]::GetFullPath($exePath)
|
||||||
|
|
||||||
# Create JSON payload pointing to your compiled .ps-exe
|
# Create JSON payload pointing to your compiled .ps-exe
|
||||||
$body = @{
|
$body = @{ filename = $absolutePath } | ConvertTo-Json
|
||||||
filename = $absolutePath
|
|
||||||
} | ConvertTo-Json
|
|
||||||
|
|
||||||
Write-Host "Pushing hot-reload to PCSX-Redux..." -ForegroundColor Magenta
|
Write-Host "Pushing hot-reload to PCSX-Redux..." -ForegroundColor Magenta
|
||||||
try {
|
try {
|
||||||
|
|||||||
@@ -0,0 +1,91 @@
|
|||||||
|
--- duffle.lua — facade over duffle_scan / duffle_isa / duffle_emit.
|
||||||
|
|
||||||
|
--- @class DuffleExport
|
||||||
|
--- bag: open module-export keys from duffle_scan / duffle_isa / duffle_emit
|
||||||
|
|
||||||
|
local scan = require("duffle_scan") ---@type DuffleExport
|
||||||
|
local isa = require("duffle_isa") ---@type DuffleExport
|
||||||
|
local emit = require("duffle_emit") ---@type DuffleExport
|
||||||
|
local M = {} ---@type DuffleExport
|
||||||
|
|
||||||
|
--- @alias Path string
|
||||||
|
--- @alias LineNum integer
|
||||||
|
--- @alias ByteOff integer
|
||||||
|
--- @alias MacroName string
|
||||||
|
--- @alias AtomName string
|
||||||
|
--- @alias Severity string
|
||||||
|
|
||||||
|
--- @class SourceFile
|
||||||
|
--- @field path Path
|
||||||
|
--- @field text string
|
||||||
|
--- @field dir string
|
||||||
|
--- @field basename string
|
||||||
|
--- @field scan SourceScan|nil
|
||||||
|
|
||||||
|
--- @class CorpusView
|
||||||
|
--- @field register_alias_registry table<string, AliasEntry>
|
||||||
|
--- @field type_name_registry table<string, TypeNameEntry>
|
||||||
|
--- @field atom_views table<AtomName, AtomViewEntry>
|
||||||
|
--- @field atom_ctxs table<AtomName, AtomCtxEntry>
|
||||||
|
--- @field atom_phases table<string, AtomPhaseGroup>
|
||||||
|
--- @field binds_by_name table<string, BindsEntry>
|
||||||
|
--- @field atoms_by_name table<AtomName, AtomEntry>
|
||||||
|
--- @field atom_infos AtomInfoEntry[]
|
||||||
|
--- @field components table<string, Component>
|
||||||
|
--- @field component_atom_infos AtomInfoEntry[]|nil
|
||||||
|
--- @field tape_chains table<string, TapeChain>|nil
|
||||||
|
--- @field source_order SourceFile[]
|
||||||
|
--- @field collisions CorpusCollision[]
|
||||||
|
|
||||||
|
--- @param src DuffleExport
|
||||||
|
--- @param label string
|
||||||
|
--- @return nil
|
||||||
|
local function merge(src, label)
|
||||||
|
for k, v in pairs(src) do ---@type string, any
|
||||||
|
if M[k] ~= nil and M[k] ~= v then
|
||||||
|
error("duffle facade name collision on " .. tostring(k) .. " from " .. label, 0)
|
||||||
|
end
|
||||||
|
M[k] = v
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
merge(scan, "duffle_scan")
|
||||||
|
merge(isa, "duffle_isa")
|
||||||
|
merge(emit, "duffle_emit")
|
||||||
|
|
||||||
|
--- @param ctx PassCtx
|
||||||
|
--- @return CorpusView
|
||||||
|
function M.corpus_view(ctx)
|
||||||
|
local corpus = ctx and ctx.shared and ctx.shared.corpus ---@type Corpus
|
||||||
|
if not corpus then error("requires ctx.shared.corpus", 0) end
|
||||||
|
return {
|
||||||
|
register_alias_registry = corpus.register_alias_registry or {},
|
||||||
|
type_name_registry = corpus.type_name_registry or {},
|
||||||
|
atom_views = corpus.atom_views or {},
|
||||||
|
atom_ctxs = corpus.atom_ctxs or {},
|
||||||
|
atom_phases = corpus.atom_phases or {},
|
||||||
|
binds_by_name = corpus.binds_by_name or {},
|
||||||
|
atoms_by_name = corpus.atoms_by_name or {},
|
||||||
|
atom_infos = corpus.atom_infos or {},
|
||||||
|
components = corpus.components or {},
|
||||||
|
component_atom_infos = corpus.component_atom_infos or {},
|
||||||
|
tape_chains = corpus.tape_chains or {},
|
||||||
|
source_order = corpus.source_order or {},
|
||||||
|
collisions = corpus.collisions or {},
|
||||||
|
}
|
||||||
|
end
|
||||||
|
|
||||||
|
--- @param rules CheckRule[]
|
||||||
|
--- @param phase string
|
||||||
|
--- @param item AtomEntry|SourceFile
|
||||||
|
--- @param pipe_ctx PassScratch
|
||||||
|
--- @param findings Finding[]
|
||||||
|
--- @return nil
|
||||||
|
function M.run_check_rules(rules, phase, item, pipe_ctx, findings)
|
||||||
|
for _, rule in ipairs(rules) do ---@type integer, CheckRule
|
||||||
|
local fn = rule[phase] ---@type (fun(item: AtomEntry|SourceFile, pipe_ctx: PassScratch, findings: Finding[]): nil)|nil
|
||||||
|
if fn then fn(item, pipe_ctx, findings) end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return M
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,866 @@
|
|||||||
|
--- duffle_isa.lua — encoder / GTE / hardware tables.
|
||||||
|
|
||||||
|
--- @class InstructionImm
|
||||||
|
--- @field arg integer
|
||||||
|
--- @field signed boolean|nil
|
||||||
|
--- @field width integer
|
||||||
|
|
||||||
|
--- @class InstructionValue
|
||||||
|
--- @field dest integer
|
||||||
|
--- @field op string
|
||||||
|
--- @field sources integer[]|nil
|
||||||
|
--- @field immediate integer|nil
|
||||||
|
--- @field source integer|nil
|
||||||
|
|
||||||
|
--- @class InstructionRow
|
||||||
|
--- @field cycles integer
|
||||||
|
--- @field kind string
|
||||||
|
--- @field reads integer[]|nil
|
||||||
|
--- @field writes integer[]|nil
|
||||||
|
--- @field imm InstructionImm[]|nil
|
||||||
|
--- @field value InstructionValue|nil
|
||||||
|
--- @field delay_slot boolean|nil
|
||||||
|
--- @field suppress_arg1 table<string, string>|nil -- bag: GPR ident -> reason
|
||||||
|
|
||||||
|
--- @class TapeAtomMacroRow
|
||||||
|
--- @field kind string
|
||||||
|
--- @field binds boolean
|
||||||
|
|
||||||
|
--- @class GteCommandPort
|
||||||
|
--- @field register string
|
||||||
|
--- @field role string
|
||||||
|
|
||||||
|
--- @class GteCommandLatch
|
||||||
|
--- @field register string
|
||||||
|
--- @field required integer
|
||||||
|
|
||||||
|
--- @class GteCommandRow
|
||||||
|
--- @field aliases string[]
|
||||||
|
--- @field cycles integer
|
||||||
|
--- @field inputs string[]
|
||||||
|
--- @field outputs GteCommandPort[]
|
||||||
|
--- @field latch GteCommandLatch[]
|
||||||
|
|
||||||
|
--- @class GteCrAliasGroup
|
||||||
|
--- @field [1] integer -- C2 control-register slot
|
||||||
|
--- @field [2] string[] -- aliases that share that slot
|
||||||
|
|
||||||
|
--- @class GtePackedSlotRelation
|
||||||
|
--- @field slot integer
|
||||||
|
--- @field first string
|
||||||
|
--- @field second string
|
||||||
|
|
||||||
|
--- @class HardwareRelationPort
|
||||||
|
--- @field domain string
|
||||||
|
--- @field arg integer
|
||||||
|
|
||||||
|
--- @class HardwareRelationVisibility
|
||||||
|
--- @field kind string
|
||||||
|
--- @field required integer
|
||||||
|
|
||||||
|
--- @class HardwareRelationEvidence
|
||||||
|
--- @field confidence string
|
||||||
|
--- @field source string
|
||||||
|
|
||||||
|
--- @class HardwareRelationRow
|
||||||
|
--- @field id string
|
||||||
|
--- @field semantic string
|
||||||
|
--- @field consumer string
|
||||||
|
--- @field token string
|
||||||
|
--- @field direction string
|
||||||
|
--- @field reads HardwareRelationPort
|
||||||
|
--- @field writes HardwareRelationPort
|
||||||
|
--- @field visibility HardwareRelationVisibility|nil
|
||||||
|
--- @field evidence HardwareRelationEvidence
|
||||||
|
--- @field violation_kind string
|
||||||
|
--- @field destination_match string|nil
|
||||||
|
--- @field fanout_to string[]|nil
|
||||||
|
--- @field required integer|nil
|
||||||
|
--- @field clear_on_consumer boolean|nil
|
||||||
|
--- @field stage boolean|nil
|
||||||
|
--- @field cu2_transition boolean|nil
|
||||||
|
--- @field status_register integer|nil
|
||||||
|
|
||||||
|
--- @class Cu2TransitionPolicy
|
||||||
|
--- @field status_register integer
|
||||||
|
--- @field enable_bit integer
|
||||||
|
--- @field required integer
|
||||||
|
--- @field visibility_kind string
|
||||||
|
--- @field evidence HardwareRelationEvidence
|
||||||
|
|
||||||
|
--- @class GprRole
|
||||||
|
--- @field name string
|
||||||
|
--- @field pool boolean
|
||||||
|
--- @field optional boolean
|
||||||
|
--- @field carrier boolean
|
||||||
|
|
||||||
|
--- @class DuffleIsa
|
||||||
|
--- @field GPR_ROLE table<string, GprRole>
|
||||||
|
--- @field TAPE_ATOM_MACROS table<string, TapeAtomMacroRow>
|
||||||
|
--- @field DELAY_MARKERS table<string, boolean>
|
||||||
|
--- @field INSTRUCTION table<string, InstructionRow>
|
||||||
|
--- @field GTE_COMMAND table<string, GteCommandRow>
|
||||||
|
--- @field ALIAS_TO_CANONICAL table<string, string>
|
||||||
|
--- @field instr fun(ident: string): InstructionRow|nil
|
||||||
|
--- @field gte_canon fun(ident: string): string
|
||||||
|
--- @field gte fun(ident: string): GteCommandRow|nil
|
||||||
|
--- @field GTE_CR_ALIAS_GROUPS GteCrAliasGroup[]
|
||||||
|
--- @field GTE_PACKED_SLOT_RELATIONS GtePackedSlotRelation[]
|
||||||
|
--- @field OPERAND_READ_POSITIONS table<string, integer[]>
|
||||||
|
--- @field GP0_CMD_SIZE table<integer, integer>
|
||||||
|
--- @field GP0_CMD_BY_SHAPE table<string, integer>
|
||||||
|
--- @field UNKNOWN_INSTRUCTION_CYCLES integer
|
||||||
|
--- @field HARDWARE_RELATIONS HardwareRelationRow[]
|
||||||
|
--- @field CU2_TRANSITION_POLICY Cu2TransitionPolicy
|
||||||
|
|
||||||
|
local M = {} ---@type DuffleIsa
|
||||||
|
|
||||||
|
-- Section 7: domain tables
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
-- One GprRole row per name. Construction order is the auto_reg pool order,
|
||||||
|
-- then R_AT, then the three carriers. Index by name into M.GPR_ROLE.
|
||||||
|
--- @type table<string, GprRole>
|
||||||
|
M.GPR_ROLE = {
|
||||||
|
{ name = "R_V0", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_V1", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_T0", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_T1", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_T2", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_T3", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_T4", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_T5", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_T6", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_T7", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_A0", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_A1", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_A2", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_A3", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_S0", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_S1", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_S2", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_S3", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_S4", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_S5", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_S6", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_S7", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_T8", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_T9", pool = true, optional = true, carrier = false },
|
||||||
|
{ name = "R_AT", pool = false, optional = true, carrier = false },
|
||||||
|
{ name = "R_TapePtr", pool = false, optional = true, carrier = true },
|
||||||
|
{ name = "R_AtomJmp", pool = false, optional = true, carrier = true },
|
||||||
|
{ name = "R_ScratchBase", pool = false, optional = true, carrier = true },
|
||||||
|
}
|
||||||
|
for _, row in ipairs(M.GPR_ROLE) do ---@type integer, GprRole
|
||||||
|
M.GPR_ROLE[row.name] = row
|
||||||
|
end
|
||||||
|
|
||||||
|
-- atom_info sub-calls: atom_bind, atom_reads, atom_writes, atom_view, atom_reg_types, atom_ctx, atom_phase.
|
||||||
|
--- @type table<string, TapeAtomMacroRow>
|
||||||
|
M.TAPE_ATOM_MACROS = {
|
||||||
|
["atom_info"] = { kind = "info", binds = false },
|
||||||
|
}
|
||||||
|
|
||||||
|
-- Empty C macros that prefix the next encoder. Zero words.
|
||||||
|
-- BdSlot_ nop is one nop word. The marker is not the BD instruction.
|
||||||
|
--- @type table<string, boolean> -- bag: marker prefix -> true
|
||||||
|
M.DELAY_MARKERS = {
|
||||||
|
["GteDelay_"] = true,
|
||||||
|
["LdSlot_"] = true,
|
||||||
|
["BdSlot_"] = true,
|
||||||
|
["DmaSlot_"] = true,
|
||||||
|
}
|
||||||
|
|
||||||
|
-- One row per encoder. Read through duffle.instr.
|
||||||
|
--- @type table<string, InstructionRow>
|
||||||
|
M.INSTRUCTION = {
|
||||||
|
["BdSlot_"] = { cycles = 0, kind = "marker", },
|
||||||
|
["LdSlot_"] = { cycles = 0, kind = "marker", },
|
||||||
|
["add_s"] = { cycles = 1, kind = "alu", },
|
||||||
|
["add_si"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, },}, },
|
||||||
|
["add_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["add_u_self"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, value = { dest = 1, op = "add_u", sources = { 1, 2 }, }, },
|
||||||
|
["add_ui"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, value = { dest = 1, immediate = 3, op = "add_ui", source = 2, }, },
|
||||||
|
["add_ui_self"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = { 1 }, imm = { { arg = 2, signed = true, width = 16, }, }, value = { dest = 1, immediate = 2, op = "add_ui", source = 1, }, },
|
||||||
|
["and"] = { cycles = 1, kind = "alu", },
|
||||||
|
["and_i"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, width = 16, }, }, value = { dest = 1, immediate = 3, op = "and_i", source = 2, }, },
|
||||||
|
["and_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["atom_bind"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||||
|
["atom_info"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||||
|
["atom_label"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||||
|
["atom_offset"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||||
|
["atom_reads"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||||
|
["atom_writes"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||||
|
["branch_equal"] = { cycles = 2, kind = "branch", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["branch_ge_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
|
||||||
|
["branch_gt_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
|
||||||
|
["branch_le_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
|
||||||
|
["branch_lt_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
|
||||||
|
["branch_ne"] = { cycles = 2, kind = "branch", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["call_addr"] = { cycles = 2, kind = "call", reads = {}, writes = { 1 }, },
|
||||||
|
["call_reg"] = { cycles = 2, kind = "call", reads = { 1 }, writes = { 2 }, },
|
||||||
|
["div_s"] = { cycles = 35, kind = "alu", reads = { 1, 2 }, writes = {}, },
|
||||||
|
["div_u"] = { cycles = 35, kind = "alu", reads = { 1, 2 }, writes = {}, },
|
||||||
|
["gte_load_v0"] = { cycles = 2, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
|
||||||
|
["gte_load_v0v1v2"] = { cycles = 6, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
|
||||||
|
["gte_load_v1"] = { cycles = 2, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
|
||||||
|
["gte_load_v2"] = { cycles = 2, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
|
||||||
|
["gte_lw"] = { cycles = 1, kind = "load", reads = { 2 }, writes = {}, },
|
||||||
|
["gte_lwc2"] = { cycles = 1, kind = "load", },
|
||||||
|
["gte_mv_from_ctrl_r"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = { 1 }, },
|
||||||
|
["gte_mv_from_data_r"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = { 1 }, },
|
||||||
|
["gte_mv_to_ctrl_r"] = { cycles = 1, kind = "cop2_xfer", reads = { 1 }, writes = {}, },
|
||||||
|
["gte_mv_to_data_r"] = { cycles = 1, kind = "cop2_xfer", reads = { 1 }, writes = {}, },
|
||||||
|
["gte_stotz"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = {}, },
|
||||||
|
["gte_stsxy3"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = {}, },
|
||||||
|
["gte_sw"] = { cycles = 1, kind = "store", reads = { 2 }, writes = {}, },
|
||||||
|
["gte_swc2"] = { cycles = 1, kind = "store", },
|
||||||
|
["jump"] = { cycles = 2, kind = "jump", reads = {}, writes = {}, },
|
||||||
|
["jump_link"] = { cycles = 2, kind = "call", reads = { 1 }, writes = { 2 }, },
|
||||||
|
["jump_reg"] = { cycles = 2, kind = "jump", reads = { 1 }, writes = {}, suppress_arg1 = { R_AtomJmp = "fixed mac_yield handshake", }, },
|
||||||
|
["jump_rel"] = { cycles = 2, kind = "branch", delay_slot = true, },
|
||||||
|
["li_s"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, value = { dest = 1, immediate = 3, op = "add_ui", source = 2, }, },
|
||||||
|
["load_byte"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["load_byte_u"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["load_half"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["load_half_u"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["load_imm"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, },
|
||||||
|
["load_ui"] = { cycles = 1, kind = "alu", reads = {}, writes = { 1 }, },
|
||||||
|
["load_upper_i"] = { cycles = 1, kind = "alu", reads = {}, writes = { 1 }, imm = { { arg = 2, width = 16, }, }, value = { dest = 1, immediate = 2, op = "load_upper_i", }, },
|
||||||
|
["load_word"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["mac_yield"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||||
|
["mask_upper"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, },
|
||||||
|
["mov_from_high"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, },
|
||||||
|
["mov_from_low"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, },
|
||||||
|
["mov_to_high"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = {}, },
|
||||||
|
["mov_to_low"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = {}, },
|
||||||
|
["mult_s"] = { cycles = 12, kind = "alu", reads = { 1, 2 }, writes = {}, },
|
||||||
|
["mult_u"] = { cycles = 12, kind = "alu", reads = { 1, 2 }, writes = {}, },
|
||||||
|
["nop"] = { cycles = 1, kind = "nop", reads = {}, writes = {}, },
|
||||||
|
["nop2"] = { cycles = 2, kind = "nop", reads = {}, writes = {}, },
|
||||||
|
["nor_u"] = { cycles = 1, kind = "alu", },
|
||||||
|
["or_i"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, width = 16, }, }, value = { dest = 1, immediate = 3, op = "or_i", source = 2, }, },
|
||||||
|
["or_i_self"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = { 1 }, imm = { { arg = 2, width = 16, }, }, value = { dest = 1, immediate = 2, op = "or_i", source = 1, }, },
|
||||||
|
["or_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["or_u_self"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, value = { dest = 1, op = "or", sources = { 1, 2 }, }, },
|
||||||
|
["set_lt_s"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["set_lt_si"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, },
|
||||||
|
["set_lt_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["set_lt_ui"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, },
|
||||||
|
["shift_aright"] = { cycles = 1, kind = "alu", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
|
||||||
|
["shift_aright_var"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
|
||||||
|
["shift_lleft"] = { cycles = 1, kind = "alu", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
|
||||||
|
["shift_lleft_self"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = { 1 }, imm = { { arg = 2, width = 5, }, }, value = { dest = 1, immediate = 2, op = "shift_lleft", source = 1, }, },
|
||||||
|
["shift_lleft_var"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["shift_lright"] = { cycles = 1, kind = "alu", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
|
||||||
|
["slt_s"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["slt_si"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["slt_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["slt_ui"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16,}, }, },
|
||||||
|
["store_byte"] = { cycles = 1, kind = "store", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["store_half"] = { cycles = 1, kind = "store", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["store_word"] = { cycles = 1, kind = "store", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["sub_s"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["sub_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["sys_mov_from_cop0"] = { cycles = 1, kind = "cop0_xfer", reads = {}, writes = { 1 }, },
|
||||||
|
["sys_mov_to_cop0"] = { cycles = 1, kind = "cop0_xfer", reads = { 1 }, writes = {}, },
|
||||||
|
["xor_i"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, width = 16, }, }, value = { dest = 1, immediate = 3, op = "xor_i", source = 2, }, },
|
||||||
|
["xor_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
}
|
||||||
|
|
||||||
|
-- One row per GTE command. Alias cycle numbers live here, not on INSTRUCTION.
|
||||||
|
--- @type table<string, GteCommandRow>
|
||||||
|
M.GTE_COMMAND = {
|
||||||
|
["gte_cmdw_avsz3"] = {
|
||||||
|
aliases = { "gte_avg_sort_z3", "gte_avsz3", "gte_cmdw_avg_sort_z3" },
|
||||||
|
cycles = 5,
|
||||||
|
inputs = { "C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3", "gte_cr_ZSF3" },
|
||||||
|
outputs = {
|
||||||
|
{ register = "C2_OTZ", role = "otz", },
|
||||||
|
},
|
||||||
|
latch = {
|
||||||
|
{ register = "C2_OTZ", required = 4, },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
["gte_cmdw_avsz4"] = {
|
||||||
|
aliases = { "gte_avg_sort_z4", "gte_avsz4", "gte_cmdw_avg_sort_z4" },
|
||||||
|
cycles = 6,
|
||||||
|
inputs = { "C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3", "gte_cr_ZSF4" },
|
||||||
|
outputs = {
|
||||||
|
{ register = "C2_OTZ", role = "otz", },
|
||||||
|
},
|
||||||
|
latch = {
|
||||||
|
{ register = "C2_OTZ", required = 4, },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
["gte_cmdw_gpf"] = {
|
||||||
|
aliases = {},
|
||||||
|
cycles = 5,
|
||||||
|
inputs = { "C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3" },
|
||||||
|
outputs = {
|
||||||
|
{ register = "C2_MAC1", role = "mac_result", },
|
||||||
|
{ register = "C2_MAC2", role = "mac_result", },
|
||||||
|
{ register = "C2_MAC3", role = "mac_result", },
|
||||||
|
{ register = "C2_IR1", role = "latest_color", },
|
||||||
|
{ register = "C2_IR2", role = "latest_color", },
|
||||||
|
{ register = "C2_IR3", role = "latest_color", },
|
||||||
|
},
|
||||||
|
latch = {
|
||||||
|
{ register = "C2_MAC1", required = 4, },
|
||||||
|
{ register = "C2_MAC2", required = 4, },
|
||||||
|
{ register = "C2_MAC3", required = 4, },
|
||||||
|
{ register = "C2_IR1", required = 4, },
|
||||||
|
{ register = "C2_IR2", required = 4, },
|
||||||
|
{ register = "C2_IR3", required = 4, },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
["gte_cmdw_mvmva"] = {
|
||||||
|
aliases = {},
|
||||||
|
cycles = 8,
|
||||||
|
inputs = {
|
||||||
|
"C2_VXY0", "C2_VZ0",
|
||||||
|
"C2_VXY1", "C2_VZ1",
|
||||||
|
"C2_VXY2", "C2_VZ2",
|
||||||
|
"C2_IR1", "C2_IR2", "C2_IR3",
|
||||||
|
"gte_cr_RT11", "gte_cr_RT12", "gte_cr_RT13",
|
||||||
|
"gte_cr_RT21", "gte_cr_RT22", "gte_cr_RT23",
|
||||||
|
"gte_cr_RT31", "gte_cr_RT32", "gte_cr_RT33",
|
||||||
|
"gte_cr_TRX", "gte_cr_TRY", "gte_cr_TRZ"
|
||||||
|
},
|
||||||
|
outputs = {
|
||||||
|
{ register = "C2_IR1", role = "latest_color", },
|
||||||
|
{ register = "C2_IR2", role = "latest_color", },
|
||||||
|
{ register = "C2_IR3", role = "latest_color", },
|
||||||
|
},
|
||||||
|
latch = {
|
||||||
|
{ register = "C2_IR1", required = 4, },
|
||||||
|
{ register = "C2_IR2", required = 4, },
|
||||||
|
{ register = "C2_IR3", required = 4, },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
["gte_cmdw_nclip"] = {
|
||||||
|
aliases = { "gte_nclip" },
|
||||||
|
cycles = 8,
|
||||||
|
inputs = { "C2_SXY0", "C2_SXY1", "C2_SXY2" },
|
||||||
|
outputs = {
|
||||||
|
{ register = "C2_SZ3", role = "mac_result", },
|
||||||
|
},
|
||||||
|
latch = {
|
||||||
|
{ register = "C2_SZ3", required = 4, },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
["gte_cmdw_op"] = {
|
||||||
|
aliases = { "gte_cmdw_outer_product", "gte_cmdw_wedge" },
|
||||||
|
cycles = 6,
|
||||||
|
inputs = {},
|
||||||
|
outputs = {
|
||||||
|
{ register = "C2_IR1", role = "latest_color", },
|
||||||
|
{ register = "C2_IR2", role = "latest_color", },
|
||||||
|
{ register = "C2_IR3", role = "latest_color", },
|
||||||
|
},
|
||||||
|
latch = {
|
||||||
|
{ register = "C2_IR1", required = 4, },
|
||||||
|
{ register = "C2_IR2", required = 4, },
|
||||||
|
{ register = "C2_IR3", required = 4, },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
["gte_cmdw_rtps"] = {
|
||||||
|
aliases = { "gte_cmdw_rotate_translate_perspective_single", "gte_rtps" },
|
||||||
|
cycles = 15,
|
||||||
|
inputs = {
|
||||||
|
"C2_VXY0", "C2_VZ0",
|
||||||
|
"C2_VXY1", "C2_VZ1",
|
||||||
|
"C2_VXY2", "C2_VZ2",
|
||||||
|
"C2_RGB", "C2_OTZ",
|
||||||
|
"C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3",
|
||||||
|
"C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3",
|
||||||
|
"gte_cr_RT11", "gte_cr_RT12", "gte_cr_RT13",
|
||||||
|
"gte_cr_RT21", "gte_cr_RT22", "gte_cr_RT23",
|
||||||
|
"gte_cr_RT31", "gte_cr_RT32", "gte_cr_RT33",
|
||||||
|
"gte_cr_TRX", "gte_cr_TRY", "gte_cr_TRZ",
|
||||||
|
"gte_cr_OFX", "gte_cr_OFY",
|
||||||
|
"gte_cr_H",
|
||||||
|
"gte_cr_DQA", "gte_cr_DQB"
|
||||||
|
},
|
||||||
|
outputs = {
|
||||||
|
{ register = "C2_SXY2", role = "latest_screen_xy", },
|
||||||
|
{ register = "C2_SZ2", role = "latest_screen_z", },
|
||||||
|
{ register = "C2_OTZ", role = "otz", },
|
||||||
|
{ register = "C2_IR0", role = "latest_color", },
|
||||||
|
},
|
||||||
|
latch = {
|
||||||
|
{ register = "C2_SXY2", required = 4, },
|
||||||
|
{ register = "C2_SZ2", required = 4, },
|
||||||
|
{ register = "C2_OTZ", required = 4, },
|
||||||
|
{ register = "C2_IR0", required = 4, },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
["gte_cmdw_rtpt"] = {
|
||||||
|
aliases = { "gte_cmdw_rotate_translate_perspective_triple", "gte_rtpt" },
|
||||||
|
cycles = 23,
|
||||||
|
inputs = {
|
||||||
|
"C2_VXY0", "C2_VZ0",
|
||||||
|
"C2_VXY1", "C2_VZ1",
|
||||||
|
"C2_VXY2", "C2_VZ2",
|
||||||
|
"C2_RGB", "C2_OTZ",
|
||||||
|
"C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3",
|
||||||
|
"C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3",
|
||||||
|
"gte_cr_RT11", "gte_cr_RT12", "gte_cr_RT13",
|
||||||
|
"gte_cr_RT21", "gte_cr_RT22", "gte_cr_RT23",
|
||||||
|
"gte_cr_RT31", "gte_cr_RT32", "gte_cr_RT33",
|
||||||
|
"gte_cr_TRX", "gte_cr_TRY", "gte_cr_TRZ",
|
||||||
|
"gte_cr_OFX", "gte_cr_OFY",
|
||||||
|
"gte_cr_H",
|
||||||
|
"gte_cr_DQA", "gte_cr_DQB"
|
||||||
|
},
|
||||||
|
outputs = {
|
||||||
|
{ register = "C2_SXY0", role = "screen_xy[0]", },
|
||||||
|
{ register = "C2_SXY1", role = "screen_xy[1]", },
|
||||||
|
{ register = "C2_SXY2", role = "latest_screen_xy", },
|
||||||
|
{ register = "C2_SZ3", role = "latest_screen_z", },
|
||||||
|
{ register = "C2_OTZ", role = "otz", },
|
||||||
|
},
|
||||||
|
latch = {
|
||||||
|
{ register = "C2_SXY0", required = 4, },
|
||||||
|
{ register = "C2_SXY1", required = 4, },
|
||||||
|
{ register = "C2_SXY2", required = 4, },
|
||||||
|
{ register = "C2_SZ3", required = 4, },
|
||||||
|
{ register = "C2_OTZ", required = 4, },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
["gte_cmdw_sqr"] = {
|
||||||
|
aliases = {},
|
||||||
|
cycles = 5,
|
||||||
|
inputs = { "C2_IR1", "C2_IR2", "C2_IR3" },
|
||||||
|
outputs = {
|
||||||
|
{ register = "C2_MAC1", role = "mac_result", },
|
||||||
|
{ register = "C2_MAC2", role = "mac_result", },
|
||||||
|
{ register = "C2_MAC3", role = "mac_result", },
|
||||||
|
{ register = "C2_IR1", role = "latest_color", },
|
||||||
|
{ register = "C2_IR2", role = "latest_color", },
|
||||||
|
{ register = "C2_IR3", role = "latest_color", },
|
||||||
|
},
|
||||||
|
latch = {
|
||||||
|
{ register = "C2_MAC1", required = 4, },
|
||||||
|
{ register = "C2_MAC2", required = 4, },
|
||||||
|
{ register = "C2_MAC3", required = 4, },
|
||||||
|
{ register = "C2_IR1", required = 4, },
|
||||||
|
{ register = "C2_IR2", required = 4, },
|
||||||
|
{ register = "C2_IR3", required = 4, },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
--- @param ident string
|
||||||
|
--- @return InstructionRow|nil
|
||||||
|
function M.instr (ident) return M.INSTRUCTION [ident] end
|
||||||
|
--- @param ident string
|
||||||
|
--- @return string
|
||||||
|
function M.gte_canon(ident) return M.ALIAS_TO_CANONICAL [ident] or ident end
|
||||||
|
--- @param ident string
|
||||||
|
--- @return GteCommandRow|nil
|
||||||
|
function M.gte (ident) return M.GTE_COMMAND[M.gte_canon(ident)] end
|
||||||
|
|
||||||
|
--- @return nil
|
||||||
|
local function build_alias_map()
|
||||||
|
--- @type table<string, string> -- bag: alias or canon -> canon
|
||||||
|
M.ALIAS_TO_CANONICAL = {}
|
||||||
|
for canon, row in pairs(M.GTE_COMMAND) do ---@type string, GteCommandRow
|
||||||
|
M.ALIAS_TO_CANONICAL[canon] = canon
|
||||||
|
for _, alias in ipairs(row.aliases or {}) do ---@type integer, string
|
||||||
|
M.ALIAS_TO_CANONICAL[alias] = canon
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
build_alias_map()
|
||||||
|
|
||||||
|
|
||||||
|
--- GTE control-register alias groups.
|
||||||
|
--- Aliases within a group write to the same C2 control-register slot (the HW double-maps some C2 slots across multiple PSX SDK / libgte conventions).
|
||||||
|
--- Aliases across groups write to distinct C2 slots.
|
||||||
|
---
|
||||||
|
--- Cross-alias writes inside one atom body, or across the wave-context boundary, silently clobber each other.
|
||||||
|
--- The `check_gte_cr_alias_writes` check warns about each pair per source. See `docs/gte_reference.md` §"Control-register alias table"
|
||||||
|
--- for the HW rationale and the libgte outer-product convention.
|
||||||
|
--- @type GteCrAliasGroup[]
|
||||||
|
M.GTE_CR_ALIAS_GROUPS = {
|
||||||
|
{ 24, { "gte_cr_RBK", "gte_cr_OFX" } }, -- background R vs screen offset X
|
||||||
|
{ 25, { "gte_cr_GBK", "gte_cr_OFY" } }, -- background G vs screen offset Y
|
||||||
|
{ 26, { "gte_cr_BBK", "gte_cr_H" } }, -- background B vs projection plane distance H
|
||||||
|
}
|
||||||
|
|
||||||
|
-- Packed RT slots named by the gte.h packed-slot comment. First must be written before second.
|
||||||
|
--- @type GtePackedSlotRelation[]
|
||||||
|
M.GTE_PACKED_SLOT_RELATIONS = {
|
||||||
|
{ slot = 2, first = "gte_cr_RT13", second = "gte_cr_RT22" },
|
||||||
|
}
|
||||||
|
|
||||||
|
-- Operand-class table for the COP2->GPR load-delay check.
|
||||||
|
-- Maps each emitting-token ident to the set of GPR operand positions it reads.
|
||||||
|
-- Covers the current encoder vocabulary (`code/duffle/mips.h` + `code/duffle/gte.h`); add rows here as new encoders land.
|
||||||
|
--
|
||||||
|
-- Semantics:
|
||||||
|
-- * A "GPR operand position" is the textual slot in the macro's argument list, 1-based; e.g. `load_word(rt, base, off)` has positional operands 1 (rt), 2 (base), 3 (off).
|
||||||
|
-- The table reads operands 1 + 2 + 3 to find what GPRs the macro touches.
|
||||||
|
-- * The check tracks one entry per destination GPR per MFC2 / CFC2 event.
|
||||||
|
-- A subsequent event counts as a "use" iff any of its read operand positions reference that destination GPR's ident (e.g. `R_T0`).
|
||||||
|
-- * Branch delay slots are out of scope (MIPS control-flow; tracked separately).
|
||||||
|
--- @type table<string, integer[]> -- bag: encoder ident -> GPR operand positions
|
||||||
|
M.OPERAND_READ_POSITIONS = {
|
||||||
|
-- CPU ALU with one or two GPR operands. Reads every GPR operand.
|
||||||
|
["add_ui"] = {1, 2},
|
||||||
|
["li_s"] = {1, 2}, -- rt (write), imm16 (immediate)
|
||||||
|
["add_ui_self"] = {1},
|
||||||
|
["add_si"] = {1, 2},
|
||||||
|
["add_u"] = {1, 2, 3},
|
||||||
|
["add_u_self"] = {1, 2},
|
||||||
|
["sub_s"] = {1, 2, 3},
|
||||||
|
["sub_u"] = {1, 2, 3},
|
||||||
|
["and_i"] = {1, 2},
|
||||||
|
["and"] = {1, 2, 3},
|
||||||
|
["or_i"] = {1, 2},
|
||||||
|
["or_i_self"] = {1},
|
||||||
|
["or"] = {1, 2, 3},
|
||||||
|
["or_self"] = {1, 2},
|
||||||
|
["xor_i"] = {1, 2},
|
||||||
|
["xor"] = {1, 2, 3},
|
||||||
|
["slt_s"] = {1, 2, 3},
|
||||||
|
["slt_u"] = {1, 2, 3},
|
||||||
|
["slt_si"] = {1, 2},
|
||||||
|
["slt_ui"] = {1, 2},
|
||||||
|
["mult_s"] = {1, 2},
|
||||||
|
["mult_u"] = {1, 2},
|
||||||
|
["div_s"] = {1, 2},
|
||||||
|
["div_u"] = {1, 2},
|
||||||
|
-- Shifts: shift_lleft(rd, rt, shamt); the rt operand is the value, rd is dest.
|
||||||
|
["shift_lleft"] = {1, 2},
|
||||||
|
["shift_lright"] = {1, 2},
|
||||||
|
["shift_aright"] = {1, 2},
|
||||||
|
["shift_lleft_self"] = {1},
|
||||||
|
-- Loads: load_word(rt, base, off); the rt operand is the destination (it's written, not read) and base + off are non-GPR operands.
|
||||||
|
-- The check treats the rt operand as a write, so the read-positions table for `load_*` is empty.
|
||||||
|
["load_word"] = {},
|
||||||
|
["load_half_u"] = {},
|
||||||
|
["load_byte_u"] = {},
|
||||||
|
["load_half"] = {},
|
||||||
|
["load_byte"] = {},
|
||||||
|
["load_upper_i"] = {},
|
||||||
|
["load_ui"] = {},
|
||||||
|
-- Stores write to memory; base + rt operands are non-read for load-delay purposes.
|
||||||
|
["store_word"] = {},
|
||||||
|
["store_half"] = {},
|
||||||
|
["store_byte"] = {},
|
||||||
|
-- Branches read rs (+ rt for beq/bne). The branch delay slot is out of scope.
|
||||||
|
["branch_equal"] = {1, 2},
|
||||||
|
["branch_ne"] = {1, 2},
|
||||||
|
["branch_le_zero"] = {1},
|
||||||
|
["branch_lt_zero"] = {1},
|
||||||
|
["branch_ge_zero"] = {1},
|
||||||
|
["branch_gt_zero"] = {1},
|
||||||
|
-- Jumps / link: jr / jalr read rs only (the target). RD is the destination link.
|
||||||
|
["jump_reg"] = {1},
|
||||||
|
["jump_link"] = {1},
|
||||||
|
["call_reg"] = {1},
|
||||||
|
["call_addr"] = {},
|
||||||
|
["jump"] = {},
|
||||||
|
-- mask_upper is a 2-word macro: shift_lleft then shift_lright. The first reads rt.
|
||||||
|
["mask_upper"] = {1, 2},
|
||||||
|
-- move from/to HI/LO.
|
||||||
|
["mov_from_high"] = {},
|
||||||
|
["mov_from_low"] = {},
|
||||||
|
["mov_to_high"] = {1},
|
||||||
|
["mov_to_low"] = {1},
|
||||||
|
-- GTE transfers / loads / stores / commands: the relevant table values live in the check itself.
|
||||||
|
-- `gte_mv_to_*` writes its rt operand; `gte_mv_from_*` writes its rt operand; `gte_*` commands are atomic-from-the-CPU-POV
|
||||||
|
-- once they issue (the CPU holds until the command completes, so load-delay violations don't surface here).
|
||||||
|
["gte_mv_from_data_r"] = {},
|
||||||
|
["gte_mv_from_ctrl_r"] = {},
|
||||||
|
["gte_mv_to_data_r"] = {},
|
||||||
|
["gte_mv_to_ctrl_r"] = {},
|
||||||
|
["gte_lw"] = {},
|
||||||
|
["gte_sw"] = {},
|
||||||
|
["shift_lleft_var"] = {1, 2, 3}, -- rd, rt, rs (variable shift amount)
|
||||||
|
["shift_aright_var"] = {1, 2, 3},
|
||||||
|
}
|
||||||
|
|
||||||
|
-- GP0 packet sizes (total words including the 1-word tag) per GP0 cmd byte.
|
||||||
|
-- Per PSX-SPX `docs/psx-spx/docs/graphicsprocessingunitgpu.md` §"GPU Render Polygon Commands":
|
||||||
|
-- Each polygon command's word count = 1 (tag/cmd) + per-vertex (vertex + optional color + optional UV).
|
||||||
|
-- F3: cmd + 3 vertices = 4 words; +1 tag = 5
|
||||||
|
-- F4: cmd + 4 vertices = 5 words; +1 tag = 6
|
||||||
|
-- G3: cmd + 3×(color + vertex) = 6 words; +1 tag = 7
|
||||||
|
-- G4: cmd + 4×(color + vertex) = 8 words; +1 tag = 9
|
||||||
|
-- FT3: cmd + tpage + clut + 3×(vertex + UV) = 7 words; +1 tag = 8
|
||||||
|
-- FT4: cmd + tpage + clut + 4×(vertex + UV) = 9 words; +1 tag = 10
|
||||||
|
-- GT3: cmd + tpage + clut + 3×(color + vertex + UV) = 9 words; +1 tag = 10
|
||||||
|
-- GT4: cmd + tpage + clut + 4×(color + vertex + UV) = 12 words; +1 tag = 13
|
||||||
|
--
|
||||||
|
-- Cross-checked against code/duffle/gp.h struct sizes + the set_poly_* macros
|
||||||
|
-- (which encode "len" = "words after tag"):
|
||||||
|
-- set_poly_f3(p) -> set_len(p, 4) -> 5 total GP0 0x20
|
||||||
|
-- set_poly_ft3(p) -> set_len(p, 7) -> 8 total GP0 0x24
|
||||||
|
-- set_poly_f4(p) -> set_len(p, 5) -> 6 total GP0 0x28
|
||||||
|
-- set_poly_ft4(p) -> set_len(p, 9) -> 10 total GP0 0x2C
|
||||||
|
-- set_poly_g3(p) -> set_len(p, 6) -> 7 total GP0 0x30
|
||||||
|
-- set_poly_gt3(p) -> set_len(p, 9) -> 10 total GP0 0x34
|
||||||
|
-- set_poly_g4(p) -> set_len(p, 8) -> 9 total GP0 0x38
|
||||||
|
-- set_poly_gt4(p) -> set_len(p, 12) -> 13 total GP0 0x3C
|
||||||
|
--- @type table<integer, integer> -- bag: GP0 cmd byte -> word count
|
||||||
|
M.GP0_CMD_SIZE = {
|
||||||
|
[0x20] = 5, -- Poly_F3
|
||||||
|
[0x24] = 8, -- Poly_FT3
|
||||||
|
[0x28] = 6, -- Poly_F4
|
||||||
|
[0x2C] = 10, -- Poly_FT4
|
||||||
|
[0x30] = 7, -- Poly_G3
|
||||||
|
[0x34] = 10, -- Poly_GT3
|
||||||
|
[0x38] = 9, -- Poly_G4
|
||||||
|
[0x3C] = 13, -- Poly_GT4
|
||||||
|
}
|
||||||
|
|
||||||
|
-- Shape suffix (after `ac_format_` / `mac_format_` prefix) -> GP0 cmd byte.
|
||||||
|
-- Lets the static-analysis check derive the cmd byte from a macro name like `mac_format_g4_color` -> `g4` -> 0x38 -> 9 expected words.
|
||||||
|
--- @type table<string, integer> -- bag: shape suffix -> GP0 cmd byte
|
||||||
|
M.GP0_CMD_BY_SHAPE = {
|
||||||
|
["f3"] = 0x20, ["ft3"] = 0x24,
|
||||||
|
["f4"] = 0x28, ["ft4"] = 0x2C,
|
||||||
|
["g3"] = 0x30, ["gt3"] = 0x34,
|
||||||
|
["g4"] = 0x38, ["gt4"] = 0x3C,
|
||||||
|
}
|
||||||
|
|
||||||
|
--- @type integer
|
||||||
|
M.UNKNOWN_INSTRUCTION_CYCLES = 1
|
||||||
|
|
||||||
|
-- Hardware-relation policy table.
|
||||||
|
--
|
||||||
|
-- The forward walker in `passes/static_analysis.lua::analyze_hardware_relations` reads every emitted word_event, matches its `encoder` against `row.token`, and:
|
||||||
|
-- * stages the event as a producer in `atom.paths.forward_state`; or
|
||||||
|
-- * matches it as a consumer against pending producers and records a hazard on `atom.paths.hazards` when the gap is below `visibility.required`.
|
||||||
|
--
|
||||||
|
-- Each row is the contract for one CPU-to-coprocessor transfer semantic (the coprocessor-to-CPU path mirrors the same shape).
|
||||||
|
-- The `reads` / `writes` sub-tables carry the argument positions the analyzer inspects:
|
||||||
|
-- * `writes.arg` is the destination operand (the producer's effect); the analyzer stages this register as a pending producer.
|
||||||
|
-- * `reads` (when present) lists the operand positions the same token reads back from hardware; for MTC2 / CTC2 the producer reads the GPR source it is loading from.
|
||||||
|
-- The `fanout_to` field (MTC2-IRGB row only) tells the consumer-match logic which downstream COP2 registers are transitively updated by the write.
|
||||||
|
--
|
||||||
|
-- Visibility semantics:
|
||||||
|
-- * `kind = "post_producer_words"` means the consumer observes the producer's effect after `required` independent emitted words that are
|
||||||
|
-- strictly between the producer and the consumer. The producer's own emitted slot is implicit (it counts as the slot of issue, not toward `required`)
|
||||||
|
-- per the PSX-SPX rule: "Store delays are counted in numbers of clock cycles (not in numbers of opcodes).
|
||||||
|
-- For 3 cycle delay, one must usually insert 3 cached opcodes (or one uncached opcode)."
|
||||||
|
-- * `required` is the minimum count of intervening emitted words between producer and consumer.
|
||||||
|
-- `required = 0` permits the consumer on the very next slot; `required < 0` would place the consumer on the same slot as the producer
|
||||||
|
-- and is reserved for future "self-retires" relations.
|
||||||
|
--
|
||||||
|
-- Evidence:
|
||||||
|
-- * `evidence.confidence` is one of `"exact"`, `"conservative"`, `"unknown"`. The severity comes from `violation_kind`;
|
||||||
|
-- A hardware measurement that the vendor caveats may still classify as `"conservative"` even when the underlying timing is numerically known.
|
||||||
|
-- * `evidence.source` is the upstream reference (file + line range) the row is sourced from. New rows must carry this citation.
|
||||||
|
--
|
||||||
|
-- Consumers:
|
||||||
|
-- * passes/static_analysis.lua::analyze_hardware_relations (forward walker).
|
||||||
|
-- * passes/static_analysis.lua::transfer_hazards CHECK_RULES reader (renders hazards onto `findings`).
|
||||||
|
-- This table is consumed by the hardware-relation analyzer and hazard renderer.
|
||||||
|
--- @type HardwareRelationRow[]
|
||||||
|
M.HARDWARE_RELATIONS = {
|
||||||
|
-- CPU → COP2 data register (MTC2). The ordinary default is 2 cached words between producer and consumer (cpuspecifications.md:407-419).
|
||||||
|
{
|
||||||
|
id = "mtc2_gpr_visibility",
|
||||||
|
semantic = "MTC2",
|
||||||
|
consumer = "cop2_input",
|
||||||
|
token = "gte_mv_to_data_r",
|
||||||
|
direction = "gpr_to_cop2_data",
|
||||||
|
reads = { domain = "gpr", arg = 1 },
|
||||||
|
writes = { domain = "cop2.data", arg = 2 },
|
||||||
|
visibility = { kind = "post_producer_words", required = 2 },
|
||||||
|
evidence = {
|
||||||
|
confidence = "exact",
|
||||||
|
source = "cpuspecifications.md:407-419",
|
||||||
|
},
|
||||||
|
violation_kind = "error",
|
||||||
|
},
|
||||||
|
-- CPU → COP2 data register when the destination is C2_IRGB (data 28).
|
||||||
|
-- C2_IRGB drives the IR1/IR2/IR3 color-conversion fan-out, which extends the propagation delay to 3 cached words.
|
||||||
|
-- `destination_match = "C2_IRGB"` is the row's filter; the analyzer consults this when the producer's destination operand equals "C2_IRGB".
|
||||||
|
-- C2_ORGB (data 29) is read-only and is never classified as a writable fan-out destination.
|
||||||
|
{
|
||||||
|
id = "mtc2_irgb_visibility",
|
||||||
|
semantic = "MTC2",
|
||||||
|
consumer = "cop2_input",
|
||||||
|
token = "gte_mv_to_data_r",
|
||||||
|
direction = "gpr_to_cop2_data",
|
||||||
|
reads = { domain = "gpr", arg = 1 },
|
||||||
|
writes = { domain = "cop2.data", arg = 2 },
|
||||||
|
destination_match = "C2_IRGB",
|
||||||
|
fanout_to = { "C2_IR1", "C2_IR2", "C2_IR3" },
|
||||||
|
visibility = { kind = "post_producer_words", required = 3 },
|
||||||
|
evidence = {
|
||||||
|
confidence = "exact",
|
||||||
|
source = "cpuspecifications.md:407-419",
|
||||||
|
},
|
||||||
|
violation_kind = "error",
|
||||||
|
},
|
||||||
|
-- CPU → COP2 control register (CTC2). Ordinary minimum 2;
|
||||||
|
-- no IRGB-style fan-out exists for control registers (per spec §3.6: only C2_IRGB has the 3-cycle fan-out on the data side).
|
||||||
|
{
|
||||||
|
id = "ctc2_gpr_visibility",
|
||||||
|
semantic = "CTC2",
|
||||||
|
consumer = "cop2_input",
|
||||||
|
token = "gte_mv_to_ctrl_r",
|
||||||
|
direction = "gpr_to_cop2_control",
|
||||||
|
reads = { domain = "gpr", arg = 1 },
|
||||||
|
writes = { domain = "cop2.ctrl", arg = 2 },
|
||||||
|
visibility = { kind = "post_producer_words", required = 2 },
|
||||||
|
evidence = {
|
||||||
|
confidence = "exact",
|
||||||
|
source = "cpuspecifications.md:407-419",
|
||||||
|
},
|
||||||
|
violation_kind = "error",
|
||||||
|
},
|
||||||
|
-- COP2 data → GPR (MFC2). One cached slot between the transfer and the first GPR consumer;
|
||||||
|
-- the GPR is not updated until the instruction AFTER the MFC2 completes (geometrytransformationenginegte.md:29-32).
|
||||||
|
{
|
||||||
|
id = "mfc2_gpr_visibility",
|
||||||
|
semantic = "MFC2",
|
||||||
|
consumer = "gpr_read",
|
||||||
|
token = "gte_mv_from_data_r",
|
||||||
|
direction = "cop2_data_to_gpr",
|
||||||
|
reads = { domain = "cop2.data", arg = 2 },
|
||||||
|
writes = { domain = "gpr", arg = 1 },
|
||||||
|
visibility = { kind = "post_producer_words", required = 1 },
|
||||||
|
evidence = {
|
||||||
|
confidence = "exact",
|
||||||
|
source = "geometrytransformationenginegte.md:29-32",
|
||||||
|
},
|
||||||
|
violation_kind = "error",
|
||||||
|
},
|
||||||
|
-- COP2 control → GPR (CFC2). Same delay as MFC2 (cpuspecifications.md treats the two load-from-COP2 paths symmetrically).
|
||||||
|
{
|
||||||
|
id = "cfc2_gpr_visibility",
|
||||||
|
semantic = "CFC2",
|
||||||
|
consumer = "gpr_read",
|
||||||
|
token = "gte_mv_from_ctrl_r",
|
||||||
|
direction = "cop2_control_to_gpr",
|
||||||
|
reads = { domain = "cop2.ctrl", arg = 2 },
|
||||||
|
writes = { domain = "gpr", arg = 1 },
|
||||||
|
visibility = { kind = "post_producer_words", required = 1 },
|
||||||
|
evidence = {
|
||||||
|
confidence = "exact",
|
||||||
|
source = "cpuspecifications.md:382-419",
|
||||||
|
},
|
||||||
|
violation_kind = "error",
|
||||||
|
},
|
||||||
|
-- COP0 control → GPR (MFC0).
|
||||||
|
-- One cached slot; the analyzer treats `sys_mov_from_cop0(rt, 12)` (the SR/CU2 transfer) as the same shape as the COP2 load-delay path.
|
||||||
|
-- The semantic-level SR/CU2 transition models the load delay;
|
||||||
|
-- SR.CU2 bounded-value propagation is modeled separately).
|
||||||
|
{
|
||||||
|
id = "mfc0_gpr_visibility",
|
||||||
|
semantic = "MFC0",
|
||||||
|
consumer = "gpr_read",
|
||||||
|
token = "sys_mov_from_cop0",
|
||||||
|
direction = "cop0_control_to_gpr",
|
||||||
|
reads = { domain = "cop0.ctrl", arg = 2 },
|
||||||
|
writes = { domain = "gpr", arg = 1 },
|
||||||
|
visibility = { kind = "post_producer_words", required = 1 },
|
||||||
|
evidence = {
|
||||||
|
confidence = "exact",
|
||||||
|
source = "cpuspecifications.md:171-178",
|
||||||
|
},
|
||||||
|
violation_kind = "error",
|
||||||
|
},
|
||||||
|
-- Memory -> COP2 data register (LWC2).
|
||||||
|
-- The memory-side timing is not measured by the vendored GTE latch experiment, so this relation has no numeric retirement threshold.
|
||||||
|
-- The LWC2 destination has TWO retirement regimes (per PSX-SPX):
|
||||||
|
-- * GTE-command consumer (`gte_cmdw_*`): the GTE pipeline LATCHES the LWC2 result, so a `gte_cmdw_*`
|
||||||
|
-- in the very next slot uses the latched value. Gap = 0 is allowed. (Per `docs/psx-spx/docs/gtepipelinetimings.md:271-274`.)
|
||||||
|
-- * Any other consumer: standard MIPS load delay applies. Gap = 1 required. (Per `docs/psx-spx/docs/cpuspecifications.md:407-419`.)
|
||||||
|
-- Two separate relations so the walker can dispatch by consumer type and emit different severities
|
||||||
|
-- (the GTE-command path is `info` because the latch is intentional; the non-GTE-consumer path is `error` because the missing nop is a real bug).
|
||||||
|
{
|
||||||
|
id = "lwc2_to_gte_command",
|
||||||
|
semantic = "LWC2_to_GTE",
|
||||||
|
consumer = "cop2_input",
|
||||||
|
token = "gte_lw",
|
||||||
|
direction = "memory_to_cop2_data",
|
||||||
|
reads = { domain = "memory", arg = 2 },
|
||||||
|
writes = { domain = "cop2.data", arg = 1 },
|
||||||
|
required = 0, -- GTE-command consumer: gap = 0 OK (latched).
|
||||||
|
evidence = {
|
||||||
|
confidence = "measured",
|
||||||
|
source = "gtepipelinetimings.md:271-274",
|
||||||
|
},
|
||||||
|
violation_kind = "info",
|
||||||
|
clear_on_consumer = true,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
id = "lwc2_to_other_consumer",
|
||||||
|
semantic = "LWC2_to_other",
|
||||||
|
consumer = "cop2_input",
|
||||||
|
token = "gte_lw",
|
||||||
|
direction = "memory_to_cop2_data",
|
||||||
|
reads = { domain = "memory", arg = 2 },
|
||||||
|
writes = { domain = "cop2.data", arg = 1 },
|
||||||
|
required = 1, -- Non-GTE-consumer: standard MIPS load delay.
|
||||||
|
evidence = {
|
||||||
|
confidence = "inferred",
|
||||||
|
source = "cpuspecifications.md:407-419",
|
||||||
|
},
|
||||||
|
violation_kind = "error",
|
||||||
|
clear_on_consumer = true,
|
||||||
|
},
|
||||||
|
-- COP2 data register -> memory (SWC2). A read of C2 state, not a CPU-to-COP2 write.
|
||||||
|
-- The policy row stays in for direction/provenance; staging it as a later command-input producer is suppressed.
|
||||||
|
{
|
||||||
|
id = "swc2_memory_write",
|
||||||
|
semantic = "SWC2",
|
||||||
|
consumer = "gpr_read",
|
||||||
|
token = "gte_sw",
|
||||||
|
direction = "cop2_data_to_memory",
|
||||||
|
reads = { domain = "cop2.data", arg = 1 },
|
||||||
|
writes = { domain = "memory", arg = 2 },
|
||||||
|
visibility = { kind = "none", required = 0 },
|
||||||
|
evidence = {
|
||||||
|
confidence = "exact",
|
||||||
|
source = "cpuspecifications.md:79",
|
||||||
|
},
|
||||||
|
violation_kind = "info",
|
||||||
|
stage = false,
|
||||||
|
},
|
||||||
|
-- MTC0 Status/SR.CU2. The ordinary COP0 store has no general store-delay relation;
|
||||||
|
-- this row feeds the dedicated CU2 transition logic in the same forward walk and is therefore not staged in `pending`.
|
||||||
|
{
|
||||||
|
id = "mtc0_cu2_visibility",
|
||||||
|
semantic = "MTC0",
|
||||||
|
consumer = "gpr_read",
|
||||||
|
token = "sys_mov_to_cop0",
|
||||||
|
direction = "gpr_to_cop0_status",
|
||||||
|
reads = { domain = "gpr", arg = 1 },
|
||||||
|
writes = { domain = "cop0.status", arg = 2 },
|
||||||
|
status_register = 12,
|
||||||
|
visibility = { kind = "post_producer_words", required = 2 },
|
||||||
|
evidence = {
|
||||||
|
confidence = "conservative",
|
||||||
|
source = "cpuspecifications.md:543,625-628",
|
||||||
|
},
|
||||||
|
violation_kind = "warning",
|
||||||
|
stage = false,
|
||||||
|
cu2_transition = true,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
-- Bounded Status/SR.CU2 transition policy.
|
||||||
|
-- The value lattice and the transition consumer both read this immutable row; no second value pass is permitted.
|
||||||
|
-- The source says the enable/disable transition takes "2 clock cycles or so", so the boundary is conservative rather than exact.
|
||||||
|
--- @type Cu2TransitionPolicy
|
||||||
|
M.CU2_TRANSITION_POLICY = {
|
||||||
|
status_register = 12,
|
||||||
|
enable_bit = 0x40000000,
|
||||||
|
required = 2,
|
||||||
|
visibility_kind = "post_producer_words",
|
||||||
|
evidence = {
|
||||||
|
confidence = "conservative",
|
||||||
|
source = "cpuspecifications.md:543,625-628",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
return M
|
||||||
@@ -0,0 +1,89 @@
|
|||||||
|
--- duffle_paths.lua — Single-line bootstrap helper for the tape-atom Lua scripts.
|
||||||
|
---
|
||||||
|
--- Each entry script (ps1_meta.lua + the 7 passes/*.lua files) starts with one of:
|
||||||
|
--- ```lua
|
||||||
|
--- -- Entry script (ps1_meta.lua — `arg[0]` is set):
|
||||||
|
--- local duffle = dofile((arg[0]:match("(.*[/\\])") or "./") .. "duffle_paths.lua")
|
||||||
|
---
|
||||||
|
--- -- Pass module (debug.getinfo path resolution; works both standalone and when require'd):
|
||||||
|
--- local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||||
|
--- local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||||
|
--- ```
|
||||||
|
---
|
||||||
|
--- That small bootstrap: (a) locates this helper via `arg[0]` / `debug.getinfo`,
|
||||||
|
--- (b) loads it (which sets `package.path` + `package.cpath`),
|
||||||
|
--- (c) at the bottom calls `require("duffle")` (now resolvable since `package.path` was just set) and returns the duffle M.
|
||||||
|
--- Net effect: the caller gets the duffle module in one statement; no separate `dofile(...)` + `require("duffle")` dance.
|
||||||
|
---
|
||||||
|
|
||||||
|
--- @class DufflePaths
|
||||||
|
--- @field setup fun(): nil
|
||||||
|
|
||||||
|
local M = {} ---@type DufflePaths
|
||||||
|
|
||||||
|
-- Cache key for the repo root. Stored in `package.loaded` (process-global) so all 8 entry scripts + passes scripts share one resolution.
|
||||||
|
local CACHE_KEY = "__duffle_repo_root__" ---@type string
|
||||||
|
|
||||||
|
--- Resolve the repo root from this script's own path. Zero shell spawn.
|
||||||
|
--- `duffle_paths.lua` always lives at `<repo>/scripts/duffle_paths.lua`, so the repo root is the parent of the directory containing this script.
|
||||||
|
--- We derive it directly from `debug.getinfo(1, "S").source` (returns `@<path>` for the currently-running chunk).
|
||||||
|
---
|
||||||
|
--- If `debug.getinfo` can't parse this script's path (shouldn't happen — dofile always populates source), return nil and let `M.setup()` fail loud.
|
||||||
|
--- @return string|nil
|
||||||
|
local function find_repo_root()
|
||||||
|
if package.loaded[CACHE_KEY] then return package.loaded[CACHE_KEY] end
|
||||||
|
|
||||||
|
local source = debug.getinfo(1, "S").source ---@type string
|
||||||
|
-- Strip the leading `@` (Lua's dofile marker) and the trailing `/duffle_paths.lua` filename.
|
||||||
|
-- What remains is the directory containing this script, i.e. `<repo>/scripts/`.
|
||||||
|
local scripts_dir = source and source:match("^@?(.*)[/\\]duffle_paths%.lua$") ---@type string|nil
|
||||||
|
if not scripts_dir then return nil end
|
||||||
|
|
||||||
|
-- The repo root is the parent of `scripts/`. Strip the trailing `scripts/` (with or without trailing slash).
|
||||||
|
local root = scripts_dir:gsub("scripts[\\/]?$", "") ---@type string
|
||||||
|
root = root:gsub("\\", "/")
|
||||||
|
if root == "" then root = "./" end
|
||||||
|
if not root:match("/$") then root = root .. "/" end
|
||||||
|
package.loaded[CACHE_KEY] = root
|
||||||
|
return root
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Set `package.path` (for `require("duffle")` + `require("passes.X")`) and `package.cpath` (for `lpeg.dll`).
|
||||||
|
---
|
||||||
|
--- This script does NOT touch the OS environment: no `os.setenv`, no `os.putenv`, no `$PATH` mods.
|
||||||
|
--- It just sets `package.path` and `package.cpath` (the standard Lua way to register module search dirs).
|
||||||
|
--- lpeg is built by `update_deps.ps1` to `toolchain/lpeg/`, which we wire into `package.cpath` here (so `require("lpeg")` from `duffle.lua` resolves without any global state).
|
||||||
|
--- @return nil
|
||||||
|
function M.setup()
|
||||||
|
local repo_root = find_repo_root() ---@type string|nil
|
||||||
|
if not repo_root then
|
||||||
|
-- Unreachable in practice: find_repo_root() derives the repo root from this script's own source path via debug.getinfo(1, "S").source (no subprocess, no git CLI, <1ms).
|
||||||
|
-- A nil return means the source path did not match the expected <repo>/scripts/duffle_paths.lua layout — a packaging bug, not a "missing git repo" condition.
|
||||||
|
-- os.exit(2) is retained so a real failure surfaces loud rather than silently producing an unconfigured module table.
|
||||||
|
os.exit(2)
|
||||||
|
end
|
||||||
|
|
||||||
|
local scripts_dir = repo_root .. "scripts/" ---@type string
|
||||||
|
local passes_dir = repo_root .. "scripts/passes/" ---@type string
|
||||||
|
package.path = scripts_dir .. "?.lua;"
|
||||||
|
.. scripts_dir .. "?/init.lua;"
|
||||||
|
.. passes_dir .. "?.lua;"
|
||||||
|
.. passes_dir .. "?/init.lua;"
|
||||||
|
.. package.path
|
||||||
|
|
||||||
|
-- lpeg: built by `update_deps.ps1` to `toolchain/lpeg/lpeg.dll`.
|
||||||
|
-- lfs: compiled from pcsx-redux's vendored luafilesystem source to `toolchain/lfs/lfs.dll`.
|
||||||
|
-- Wire both directories into cpath so `require("lpeg")` and `require("lfs")` resolve.
|
||||||
|
local lpeg_dir = repo_root .. "toolchain/lpeg/" ---@type string
|
||||||
|
local lfs_dir = repo_root .. "toolchain/lfs/" ---@type string
|
||||||
|
package.cpath = lpeg_dir .. "?.dll;"
|
||||||
|
.. lfs_dir .. "?.dll;"
|
||||||
|
.. package.cpath
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Run the setup as a side effect.
|
||||||
|
M.setup()
|
||||||
|
|
||||||
|
-- Now that package.path includes scripts/, `require("duffle")` resolves.
|
||||||
|
-- Return the duffle module so callers can do `local duffle = dofile(...duffle_paths.lua)` in one line.
|
||||||
|
return require("duffle")
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,511 @@
|
|||||||
|
-- elf32.lua — Pure-Lua ELF32 format helpers with no lfs / no lpeg dependency.
|
||||||
|
-- The reload helper's `parse_manifest` (scripts/pcsx_debug_helper/reload.lua)
|
||||||
|
-- and the metaprogram's `read_elf_sections` + `read_nm` (scripts/elf_dwarf.lua)
|
||||||
|
-- both parsed ELF32 headers from wire bytes.
|
||||||
|
--
|
||||||
|
-- This module contains the format constants and the byte-level walker.
|
||||||
|
--- The metaprogram side keeps `read_u32_le` / `read_u16_le` as local forwarders; the helper side calls `E.*` directly.
|
||||||
|
--
|
||||||
|
-- **Adapter contract (explicit pass style):**
|
||||||
|
-- The helper VM's `Support.File` exposes byte-read methods that require `self` (fileffi.lua:225-227),
|
||||||
|
-- so callers wrap once in a 1-line adapter that strips `self`.
|
||||||
|
-- The parsers here operate on the unwrapped form.
|
||||||
|
-- Reads are flat function calls — `E.read_u8(adapter, off)`, `E.read_u32(adapter, off)`, `E.size(adapter)`.
|
||||||
|
-- read_u8(adapter, off) -> integer | nil
|
||||||
|
-- read_u16(adapter, off) -> integer | nil
|
||||||
|
-- read_u32(adapter, off) -> integer | nil
|
||||||
|
-- size(adapter) -> integer
|
||||||
|
--
|
||||||
|
-- **Convention:** every offset in the constants tables is a zero-based wire offset.
|
||||||
|
-- The `+ 1` conversion happens only at the `string.byte` boundary inside the readers.
|
||||||
|
--
|
||||||
|
-- spec: System V ABI gABI v1.2 §"ELF Header" (Table 1) + §"Section Header Table"
|
||||||
|
-- spec: System V ABI gABI v1.2 §"Symbol Table" (Elf32_Sym layout)
|
||||||
|
|
||||||
|
--- @class Elf32Adapter
|
||||||
|
--- @field read_u8_at fun(off: integer): integer|nil
|
||||||
|
--- @field read_u16_at fun(off: integer): integer|nil
|
||||||
|
--- @field read_u32_at fun(off: integer): integer|nil
|
||||||
|
--- @field read_size fun(): integer
|
||||||
|
|
||||||
|
--- @class Elf32Header
|
||||||
|
--- @field e_entry integer
|
||||||
|
--- @field e_shoff integer
|
||||||
|
--- @field e_shentsize integer
|
||||||
|
--- @field e_shnum integer
|
||||||
|
--- @field e_shstrndx integer
|
||||||
|
--- @field error string|nil
|
||||||
|
|
||||||
|
--- @class Elf32Section
|
||||||
|
--- @field sh_name integer
|
||||||
|
--- @field sh_type integer
|
||||||
|
--- @field sh_flags integer
|
||||||
|
--- @field sh_addr integer
|
||||||
|
--- @field sh_offset integer
|
||||||
|
--- @field sh_size integer
|
||||||
|
--- @field sh_link integer
|
||||||
|
--- @field name string
|
||||||
|
|
||||||
|
--- @class Elf32Sym
|
||||||
|
--- @field value integer
|
||||||
|
--- @field size integer
|
||||||
|
--- @field info integer
|
||||||
|
--- @field shndx integer
|
||||||
|
|
||||||
|
--- @class Elf32HeaderLayout
|
||||||
|
--- @field magic_offset integer
|
||||||
|
--- @field magic string
|
||||||
|
--- @field class_offset integer
|
||||||
|
--- @field endian_offset integer
|
||||||
|
--- @field header_bytes integer
|
||||||
|
--- @field e_entry_offset integer
|
||||||
|
--- @field e_shoff_offset integer
|
||||||
|
--- @field e_shentsize_offset integer
|
||||||
|
--- @field e_shnum_offset integer
|
||||||
|
--- @field e_shstrndx_offset integer
|
||||||
|
|
||||||
|
--- @class Elf32SectionLayout
|
||||||
|
--- @field sh_name_offset integer
|
||||||
|
--- @field sh_type_offset integer
|
||||||
|
--- @field sh_flags_offset integer
|
||||||
|
--- @field sh_addr_offset integer
|
||||||
|
--- @field sh_offset_offset integer
|
||||||
|
--- @field sh_size_offset integer
|
||||||
|
--- @field sh_link_offset integer
|
||||||
|
--- @field sh_entsize_bytes integer
|
||||||
|
|
||||||
|
--- @class Elf32SymLayout
|
||||||
|
--- @field st_name integer
|
||||||
|
--- @field st_value integer
|
||||||
|
--- @field st_size integer
|
||||||
|
--- @field st_info integer
|
||||||
|
--- @field sym_entry_bytes integer
|
||||||
|
|
||||||
|
--- @class Elf32Mod
|
||||||
|
--- @field ELFCLASS32 integer
|
||||||
|
--- @field ELFDATA2LSB integer
|
||||||
|
--- @field EM_MIPS integer
|
||||||
|
--- @field SHT_SYMTAB integer
|
||||||
|
--- @field SHT_STRTAB integer
|
||||||
|
--- @field SHT_NOBITS integer
|
||||||
|
--- @field SHF_WRITE integer
|
||||||
|
--- @field SHF_ALLOC integer
|
||||||
|
--- @field SHF_EXECINSTR integer
|
||||||
|
--- @field ELF32_HEADER Elf32HeaderLayout
|
||||||
|
--- @field ELF32_SECTION Elf32SectionLayout
|
||||||
|
--- @field ELF32_SYM Elf32SymLayout
|
||||||
|
--- @field dw_dwarf32_terminator integer
|
||||||
|
--- @field read_u32 fun(adapter: Elf32Adapter, off: integer): integer|nil
|
||||||
|
--- @field read_u16 fun(adapter: Elf32Adapter, off: integer): integer|nil
|
||||||
|
--- @field read_u8 fun(adapter: Elf32Adapter, off: integer): integer|nil
|
||||||
|
--- @field size fun(adapter: Elf32Adapter): integer
|
||||||
|
--- @field read_u32_le fun(buf: string, off: integer): integer
|
||||||
|
--- @field read_u16_le fun(buf: string, off: integer): integer
|
||||||
|
--- @field validate_adapter fun(adapter: any): boolean, string|nil
|
||||||
|
--- @field get_str fun(strtab: string, off: integer): string|nil
|
||||||
|
--- @field parse_elf32_headers fun(adapter: Elf32Adapter): Elf32Header|nil, string|nil
|
||||||
|
--- @field walk_sections fun(adapter: Elf32Adapter, hdr: Elf32Header): Elf32Section[]|nil, string|nil
|
||||||
|
--- @field read_section_bytes fun(adapter: Elf32Adapter, section: Elf32Section): string|nil
|
||||||
|
--- @field read_named_section fun(adapter: Elf32Adapter, sections: Elf32Section[], name: string): string|nil, string|nil
|
||||||
|
--- @field collect_symbols fun(adapter: Elf32Adapter, sections: Elf32Section[]): table<string, Elf32Sym>|nil, string|nil
|
||||||
|
|
||||||
|
local M = {} ---@type Elf32Mod
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Little-endian readers (bit-weighted accumulator, math.floor only)
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Read a 4-byte little-endian unsigned integer from `adapter` at zero-based wire offset `off`.
|
||||||
|
---
|
||||||
|
--- Bit weights are written as `0x100`, `0x10000`, `0x1000000` (i.e. 2^8, 2^16, 2^24) so the LE byte positions are visually explicit:
|
||||||
|
--- byte 0 contributes its value directly;
|
||||||
|
--- byte 1 is shifted left by 8; byte 2 by 16; byte 3 by 24.
|
||||||
|
---
|
||||||
|
--- math.floor (not LuaJIT's `>>`) keeps the body portable across LuaJIT 2.0/2.1 and plain Lua 5.x. `string.byte` receives `+ 1` at the boundary.
|
||||||
|
---
|
||||||
|
--- **Call form:** explicit-pass. The reader receives `adapter` as the first positional argument and the offset as the second; no `self` is passed.
|
||||||
|
--- Test fixtures declare `function(offset) ... end` and the parsers call them via dot syntax `adapter.read_u8_at(off)`.
|
||||||
|
--- The colon form `adapter:read_u8_at(off)` would prepend the adapter table as `offset` and break the contract.
|
||||||
|
--- @param adapter Elf32Adapter
|
||||||
|
--- @param off integer -- zero-based wire offset
|
||||||
|
--- @return integer|nil
|
||||||
|
function M.read_u32(adapter, off)
|
||||||
|
return adapter.read_u8_at(off)
|
||||||
|
+ adapter.read_u8_at(off + 0x01) * 0x00000100
|
||||||
|
+ adapter.read_u8_at(off + 0x02) * 0x00010000
|
||||||
|
+ adapter.read_u8_at(off + 0x03) * 0x01000000
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Read a 2-byte little-endian unsigned integer from `adapter` at zero-based wire offset `off`.
|
||||||
|
--- @param adapter Elf32Adapter
|
||||||
|
--- @param off integer -- zero-based wire offset
|
||||||
|
--- @return integer|nil
|
||||||
|
function M.read_u16(adapter, off)
|
||||||
|
return adapter.read_u8_at(off)
|
||||||
|
+ adapter.read_u8_at(off + 0x01) * 0x00000100
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Read a 1-byte unsigned integer from `adapter` at zero-based wire offset `off`.
|
||||||
|
--- @param adapter Elf32Adapter
|
||||||
|
--- @param off integer -- zero-based wire offset
|
||||||
|
--- @return integer|nil
|
||||||
|
function M.read_u8(adapter, off)
|
||||||
|
return adapter.read_u8_at(off)
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Total adapter byte length.
|
||||||
|
--- @param adapter Elf32Adapter
|
||||||
|
--- @return integer
|
||||||
|
function M.size(adapter)
|
||||||
|
return adapter.read_size()
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Forwarders kept for backward compat with scripts/elf_dwarf.lua.
|
||||||
|
--- The metaprogram side keeps `read_u32_le` / `read_u16_le`;
|
||||||
|
--- both layers now use the same byte-level helpers under the hood.
|
||||||
|
--- @param buf string
|
||||||
|
--- @param off integer
|
||||||
|
--- @return integer
|
||||||
|
function M.read_u32_le(buf, off)
|
||||||
|
local byte_off = off + 1 ---@type integer
|
||||||
|
return buf:byte(byte_off)
|
||||||
|
+ buf:byte(byte_off + 0x01) * 0x00000100
|
||||||
|
+ buf:byte(byte_off + 0x02) * 0x00010000
|
||||||
|
+ buf:byte(byte_off + 0x03) * 0x01000000
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Read a 2-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
|
||||||
|
--- @param buf string
|
||||||
|
--- @param off integer -- zero-based wire offset
|
||||||
|
--- @return integer
|
||||||
|
function M.read_u16_le(buf, off)
|
||||||
|
local byte_off = off + 1 ---@type integer
|
||||||
|
return buf:byte(byte_off) + buf:byte(byte_off + 0x01) * 0x00000100
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Format constants
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
-- ELF format constants (System V ABI gABI v1.2).
|
||||||
|
M.ELFCLASS32 = 1 -- spec: gABI v1.2 §"ELF Header" — EI_CLASS byte
|
||||||
|
M.ELFDATA2LSB = 1 -- spec: gABI v1.2 §"ELF Header" — EI_DATA byte
|
||||||
|
M.EM_MIPS = 8 -- spec: gABI v1.2 §"Machine Information" — MIPS architecture
|
||||||
|
|
||||||
|
-- Section type constants (System V ABI gABI v1.2 §"Section Header Table").
|
||||||
|
M.SHT_SYMTAB = 2 -- spec: gABI v1.2 §"Section Types" — symbol table
|
||||||
|
M.SHT_STRTAB = 3 -- spec: gABI v1.2 §"Section Types" — string table
|
||||||
|
M.SHT_NOBITS = 8 -- spec: gABI v1.2 §"Section Types" — no space in file
|
||||||
|
|
||||||
|
-- Section flag constants (System V ABI gABI v1.2 §"Section Header Table").
|
||||||
|
M.SHF_WRITE = 0x1 -- spec: gABI v1.2 §"Section Attributes" — writable
|
||||||
|
M.SHF_ALLOC = 0x2 -- spec: gABI v1.2 §"Section Attributes" — occupies memory
|
||||||
|
M.SHF_EXECINSTR = 0x4 -- spec: gABI v1.2 §"Section Attributes" — executable
|
||||||
|
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
-- ELF32 header layout (System V ABI gABI v1.2 §"ELF Header" Table 1)
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
-- All offsets are zero-based wire offsets. The header is 52 bytes total (header_bytes = 0x34 = 52).
|
||||||
|
--- @type Elf32HeaderLayout
|
||||||
|
M.ELF32_HEADER = {
|
||||||
|
magic_offset = 0x00, -- 4 bytes; expected "\127ELF"
|
||||||
|
magic = "\127ELF",
|
||||||
|
class_offset = 0x04, -- 1 byte; 1 = ELF32, 2 = ELF64
|
||||||
|
endian_offset = 0x05, -- 1 byte; 1 = little-endian, 2 = big-endian
|
||||||
|
header_bytes = 0x34, -- ELF32 header is 52 bytes total
|
||||||
|
e_entry_offset = 0x18, -- 4-byte LE; entry-point virtual address
|
||||||
|
e_shoff_offset = 0x20, -- 4-byte LE; section-header table file offset
|
||||||
|
e_shentsize_offset = 0x2E, -- 2-byte LE; section-header entry size in bytes
|
||||||
|
e_shnum_offset = 0x30, -- 2-byte LE; number of section headers
|
||||||
|
e_shstrndx_offset = 0x32, -- 2-byte LE; index of section-name string table
|
||||||
|
}
|
||||||
|
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
-- ELF32 section-header layout (System V ABI gABI v1.2 §"Section Header Table")
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
-- Each entry is 40 bytes (sh_entsize_bytes = 0x28 = 40);
|
||||||
|
-- zero-based, field offsets relative to the start of the entry.
|
||||||
|
--- @type Elf32SectionLayout
|
||||||
|
M.ELF32_SECTION = {
|
||||||
|
sh_name_offset = 0x00, -- 4-byte LE; offset into .shstrtab
|
||||||
|
sh_type_offset = 0x04, -- 4-byte LE; section type (SHT_*)
|
||||||
|
sh_flags_offset = 0x08, -- 4-byte LE; section flags (SHF_*)
|
||||||
|
sh_addr_offset = 0x0C, -- 4-byte LE; virtual address at execution
|
||||||
|
sh_offset_offset = 0x10, -- 4-byte LE; section's file offset
|
||||||
|
sh_size_offset = 0x14, -- 4-byte LE; section's size in bytes
|
||||||
|
sh_link_offset = 0x18, -- 4-byte LE; link to a related section
|
||||||
|
sh_entsize_bytes = 0x28, -- spec: gABI v1.2 §"Section Header Table" — 40 bytes per entry
|
||||||
|
}
|
||||||
|
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
-- ELF32 symbol-table entry layout (System V ABI gABI v1.2 §"Symbol Table")
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
-- Each entry is 16 bytes (sym_entry_bytes = 0x10 = 16);
|
||||||
|
-- zero-based, field offsets relative to the start of the entry.
|
||||||
|
--- @type Elf32SymLayout
|
||||||
|
M.ELF32_SYM = {
|
||||||
|
st_name = 0x00, -- 4-byte LE; offset into the linked string table
|
||||||
|
st_value = 0x04, -- 4-byte LE; symbol value (address / absolute)
|
||||||
|
st_size = 0x08, -- 4-byte LE; symbol size in bytes
|
||||||
|
st_info = 0x0C, -- 1 byte; binding (high nibble) + type (low nibble)
|
||||||
|
sym_entry_bytes = 0x10, -- spec: gABI v1.2 §"Symbol Table" — 16 bytes per entry
|
||||||
|
}
|
||||||
|
|
||||||
|
-- DWARF32 initial-length terminator (DWARF4 §7.4) — kept here so the metaprogram's elf_dwarf.lua can drop its own copy of the same constant.
|
||||||
|
M.dw_dwarf32_terminator = 0xFFFFFFFF
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Adapter validation
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Validate that `adapter` exposes the byte-read surface.
|
||||||
|
--- Returns true on success, false + a stable error code on failure.
|
||||||
|
--- The helper side calls this before parse_manifest to reject callers before any byte is read.
|
||||||
|
--- @param adapter any
|
||||||
|
--- @return boolean, string|nil
|
||||||
|
function M.validate_adapter(adapter)
|
||||||
|
if type(adapter) ~= "table" then return false, "bad_file_adapter" end
|
||||||
|
if type(adapter.read_u8_at) ~= "function" then return false, "bad_file_adapter" end
|
||||||
|
if type(adapter.read_u16_at) ~= "function" then return false, "bad_file_adapter" end
|
||||||
|
if type(adapter.read_u32_at) ~= "function" then return false, "bad_file_adapter" end
|
||||||
|
if type(adapter.read_size) ~= "function" then return false, "bad_file_adapter" end
|
||||||
|
return true, nil
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- String-table reader
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Extract a NUL-terminated C string from `strtab` at zero-based offset `off`.
|
||||||
|
--- Returns nil if `off` is out of range or the string is not NUL-terminated.
|
||||||
|
--- @param strtab string
|
||||||
|
--- @param off integer
|
||||||
|
--- @return string|nil
|
||||||
|
function M.get_str(strtab, off)
|
||||||
|
if off < 0 or off >= #strtab then return nil end
|
||||||
|
local end_pos = strtab:find("\0", off + 1, true) ---@type integer|nil
|
||||||
|
if not end_pos then return nil end
|
||||||
|
return strtab:sub(off + 1, end_pos - 1)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Header / section / symbol walkers
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Read the ELF32 header through `adapter` and validate the magic, class, and data encoding.
|
||||||
|
--- Returns a table on success:
|
||||||
|
--- { e_entry, e_shoff, e_shentsize, e_shnum, e_shstrndx, error = nil }
|
||||||
|
--- On failure returns nil + a stable error code:
|
||||||
|
--- bad_magic, unsupported_elf_class, unsupported_elf_data, truncated_header
|
||||||
|
--- The header's machine field is NOT validated here — callers (e.g. the helper's prime path) decide whether to require EM_MIPS before symbol reads.
|
||||||
|
--- @param adapter Elf32Adapter
|
||||||
|
--- @return Elf32Header|nil, string|nil
|
||||||
|
function M.parse_elf32_headers(adapter)
|
||||||
|
local ok, err = M.validate_adapter(adapter) ---@type boolean, string|nil
|
||||||
|
if not ok then return nil, err end
|
||||||
|
|
||||||
|
-- 4-byte magic: 0x7F 'E' 'L' 'F'.
|
||||||
|
-- The byte readers take the adapter explicitly.
|
||||||
|
-- The production `Support.File` adapter is wrapped by the caller to drop its implicit `self` so the parser shape is flat pass-style.
|
||||||
|
local b1 = M.read_u8(adapter, 0) ---@type integer|nil
|
||||||
|
local b2 = M.read_u8(adapter, 1) ---@type integer|nil
|
||||||
|
local b3 = M.read_u8(adapter, 2) ---@type integer|nil
|
||||||
|
local b4 = M.read_u8(adapter, 3) ---@type integer|nil
|
||||||
|
if not (b1 and b2 and b3 and b4)
|
||||||
|
or not (b1 == 0x7f and b2 == 0x45 and b3 == 0x4c and b4 == 0x46) then
|
||||||
|
return nil, "bad_magic"
|
||||||
|
end
|
||||||
|
|
||||||
|
local class = M.read_u8(adapter, M.ELF32_HEADER.class_offset) ---@type integer|nil
|
||||||
|
if class ~= M.ELFCLASS32 then
|
||||||
|
return nil, "unsupported_elf_class"
|
||||||
|
end
|
||||||
|
|
||||||
|
local data = M.read_u8(adapter, M.ELF32_HEADER.endian_offset) ---@type integer|nil
|
||||||
|
if data ~= M.ELFDATA2LSB then
|
||||||
|
return nil, "unsupported_elf_data"
|
||||||
|
end
|
||||||
|
|
||||||
|
local e_entry = M.read_u32(adapter, M.ELF32_HEADER.e_entry_offset) ---@type integer|nil
|
||||||
|
local e_shoff = M.read_u32(adapter, M.ELF32_HEADER.e_shoff_offset) ---@type integer|nil
|
||||||
|
local e_shentsize = M.read_u16(adapter, M.ELF32_HEADER.e_shentsize_offset) ---@type integer|nil
|
||||||
|
local e_shnum = M.read_u16(adapter, M.ELF32_HEADER.e_shnum_offset) ---@type integer|nil
|
||||||
|
local e_shstrndx = M.read_u16(adapter, M.ELF32_HEADER.e_shstrndx_offset) ---@type integer|nil
|
||||||
|
if not (e_entry and e_shoff and e_shentsize and e_shnum and e_shstrndx) then
|
||||||
|
return nil, "truncated_header"
|
||||||
|
end
|
||||||
|
|
||||||
|
return {
|
||||||
|
e_entry = e_entry,
|
||||||
|
e_shoff = e_shoff,
|
||||||
|
e_shentsize = e_shentsize,
|
||||||
|
e_shnum = e_shnum,
|
||||||
|
e_shstrndx = e_shstrndx,
|
||||||
|
error = nil,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Read one section-header entry from `adapter` at `sh_off`.
|
||||||
|
--- Returns a table with the wire fields plus a (yet-unresolved) `name` field.
|
||||||
|
--- @param adapter Elf32Adapter
|
||||||
|
--- @param sh_off integer
|
||||||
|
--- @return Elf32Section|nil, string|nil
|
||||||
|
local function read_section_entry(adapter, sh_off)
|
||||||
|
local entry = { ---@type Elf32Section
|
||||||
|
sh_name = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_name_offset),
|
||||||
|
sh_type = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_type_offset),
|
||||||
|
sh_flags = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_flags_offset),
|
||||||
|
sh_addr = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_addr_offset),
|
||||||
|
sh_offset = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_offset_offset),
|
||||||
|
sh_size = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_size_offset),
|
||||||
|
sh_link = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_link_offset),
|
||||||
|
name = "",
|
||||||
|
}
|
||||||
|
if not (entry.sh_name and entry.sh_type and entry.sh_flags and entry.sh_addr
|
||||||
|
and entry.sh_offset and entry.sh_size and entry.sh_link) then
|
||||||
|
return nil, "truncated_section_headers"
|
||||||
|
end
|
||||||
|
return entry, nil
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Walk every section header in `hdr` and return a 1-based array of entries
|
||||||
|
--- (the section at logical index 0 is at array position 1, etc.).
|
||||||
|
--- Each entry has the wire fields plus a resolved `name` derived from `.shstrtab`.
|
||||||
|
--- Returns nil + a stable error code on failure: truncated_section_headers, missing_shstrtab, truncated_strtab
|
||||||
|
--- @param adapter Elf32Adapter
|
||||||
|
--- @param hdr Elf32Header
|
||||||
|
--- @return Elf32Section[]|nil, string|nil
|
||||||
|
function M.walk_sections(adapter, hdr)
|
||||||
|
if not hdr or hdr.error then return nil, hdr and hdr.error or "truncated_section_headers" end
|
||||||
|
|
||||||
|
local file_size = M.size(adapter) ---@type integer
|
||||||
|
if hdr.e_shoff + hdr.e_shnum * hdr.e_shentsize > file_size then
|
||||||
|
return nil, "truncated_section_headers"
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Read every section header first; we need .shstrtab to resolve names.
|
||||||
|
local sections = {} ---@type Elf32Section[]
|
||||||
|
for i = 0, hdr.e_shnum - 1 do ---@type integer
|
||||||
|
local sh_off = hdr.e_shoff + i * hdr.e_shentsize ---@type integer
|
||||||
|
local entry, err = read_section_entry(adapter, sh_off) ---@type Elf32Section|nil, string|nil
|
||||||
|
if not entry then return nil, err end
|
||||||
|
sections[i + 1] = entry
|
||||||
|
end
|
||||||
|
|
||||||
|
if hdr.e_shstrndx >= hdr.e_shnum then
|
||||||
|
return nil, "missing_shstrtab"
|
||||||
|
end
|
||||||
|
|
||||||
|
local shstrtab = sections[hdr.e_shstrndx + 1] ---@type Elf32Section|nil
|
||||||
|
if not shstrtab or shstrtab.sh_type ~= M.SHT_STRTAB then
|
||||||
|
return nil, "missing_shstrtab"
|
||||||
|
end
|
||||||
|
if shstrtab.sh_offset + shstrtab.sh_size > file_size then
|
||||||
|
return nil, "truncated_section_headers"
|
||||||
|
end
|
||||||
|
local shstrtab_bytes = M.read_section_bytes(adapter, shstrtab) ---@type string|nil
|
||||||
|
if not shstrtab_bytes then return nil, "truncated_section_headers" end
|
||||||
|
|
||||||
|
for _, s in ipairs(sections) do ---@type integer, Elf32Section
|
||||||
|
s.name = M.get_str(shstrtab_bytes, s.sh_name) or ""
|
||||||
|
end
|
||||||
|
|
||||||
|
return sections, nil
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Read the bytes of one section. Returns a string, or nil if the adapter returns nil for any byte (out-of-bounds).
|
||||||
|
--- The caller is responsible fors sizing the buffer (the section's sh_offset + sh_size must fit in adapter.size).
|
||||||
|
--- @param adapter Elf32Adapter
|
||||||
|
--- @param section Elf32Section
|
||||||
|
--- @return string|nil
|
||||||
|
function M.read_section_bytes(adapter, section)
|
||||||
|
local size = section.sh_size ---@type integer
|
||||||
|
if size == 0 then return "" end
|
||||||
|
local out = {} ---@type string[]
|
||||||
|
for i = 0, size - 1 do ---@type integer
|
||||||
|
local b = M.read_u8(adapter, section.sh_offset + i) ---@type integer|nil
|
||||||
|
if b == nil then return nil end
|
||||||
|
out[#out + 1] = string.char(b)
|
||||||
|
end
|
||||||
|
return table.concat(out)
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Convenience: walk sections, then look up the named section, then read its bytes.
|
||||||
|
--- Returns nil + a stable error code if the section is absent or out-of-bounds.
|
||||||
|
--- @param adapter Elf32Adapter
|
||||||
|
--- @param sections Elf32Section[]
|
||||||
|
--- @param name string
|
||||||
|
--- @return string|nil, string|nil
|
||||||
|
function M.read_named_section(adapter, sections, name)
|
||||||
|
if not sections then return nil, "missing_section" end
|
||||||
|
for _, s in ipairs(sections) do ---@type integer, Elf32Section
|
||||||
|
if s.name == name then
|
||||||
|
local bytes = M.read_section_bytes(adapter, s) ---@type string|nil
|
||||||
|
if not bytes then return nil, "truncated_section_data" end
|
||||||
|
return bytes, nil
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return nil, "missing_section"
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Walk every SHT_SYMTAB section in `sections` and accumulate symbols by name.
|
||||||
|
--- Each stored entry is `{ value = st_value, size = st_size, info = st_info, shndx = st_shndx }`.
|
||||||
|
--- Both STB_LOCAL and STB_GLOBAL symbols are included; the live ELF stores `smem` as a local symbol.
|
||||||
|
--- Returns nil + a stable error code on failure: missing_symtab_strtab, truncated_section_headers
|
||||||
|
--- @param adapter Elf32Adapter
|
||||||
|
--- @param sections Elf32Section[]
|
||||||
|
--- @return table<string, Elf32Sym>|nil, string|nil
|
||||||
|
function M.collect_symbols(adapter, sections)
|
||||||
|
if not sections then return nil, "missing_sections" end
|
||||||
|
local symbols = {} ---@type table<string, Elf32Sym> -- bag: symbol name -> Elf32Sym
|
||||||
|
local file_size = M.size(adapter) ---@type integer
|
||||||
|
for _, s in ipairs(sections) do ---@type integer, Elf32Section
|
||||||
|
if s.sh_type == M.SHT_SYMTAB then
|
||||||
|
local strtab = sections[s.sh_link + 1] ---@type Elf32Section|nil
|
||||||
|
if not strtab or strtab.sh_type ~= M.SHT_STRTAB then
|
||||||
|
return nil, "missing_symtab_strtab"
|
||||||
|
end
|
||||||
|
if strtab.sh_offset + strtab.sh_size > file_size then
|
||||||
|
return nil, "truncated_section_headers"
|
||||||
|
end
|
||||||
|
local strtab_bytes = M.read_section_bytes(adapter, strtab) ---@type string|nil
|
||||||
|
if not strtab_bytes then return nil, "truncated_section_headers" end
|
||||||
|
if s.sh_offset + s.sh_size > file_size then
|
||||||
|
return nil, "truncated_section_headers"
|
||||||
|
end
|
||||||
|
local symtab_bytes = M.read_section_bytes(adapter, s) ---@type string|nil
|
||||||
|
if not symtab_bytes then return nil, "truncated_section_headers" end
|
||||||
|
local n = #symtab_bytes / M.ELF32_SYM.sym_entry_bytes ---@type number
|
||||||
|
for j = 0, n - 1 do ---@type integer
|
||||||
|
local e = s.sh_offset + j * M.ELF32_SYM.sym_entry_bytes ---@type integer
|
||||||
|
local st_name = M.read_u32(adapter, e + M.ELF32_SYM.st_name) ---@type integer|nil
|
||||||
|
if st_name then
|
||||||
|
local st_value = M.read_u32(adapter, e + M.ELF32_SYM.st_value) ---@type integer|nil
|
||||||
|
local st_size = M.read_u32(adapter, e + M.ELF32_SYM.st_size) ---@type integer|nil
|
||||||
|
local st_info = M.read_u8(adapter, e + M.ELF32_SYM.st_info) ---@type integer|nil
|
||||||
|
-- st_shndx is at offset 14 (2 bytes) — derived from the layout
|
||||||
|
-- the metaprogram reads too. Inline the read to keep the
|
||||||
|
-- adapter as the only I/O surface.
|
||||||
|
local b1 = M.read_u8(adapter, e + 14) ---@type integer|nil
|
||||||
|
local b2 = M.read_u8(adapter, e + 15) ---@type integer|nil
|
||||||
|
if not (b1 and b2) then
|
||||||
|
return nil, "truncated_section_headers"
|
||||||
|
end
|
||||||
|
local st_shndx = b1 + b2 * 0x100 ---@type integer
|
||||||
|
local name = M.get_str(strtab_bytes, st_name) or "" ---@type string
|
||||||
|
if name ~= "" then
|
||||||
|
symbols[name] = {
|
||||||
|
value = st_value,
|
||||||
|
size = st_size,
|
||||||
|
info = st_info,
|
||||||
|
shndx = st_shndx,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return symbols, nil
|
||||||
|
end
|
||||||
|
|
||||||
|
return M
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,105 @@
|
|||||||
|
# scripts/gdb/gdb_tape_atoms.gdb
|
||||||
|
#
|
||||||
|
# Wrapper for the tape-atom step-debug helpers.
|
||||||
|
# The 9 user commands are defined here as STUBS (degraded-state messages).
|
||||||
|
# The real implementations + the per-atom data tables are emitted by `passes/atoms_source_map.lua`
|
||||||
|
# (post-link invocation: `ps1_meta.lua --atoms-source-map --gdb-runtime --elf <elf>`) into `build/gdb_tape_atoms_runtime.gdb`.
|
||||||
|
# Sourcing that file RE-DEFINES the commands with real implementations.
|
||||||
|
#
|
||||||
|
# If `build/gdb_tape_atoms_runtime.gdb` is missing or stale, the stubs remain (E1: no source map).
|
||||||
|
# The user just needs to re-run `build_psyq.ps1` to regenerate.
|
||||||
|
|
||||||
|
# ?? Stub commands (defined here so they're always present, even if the runtime file is missing). The runtime file overrides these if sourced. ??
|
||||||
|
|
||||||
|
define tape_atoms
|
||||||
|
echo "[gdb_tape_atoms] STUB: runtime file build/gdb_tape_atoms_runtime.gdb not found."
|
||||||
|
echo "[gdb_tape_atoms] STUB: run .\\build_psyq.ps1 to regenerate, then re-source this file."
|
||||||
|
end
|
||||||
|
document tape_atoms
|
||||||
|
List every tape atom symbol in the loaded ELF with its .rodata address and word count.
|
||||||
|
STUB state: runtime file not sourced. Run build_psyq.ps1 to regenerate.
|
||||||
|
end
|
||||||
|
|
||||||
|
define break_atom
|
||||||
|
echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||||
|
end
|
||||||
|
document break_atom
|
||||||
|
Set a breakpoint at the start of tape atom <name>. STUB state.
|
||||||
|
end
|
||||||
|
|
||||||
|
define step_atom
|
||||||
|
echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||||
|
end
|
||||||
|
document step_atom
|
||||||
|
Resume execution until the next atom boundary. STUB state.
|
||||||
|
end
|
||||||
|
|
||||||
|
define next_atom
|
||||||
|
echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||||
|
end
|
||||||
|
document next_atom
|
||||||
|
Alias for step_atom. STUB state.
|
||||||
|
end
|
||||||
|
|
||||||
|
define where_in_atom
|
||||||
|
echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||||
|
end
|
||||||
|
document where_in_atom
|
||||||
|
Report current atom name, .rodata addr, word offset, and source line (if known). STUB state.
|
||||||
|
end
|
||||||
|
|
||||||
|
define stepi_inside_atom
|
||||||
|
echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||||
|
end
|
||||||
|
document stepi_inside_atom
|
||||||
|
One MIPS-instruction step, then where_in_atom. STUB state.
|
||||||
|
end
|
||||||
|
|
||||||
|
define show_c2
|
||||||
|
printf "C2[ 0] 0x%08x\n", $c2_data[0]
|
||||||
|
printf "C2[ 7] 0x%08x [otz]\n", $c2_data[7]
|
||||||
|
printf "C2[12] 0x%08x [sxy0]\n", $c2_data[12]
|
||||||
|
printf "C2[13] 0x%08x [sxy1]\n", $c2_data[13]
|
||||||
|
printf "C2[14] 0x%08x [sxy2]\n", $c2_data[14]
|
||||||
|
printf "C2[24] 0x%08x [mac0]\n", $c2_data[24]
|
||||||
|
printf "...\n"
|
||||||
|
echo "(STUB state: only 7 representative regs shown. Run build_psyq.ps1 for full dump.)"
|
||||||
|
end
|
||||||
|
document show_c2
|
||||||
|
Pretty-print all 32 C2 data registers as hex + named alias. STUB state (7 reg subset).
|
||||||
|
end
|
||||||
|
|
||||||
|
define show_c2ctl
|
||||||
|
printf "C2CTL[ 0] 0x%08x\n", $c2_control[0]
|
||||||
|
printf "...\n"
|
||||||
|
echo "(STUB state: only 1 reg shown. Run build_psyq.ps1 for full dump.)"
|
||||||
|
end
|
||||||
|
document show_c2ctl
|
||||||
|
Pretty-print all 32 C2 control registers. STUB state (1 reg subset).
|
||||||
|
end
|
||||||
|
|
||||||
|
define wave_ctx
|
||||||
|
printf "$t4 = R_FaceCursor 0x%08x\n", $t4
|
||||||
|
printf "$t5 = R_VertBase 0x%08x\n", $t5
|
||||||
|
printf "$t6 = R_OtBase 0x%08x\n", $t6
|
||||||
|
printf "$t7 = R_PrimCursor 0x%08x\n", $t7
|
||||||
|
end
|
||||||
|
document wave_ctx
|
||||||
|
Pretty-print the 4 wave-context GPRs ($t4..$t7). (wave_ctx works in stub state too.)
|
||||||
|
end
|
||||||
|
|
||||||
|
|
||||||
|
# ?? Source the runtime file (re-defines commands with real impls + data). ??
|
||||||
|
|
||||||
|
# Try to source from project-root-relative path first (the typical case).
|
||||||
|
# If the user is in a different CWD, the source will fail and stubs remain.
|
||||||
|
# The runtime file path is computed relative to the ELF's source map convention (build/gdb_tape_atoms_runtime.gdb).
|
||||||
|
echo [gdb_tape_atoms] Wrapper loaded. Sourcing runtime file...
|
||||||
|
# Suppress the "Redefine command" prompts that would otherwise appear when the runtime file overrides the 9 stub commands defined above.
|
||||||
|
# The runtime's `define` blocks are intended to overwrite ? there's no ambiguity to confirm.
|
||||||
|
set confirm off
|
||||||
|
|
||||||
|
# Source the runtime file (re-defines commands with real impls + data).
|
||||||
|
source build/gdb_tape_atoms_runtime.gdb
|
||||||
|
set confirm on
|
||||||
|
echo [gdb_tape_atoms] Runtime sourced successfully (9 commands now have real implementations).
|
||||||
@@ -0,0 +1,94 @@
|
|||||||
|
# scripts/launch_pcsx_debug.ps1
|
||||||
|
#
|
||||||
|
# One-shot launcher for debug sessions:
|
||||||
|
# Starts pcsx-redux with the .ps-exe loaded, the gdb stub enabled,
|
||||||
|
# AND the pcsx_debug_helper Lua plugin loaded so external CLI tools (gdb's `shell` command, etc.)
|
||||||
|
# can read GTE state via http://localhost:8080/api/v1/lua/gte (the gdb stub doesn't expose COP2 at all).
|
||||||
|
#
|
||||||
|
# usage:
|
||||||
|
# .\scripts\launch_pcsx_debug.ps1
|
||||||
|
# .\scripts\launch_pcsx_debug.ps1 -ExePath build\hello_gte.ps-exe
|
||||||
|
# .\scripts\launch_pcsx_debug.ps1 -HelperZip scripts\pcsx_debug_helper.zip
|
||||||
|
#
|
||||||
|
# After launch:
|
||||||
|
# - gdb: target remote localhost:3333
|
||||||
|
# - web: curl http://localhost:8080/api/v1/lua/gte
|
||||||
|
#
|
||||||
|
# Companion: scripts/debug_psyq.ps1 (bare launch — no .ps-exe, no helper).
|
||||||
|
|
||||||
|
[CmdletBinding()]
|
||||||
|
param(
|
||||||
|
[string]$PcsxPath = (Join-Path $PSScriptRoot '..\toolchain\pcsx-redux\vsprojects\x64\Release\pcsx-redux.exe'),
|
||||||
|
[string]$ExePath = (Join-Path $PSScriptRoot '..\build\hello_gte.ps-exe'),
|
||||||
|
[string]$HelperZip = (Join-Path $PSScriptRoot 'pcsx_debug_helper.zip'),
|
||||||
|
[int] $GdbPort = 3333,
|
||||||
|
[int] $WebPort = 8080
|
||||||
|
)
|
||||||
|
|
||||||
|
$ErrorActionPreference = 'Stop'
|
||||||
|
|
||||||
|
# ── Pre-checks ──
|
||||||
|
foreach ($p in @($PcsxPath, $ExePath, $HelperZip)) {
|
||||||
|
if (-not (Test-Path $p)) {
|
||||||
|
Write-Error "Missing: $p"
|
||||||
|
exit 1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
# Kill any existing pcsx-redux so the archive file isn't locked.
|
||||||
|
Get-Process pcsx-redux -ErrorAction SilentlyContinue | Stop-Process -Force
|
||||||
|
Start-Sleep -Seconds 2
|
||||||
|
|
||||||
|
# ── Launch ──
|
||||||
|
$absExe = [System.IO.Path]::GetFullPath($ExePath)
|
||||||
|
$absZip = [System.IO.Path]::GetFullPath($HelperZip)
|
||||||
|
|
||||||
|
$args = @(
|
||||||
|
'-gdb', '-run'
|
||||||
|
'-loadexe', "`"$absExe`""
|
||||||
|
'-archive', "`"$absZip`""
|
||||||
|
)
|
||||||
|
|
||||||
|
Write-Host "Launching pcsx-redux..." -ForegroundColor Cyan
|
||||||
|
Write-Host " ps-exe : $absExe"
|
||||||
|
Write-Host " helper zip: $absZip"
|
||||||
|
Write-Host " gdb : localhost:$GdbPort"
|
||||||
|
Write-Host " web : localhost:$WebPort/api/v1/lua/gte"
|
||||||
|
Write-Host ""
|
||||||
|
|
||||||
|
Start-Process -FilePath $PcsxPath -ArgumentList $args | Out-Null
|
||||||
|
|
||||||
|
# ── Wait for both endpoints to come up ──
|
||||||
|
$deadline = (Get-Date).AddSeconds(15)
|
||||||
|
while ((Get-Date) -lt $deadline) {
|
||||||
|
$gdbUp = $false
|
||||||
|
$webUp = $false
|
||||||
|
try {
|
||||||
|
$tcp = New-Object System.Net.Sockets.TcpClient
|
||||||
|
$tcp.BeginConnect('localhost', $GdbPort, $null, $null) | Out-Null
|
||||||
|
Start-Sleep -Milliseconds 100
|
||||||
|
$gdbUp = $tcp.Connected
|
||||||
|
$tcp.Close()
|
||||||
|
} catch { $gdbUp = $false }
|
||||||
|
try {
|
||||||
|
$r = Invoke-WebRequest -Uri "http://localhost:$WebPort/" -UseBasicParsing -TimeoutSec 1 -ErrorAction SilentlyContinue
|
||||||
|
$webUp = $r.StatusCode -ne 0
|
||||||
|
} catch { $webUp = $false }
|
||||||
|
if ($gdbUp -and $webUp) { break }
|
||||||
|
Start-Sleep -Milliseconds 500
|
||||||
|
}
|
||||||
|
|
||||||
|
# ── Smoke-test the gte handler ──
|
||||||
|
try {
|
||||||
|
$r = Invoke-WebRequest -Uri "http://localhost:$WebPort/api/v1/lua/gte" -UseBasicParsing -TimeoutSec 5
|
||||||
|
$firstLine = ([System.Text.Encoding]::UTF8.GetString($r.Content) -split "`n")[0]
|
||||||
|
Write-Host "GTE handler OK: $firstLine" -ForegroundColor Green
|
||||||
|
}
|
||||||
|
catch {
|
||||||
|
Write-Warning "GTE handler NOT responding: $_"
|
||||||
|
Write-Host "Check the pcsx-redux Lua Console for debug cli messages." -ForegroundColor Yellow
|
||||||
|
}
|
||||||
|
|
||||||
|
Write-Host ""
|
||||||
|
Write-Host "pcsx-redux running. PIDs:" -ForegroundColor Cyan
|
||||||
|
Get-Process pcsx-redux | Select-Object Id, ProcessName | Format-Table
|
||||||
@@ -0,0 +1,563 @@
|
|||||||
|
--- passes/annotation.lua — Atom-annotation DSL validator.
|
||||||
|
---
|
||||||
|
--- Validates `MipsAtom_(name) atom_info(atom_bind(Binds_X), atom_reads(...), atom_writes(...)) { ... }` declarations in source files.
|
||||||
|
--- Also reads `Binds_*` struct declarations (`typedef Struct_(Binds_X) { ... };`).
|
||||||
|
---
|
||||||
|
--- `duffle.scan_source()` scans each source once upstream; `ps1_meta.lua` stores that result in `src.scan`.
|
||||||
|
---
|
||||||
|
--- Ownership: the canonical `ctx.shared.corpus` supplies cross-source registries, while each `src.scan` supplies its source's declarations and bodies.
|
||||||
|
--- A context without `ctx.shared.corpus` is rejected with an explicit canonical-corpus message.
|
||||||
|
|
||||||
|
-- Bootstrap follows the entry scripts; `scripts/duffle_paths.lua` sets package.path and package.cpath. See `ps1_meta.lua` for the rationale.
|
||||||
|
-- `debug.getinfo(1, "S").source` locates this file for standalone and orchestrated runs, then `duffle_paths.lua` returns the loaded `duffle` module.
|
||||||
|
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" ---@type string
|
||||||
|
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") ---@type DuffleExport
|
||||||
|
|
||||||
|
-- The annotation pass reads the source-derived registries from scan_source:
|
||||||
|
-- * pipe_ctx.register_alias_registry — for atom_dbg_reg_default(R_X, ...) and atom_reg_types(R_X, ...) member-identity checks
|
||||||
|
-- * pipe_ctx.type_name_registry — for atom_dbg_reg_default(<T>, ...) and atom_reg_types(<T>, ...) type-identity checks
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Type declarations
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
-- SourceFile, PassCtx, PassResult, PassShared, Corpus, Finding: see ps1_meta.lua
|
||||||
|
-- SourceScan, AtomEntry, AtomInfoEntry, BindsEntry, RegTypeDefault, AtomViewEntry: see scan_source.lua
|
||||||
|
|
||||||
|
--- @class RegTypeOccurrence
|
||||||
|
--- @field reg string
|
||||||
|
--- @field type_name string
|
||||||
|
--- @field source_line integer
|
||||||
|
|
||||||
|
--- @class Findings
|
||||||
|
--- @field errors Finding[]
|
||||||
|
--- @field warnings Finding[]
|
||||||
|
--- @field info Finding[]
|
||||||
|
|
||||||
|
-- PassScratch: see ps1_meta.lua
|
||||||
|
|
||||||
|
--- @class AnnotatedResult
|
||||||
|
--- @field atoms AtomEntry[]
|
||||||
|
--- @field annots AtomInfoEntry[]
|
||||||
|
--- @field macros MacroEntry[]
|
||||||
|
--- @field binds BindsEntry[]
|
||||||
|
--- @field errors Finding[]
|
||||||
|
--- @field warnings Finding[]
|
||||||
|
--- @field info Finding[]
|
||||||
|
--- @field source string|nil
|
||||||
|
|
||||||
|
--- @class CheckRule
|
||||||
|
--- @field per_annot (fun(item: AtomInfoEntry, pipe_ctx: PassScratch, findings: Findings): nil)|nil
|
||||||
|
|
||||||
|
--- @class SourceScan
|
||||||
|
--- @field type_occurrences RegTypeOccurrence[]|nil
|
||||||
|
|
||||||
|
--- @class AnnotationPass
|
||||||
|
--- @field validate fun(ctx: PassCtx, src: SourceFile, corpus_pipe_ctx: PassScratch|nil): AnnotatedResult
|
||||||
|
--- @field run fun(ctx: PassCtx): PassResult
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Per-check functions (the CHECK_RULES table's payload)
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
--- The dispatcher in `validate()` routes each result by convention: existence checks write errors[] and shape checks write warnings[].
|
||||||
|
--- `macro_word_drift` writes errors[] for missing or mismatched metadata and info[] for a match.
|
||||||
|
|
||||||
|
--- Check: Every annotated atom must have a matching MipsAtom_(name) declaration.
|
||||||
|
--- @param info AtomInfoEntry
|
||||||
|
--- @param pipe_ctx PassScratch
|
||||||
|
--- @param findings Findings
|
||||||
|
--- @return nil
|
||||||
|
local function check_atom_decl_exists(info, pipe_ctx, findings)
|
||||||
|
if not pipe_ctx.atom_index[info.atom_name] then
|
||||||
|
findings.errors[#findings.errors + 1] = {
|
||||||
|
line = info.info_line,
|
||||||
|
msg = string.format("annotation for '%s' has no matching MipsAtom_(%s) { ... }", info.atom_name, info.atom_name),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Check: Every atom may have AT MOST ONE annotation.
|
||||||
|
--- Post-loop: Needs full-corpus `annot_counts` from pipe_ctx.
|
||||||
|
--- @param _item AtomInfoEntry|nil
|
||||||
|
--- @param pipe_ctx PassScratch
|
||||||
|
--- @param findings Findings
|
||||||
|
--- @return nil
|
||||||
|
local function check_unique_annotation(_item, pipe_ctx, findings)
|
||||||
|
for name, n in pairs(pipe_ctx.annot_counts) do ---@type string, integer
|
||||||
|
if n > 1 then
|
||||||
|
findings.errors[#findings.errors + 1] = {
|
||||||
|
line = pipe_ctx.atom_index[name] and pipe_ctx.atom_index[name].line or 0,
|
||||||
|
msg = string.format("MipsAtom_(%s) has %d annotations (expected at most 1)", name, n),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Check: BIND atoms must reference a real Binds_* struct.
|
||||||
|
--- I keep this as a warning so the annotation pass can report the common test-fixture case; `check_abi_handoff` in static analysis supplies the build-stopping error.
|
||||||
|
--- @param info AtomInfoEntry
|
||||||
|
--- @param pipe_ctx PassScratch
|
||||||
|
--- @param findings Findings
|
||||||
|
--- @return nil
|
||||||
|
local function check_binds_struct_exists(info, pipe_ctx, findings)
|
||||||
|
if not info.binds then return end
|
||||||
|
if pipe_ctx.binds_index[info.binds] then return end
|
||||||
|
findings.warnings[#findings.warnings + 1] = {
|
||||||
|
line = info.info_line,
|
||||||
|
msg = string.format("'%s' binds '%s' but no Struct_(%s) { ... } "
|
||||||
|
.. "declaration found (also flagged as an error by check_abi_handoff in the static-analysis pass)"
|
||||||
|
, info.atom_name, info.binds, info.binds),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Check: TAPE_WORDS(mac_X, N) ↔ WORD_COUNT(mac_X, N) drift.
|
||||||
|
--- Three outcomes: missing (error), mismatch (error), match (info).
|
||||||
|
--- @param m MacroEntry
|
||||||
|
--- @param pipe_ctx PassScratch
|
||||||
|
--- @param findings Findings
|
||||||
|
--- @return nil
|
||||||
|
local function check_macro_word_drift(m, pipe_ctx, findings)
|
||||||
|
local wc = (pipe_ctx and pipe_ctx.word_counts) or {} ---@type WordCounts
|
||||||
|
local declared = wc[m.name] ---@type integer|nil
|
||||||
|
if not declared then
|
||||||
|
findings.errors[#findings.errors + 1] = {
|
||||||
|
line = m.line,
|
||||||
|
msg = string.format("TAPE_WORDS(%s, %d) but '%s' is not in metadata.h", m.name, m.words, m.name),
|
||||||
|
}
|
||||||
|
return
|
||||||
|
end
|
||||||
|
if declared ~= m.words then
|
||||||
|
findings.errors[#findings.errors + 1] = {
|
||||||
|
line = m.line,
|
||||||
|
msg = string.format("DRIFT: TAPE_WORDS(%s, %d) but metadata.h declares WORD_COUNT(%s, %d)", m.name, m.words, m.name, declared),
|
||||||
|
}
|
||||||
|
return
|
||||||
|
end
|
||||||
|
findings.info[#findings.info + 1] = {
|
||||||
|
line = m.line,
|
||||||
|
msg = string.format("OK: %s = %d words", m.name, m.words),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Check: atom_dbg_reg_default(R_X, <type>) targets an alias in `pipe_ctx.register_alias_registry` and a type in `pipe_ctx.type_name_registry`.
|
||||||
|
--- Pointer depth remains bounded to 0 or 1, and duplicate defaults remain errors.
|
||||||
|
--- @param _src SourceFile -- unused (kept for the per_source shape)
|
||||||
|
--- @param pipe_ctx PassScratch
|
||||||
|
--- @param findings Findings
|
||||||
|
--- @return nil
|
||||||
|
local function check_semantic_reg_defaults(_src, pipe_ctx, findings)
|
||||||
|
-- Detect duplicate defaults using the ordered occurrence list (the out.types hash only retains the last declaration).
|
||||||
|
local seen_first_line = {} ---@type table<string, integer> -- bag: register ident -> first source line
|
||||||
|
for _, occ in ipairs(pipe_ctx.type_occurrences or {}) do ---@type integer, RegTypeOccurrence
|
||||||
|
if seen_first_line[occ.reg] == nil then
|
||||||
|
seen_first_line[occ.reg] = occ.source_line
|
||||||
|
else
|
||||||
|
findings.errors[#findings.errors + 1] = {
|
||||||
|
line = occ.source_line,
|
||||||
|
msg = string.format(
|
||||||
|
"duplicate atom_dbg_reg_default for %q at line %d (first declared at line %d); one default per register",
|
||||||
|
occ.reg, occ.source_line, seen_first_line[occ.reg]),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
local reg_registry = pipe_ctx.register_alias_registry or {} ---@type table<string, AliasEntry>
|
||||||
|
local type_registry = pipe_ctx.type_name_registry or {} ---@type table<string, TypeNameEntry>
|
||||||
|
for reg, def in pairs(pipe_ctx.types or {}) do ---@type string, RegTypeDefault
|
||||||
|
if not reg_registry[reg] then
|
||||||
|
findings.errors[#findings.errors + 1] = {
|
||||||
|
line = def.source_line,
|
||||||
|
msg = string.format(
|
||||||
|
"atom_dbg_reg_default at line %d references unknown register %q (not in register_alias_registry)",
|
||||||
|
def.source_line, reg),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
if def.pointer_depth == nil or def.pointer_depth < 0 or def.pointer_depth > 1 then
|
||||||
|
findings.errors[#findings.errors + 1] = {
|
||||||
|
line = def.source_line,
|
||||||
|
msg = string.format(
|
||||||
|
"atom_dbg_reg_default at line %d for %q has unsupported pointer depth %d (expected 0 or 1)",
|
||||||
|
def.source_line, reg, def.pointer_depth or -1),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
if not def.type_name or not type_registry[def.type_name] then
|
||||||
|
findings.errors[#findings.errors + 1] = {
|
||||||
|
line = def.source_line,
|
||||||
|
msg = string.format(
|
||||||
|
"atom_dbg_reg_default at line %d for %q uses unknown type %q (not in type_name_registry)",
|
||||||
|
def.source_line, reg, tostring(def.type_name)),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Check: atom_reg_types(R_X, <type>) entries target an alias in `pipe_ctx.register_alias_registry` and a type in `pipe_ctx.type_name_registry`.
|
||||||
|
--- A bare `atom_reg` marker opts the `R_<n>` alias into GPR identity; references to R_T0..R_T3 require the same explicit marker.
|
||||||
|
--- @param _src SourceFile
|
||||||
|
--- @param pipe_ctx PassScratch
|
||||||
|
--- @param findings Findings
|
||||||
|
--- @return nil
|
||||||
|
local function check_atom_reg_types(_src, pipe_ctx, findings)
|
||||||
|
local reg_registry = pipe_ctx.register_alias_registry or {} ---@type table<string, AliasEntry>
|
||||||
|
local type_registry = pipe_ctx.type_name_registry or {} ---@type table<string, TypeNameEntry>
|
||||||
|
for _, ai in ipairs(pipe_ctx.atom_infos_list or {}) do ---@type integer, AtomInfoEntry
|
||||||
|
if ai.reg_type_overrides then
|
||||||
|
for reg, ov in pairs(ai.reg_type_overrides) do ---@type string, RegTypeOverride
|
||||||
|
if not reg_registry[reg] then
|
||||||
|
findings.errors[#findings.errors + 1] = {
|
||||||
|
line = ai.info_line,
|
||||||
|
msg = string.format(
|
||||||
|
"atom '%s' has atom_reg_types for %q; compute-register types are restricted to opt-in aliases (%q not in register_alias_registry)",
|
||||||
|
ai.atom_name, reg, reg),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
if not ov.type_name or not type_registry[ov.type_name] then
|
||||||
|
findings.errors[#findings.errors + 1] = {
|
||||||
|
line = ai.info_line,
|
||||||
|
msg = string.format(
|
||||||
|
"atom '%s' atom_reg_types for %q uses unknown compute type %q (not in type_name_registry)",
|
||||||
|
ai.atom_name, reg, tostring(ov.type_name)),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Check: atom_view(Binds_X) entries reference a Binds_* struct with at least one field.
|
||||||
|
--- @param _src SourceFile
|
||||||
|
--- @param pipe_ctx PassScratch
|
||||||
|
--- @param findings Findings
|
||||||
|
--- @return nil
|
||||||
|
local function check_atom_view_layout(_src, pipe_ctx, findings)
|
||||||
|
for atom_name, view in pairs(pipe_ctx.atom_views or {}) do ---@type string, AtomViewEntry
|
||||||
|
if not view.binds_name then
|
||||||
|
-- The atom had atom_reg_types but no atom_view; no layout check needed.
|
||||||
|
else
|
||||||
|
local bs = pipe_ctx.binds_index[view.binds_name] ---@type BindsEntry|nil
|
||||||
|
if not bs then
|
||||||
|
findings.errors[#findings.errors + 1] = {
|
||||||
|
line = view.info_line,
|
||||||
|
msg = string.format(
|
||||||
|
"atom '%s' has atom_view(%s) but no Struct_(%s) { ... } declaration was found",
|
||||||
|
atom_name, view.binds_name, view.binds_name),
|
||||||
|
}
|
||||||
|
elseif not bs.fields or #bs.fields == 0 then
|
||||||
|
findings.errors[#findings.errors + 1] = {
|
||||||
|
line = bs.line,
|
||||||
|
msg = string.format(
|
||||||
|
"atom '%s' has atom_view(%s) but that struct declares zero typed fields",
|
||||||
|
atom_name, view.binds_name),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Check: Binds_* structs require unique field names because atom_view uses those names for typed-field lookup in gdb.
|
||||||
|
--- @param _src SourceFile
|
||||||
|
--- @param pipe_ctx PassScratch
|
||||||
|
--- @param findings Findings
|
||||||
|
--- @return nil
|
||||||
|
local function check_binds_no_duplicate_fields(_src, pipe_ctx, findings)
|
||||||
|
for _, bs in ipairs(pipe_ctx.binds_list or {}) do ---@type integer, BindsEntry
|
||||||
|
local seen = {} ---@type table<string, integer> -- bag: field name -> occurrence count
|
||||||
|
for _, f in ipairs(bs.fields or {}) do ---@type integer, TypeField
|
||||||
|
seen[f.name] = (seen[f.name] or 0) + 1
|
||||||
|
end
|
||||||
|
for name, count in pairs(seen) do ---@type string, integer
|
||||||
|
if count > 1 then
|
||||||
|
findings.errors[#findings.errors + 1] = {
|
||||||
|
line = bs.line,
|
||||||
|
msg = string.format(
|
||||||
|
"%s has duplicate field name %q (count %d); the typed-view contract requires unique field names",
|
||||||
|
bs.name, name, count),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Check: Debug-skip markers must satisfy shape + placement constraints.
|
||||||
|
--- Walks the priority list once; each marker produces at most one error, so one source defect yields one finding.
|
||||||
|
--- Priority order (first defect wins):
|
||||||
|
--- 1. marker_kind ~= "atom_dbg_skip" -> legacy/renamed spelling (use `atom_dbg_skip`)
|
||||||
|
--- 2. marker_kind == "atom_dbg_skip" AND has_parens -> parenthesized form (the marker is bare-only)
|
||||||
|
--- 3. args ~= "" -> takes no arguments
|
||||||
|
--- 4. superseded_by_marker_line -> duplicate marker (cite superseding line)
|
||||||
|
--- 5. pending + no target_kind -> dangling (no following declaration)
|
||||||
|
--- 6. unsupported target_kind -> marker precedes an unrelated declaration
|
||||||
|
--- Valid markers stamp `debug_skip` on whole-atom, bare-component, and proc-component declaration records in scan_source.lua.
|
||||||
|
--- @param marker DebugSkipMarker
|
||||||
|
--- @param _pipe_ctx PassScratch -- Unused; kept for consistency with per_annot
|
||||||
|
--- @param findings Findings
|
||||||
|
--- @return nil
|
||||||
|
local function check_skip_marker(marker, _pipe_ctx, findings)
|
||||||
|
local kind = marker.marker_kind ---@type string
|
||||||
|
local line = marker.marker_line ---@type integer
|
||||||
|
-- Left `scan.debug_skip_markers` with production records for `atom_dbg_skip` only; other identifiers take the walker's unrelated branch.
|
||||||
|
|
||||||
|
if marker.has_parens then
|
||||||
|
findings.errors[#findings.errors + 1] = {
|
||||||
|
line = line,
|
||||||
|
msg = string.format("%s marker at line %d must be bare; the parenthesized form is no longer accepted (use `atom_dbg_skip MipsAtom_(name) { ... }`)",
|
||||||
|
kind, line),
|
||||||
|
}
|
||||||
|
return
|
||||||
|
end
|
||||||
|
|
||||||
|
if marker.args ~= nil and marker.args ~= "" then
|
||||||
|
findings.errors[#findings.errors + 1] = {
|
||||||
|
line = line,
|
||||||
|
msg = string.format("%s marker at line %d takes no arguments; found %q", kind, line, marker.args),
|
||||||
|
}
|
||||||
|
return
|
||||||
|
end
|
||||||
|
|
||||||
|
if marker.superseded_by_marker_line then
|
||||||
|
findings.errors[#findings.errors + 1] = {
|
||||||
|
line = line,
|
||||||
|
msg = string.format("duplicate %s marker at line %d; superseded by another %s marker at line %d"
|
||||||
|
, kind, line, kind, marker.superseded_by_marker_line),
|
||||||
|
}
|
||||||
|
return
|
||||||
|
end
|
||||||
|
|
||||||
|
if marker.pending and not marker.target_kind then
|
||||||
|
findings.errors[#findings.errors + 1] = {
|
||||||
|
line = line,
|
||||||
|
msg = string.format("dangling %s marker at line %d: no following MipsAtom_/MipsAtomComp_/MipsAtomComp_Proc_ declaration"
|
||||||
|
, kind, line),
|
||||||
|
}
|
||||||
|
return
|
||||||
|
end
|
||||||
|
|
||||||
|
if marker.target_kind
|
||||||
|
and marker.target_kind ~= "atom"
|
||||||
|
and marker.target_kind ~= "comp_bare"
|
||||||
|
and marker.target_kind ~= "comp_proc" then
|
||||||
|
findings.errors[#findings.errors + 1] = {
|
||||||
|
line = line,
|
||||||
|
msg = string.format("%s marker at line %d must precede MipsAtom_, MipsAtomComp_, or MipsAtomComp_Proc_; found an unrelated declaration"
|
||||||
|
, kind, line),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Warn when a source references an unregistered alias.
|
||||||
|
--- When a source uses an unregistered R_X, this check emits one pass-level info entry for that source and directs C-ABI register names to explicit alias registration.
|
||||||
|
--- @param _src SourceFile
|
||||||
|
--- @param pipe_ctx PassScratch
|
||||||
|
--- @param findings Findings
|
||||||
|
--- @return nil
|
||||||
|
local function check_wave_context_migration(_src, pipe_ctx, findings)
|
||||||
|
if not (pipe_ctx.types and next(pipe_ctx.types)) then return end
|
||||||
|
if not (pipe_ctx.atom_infos_list) then return end
|
||||||
|
local reg_registry = pipe_ctx.register_alias_registry or {} ---@type table<string, AliasEntry>
|
||||||
|
for _, ai in ipairs(pipe_ctx.atom_infos_list) do ---@type integer, AtomInfoEntry
|
||||||
|
if ai.reg_type_overrides then
|
||||||
|
for reg, _ in pairs(ai.reg_type_overrides) do ---@type string, RegTypeOverride
|
||||||
|
if not reg_registry[reg] then
|
||||||
|
findings.warnings[#findings.warnings + 1] = {
|
||||||
|
line = 0,
|
||||||
|
msg = "wave-context removed; opt in via #define atom_reg in mips.h "
|
||||||
|
.. "(every R_<alias> that should be visible to the annotation pass "
|
||||||
|
.. "must be enum-declared with the bare atom_reg marker)",
|
||||||
|
}
|
||||||
|
return
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- CHECK_RULES — data-driven check dispatch (the plex pattern)
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
--
|
||||||
|
-- Each rule entry picks one of four "shapes" of dispatch:
|
||||||
|
-- per_annot(info, pipe_ctx, findings) -- runs once per scan.atom_infos row
|
||||||
|
-- post(pipe_ctx, findings) -- runs once after all per_annot calls complete (full-corpus aggregation)
|
||||||
|
-- per_macro(macro, wc, findings) -- runs once per TAPE_WORDS / _Pragma macro declaration
|
||||||
|
-- per_skip_marker(marker, pipe_ctx, findings) -- runs once per src.scan.debug_skip_markers entry
|
||||||
|
--
|
||||||
|
-- Adding a new check = 1 row here + 1 function above. The `validate()` dispatch loop never needs editing.
|
||||||
|
|
||||||
|
local CHECK_RULES = { ---@type CheckRule[]
|
||||||
|
{ name = "atom_decl_exists", per_annot = check_atom_decl_exists },
|
||||||
|
{ name = "binds_struct_exists", per_annot = check_binds_struct_exists },
|
||||||
|
{ name = "unique_annotation", post = check_unique_annotation },
|
||||||
|
{ name = "macro_word_drift", per_macro = check_macro_word_drift },
|
||||||
|
{ name = "skip_marker_validation", per_skip_marker = check_skip_marker },
|
||||||
|
{ name = "semantic_reg_defaults", per_source = check_semantic_reg_defaults },
|
||||||
|
{ name = "atom_reg_types", per_source = check_atom_reg_types },
|
||||||
|
{ name = "atom_view_layout", per_source = check_atom_view_layout },
|
||||||
|
{ name = "binds_no_duplicate_fields", per_source = check_binds_no_duplicate_fields },
|
||||||
|
{ name = "wave_context_migration", per_source = check_wave_context_migration },
|
||||||
|
}
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Validation
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Pure check: Read from src.scan, run validations, emit findings. The scan was done once upstream.
|
||||||
|
|
||||||
|
--- Builds one pass-wide pipe_ctx from the merged `corpus.*` registries and source-ordered `corpus.atom_infos`; per-source declarations and bodies remain in `src.scan`.
|
||||||
|
--- The module ownership contract above requires callers to construct `ctx.shared.corpus` through `build_ctx`; the error message below enforces that gate.
|
||||||
|
--- @param ctx PassCtx
|
||||||
|
--- @return PassScratch
|
||||||
|
local function build_corpus_pipe_ctx(ctx)
|
||||||
|
local view = duffle.corpus_view(ctx) ---@type PassScratch
|
||||||
|
local annot_counts = {} ---@type table<string, integer> -- bag: atom name -> annotation count
|
||||||
|
for _, info in ipairs(view.atom_infos) do ---@type integer, AtomInfoEntry
|
||||||
|
if info and info.atom_name then
|
||||||
|
annot_counts[info.atom_name] = (annot_counts[info.atom_name] or 0) + 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
view.annot_counts = annot_counts
|
||||||
|
view.atom_infos_list = view.atom_infos
|
||||||
|
view.word_counts = ctx.shared.corpus.word_counts or {}
|
||||||
|
return view
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Validate one source against its pre-scanned SourceScan payload + the corpus-wide pipe_ctx.
|
||||||
|
--- @param ctx PassCtx
|
||||||
|
--- @param src SourceFile
|
||||||
|
--- @param corpus_pipe_ctx PassScratch|nil -- Built once per pass from corpus registries; nil builds the same projection here.
|
||||||
|
--- @return AnnotatedResult
|
||||||
|
local function validate(ctx, src, corpus_pipe_ctx)
|
||||||
|
corpus_pipe_ctx = corpus_pipe_ctx or build_corpus_pipe_ctx(ctx)
|
||||||
|
local scan = src.scan ---@type SourceScan
|
||||||
|
|
||||||
|
-- Build a per-source pipe_ctx: shared lookups come from `corpus_pipe_ctx`, while declarations, bodies, types, views, defaults, and occurrences come from `src.scan`.
|
||||||
|
local seen_defaults = {}; for reg, _ in pairs (scan.types or {}) do seen_defaults[reg] = (seen_defaults[reg] or 0) + 1 end ---@type table<string, integer> -- bag: register ident -> occurrence count
|
||||||
|
local atom_infos_list = {}; for _, ai in ipairs(scan.atom_infos or {}) do atom_infos_list[#atom_infos_list + 1] = ai end ---@type AtomInfoEntry[]
|
||||||
|
|
||||||
|
local pipe_ctx = { ---@type PassScratch
|
||||||
|
atom_index = {},
|
||||||
|
binds_index = {},
|
||||||
|
annot_counts = corpus_pipe_ctx.annot_counts,
|
||||||
|
types = scan.types or {},
|
||||||
|
type_occurrences = scan.type_occurrences or {},
|
||||||
|
atom_views = scan.atom_views or {},
|
||||||
|
seen_defaults = seen_defaults,
|
||||||
|
atom_infos_list = atom_infos_list,
|
||||||
|
binds_list = scan.binds or {},
|
||||||
|
-- See the module ownership contract; these shared lookup tables come from corpus_pipe_ctx.
|
||||||
|
register_alias_registry = corpus_pipe_ctx.register_alias_registry,
|
||||||
|
type_name_registry = corpus_pipe_ctx.type_name_registry,
|
||||||
|
}
|
||||||
|
local atoms = {} ---@type AtomEntry[]
|
||||||
|
for _, a in ipairs(scan.atoms) do ---@type integer, AtomEntry
|
||||||
|
if a.kind == "atom" or a.kind == "atom_proc" then
|
||||||
|
atoms[#atoms + 1] = a
|
||||||
|
pipe_ctx.atom_index[a.raw_name or a.name] = a
|
||||||
|
end
|
||||||
|
end
|
||||||
|
for _, b in ipairs(scan.binds) do pipe_ctx.binds_index[b.name] = b end ---@type integer, BindsEntry
|
||||||
|
|
||||||
|
-- Findings live in a single struct with three lists (errors / warnings / info).
|
||||||
|
-- Each check writes to the list appropriate for its severity.
|
||||||
|
local findings = { errors = {}, warnings = {}, info = {} } ---@type Findings
|
||||||
|
|
||||||
|
-- Lift parse-time errors already recorded in scan_source's atom_info payload into this pass's findings list.
|
||||||
|
for _, info in ipairs(scan.atom_infos) do ---@type integer, AtomInfoEntry
|
||||||
|
if info.errors then
|
||||||
|
for _, msg in ipairs(info.errors) do ---@type integer, string
|
||||||
|
findings.errors[#findings.errors + 1] = {
|
||||||
|
line = info.info_line,
|
||||||
|
msg = string.format("'%s': %s", info.atom_name, msg),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- THE per-annotation pipeline. ONE loop. CHECK_RULES dispatches per_annot rules.
|
||||||
|
for _, info in ipairs(scan.atom_infos) do ---@type integer, AtomInfoEntry
|
||||||
|
duffle.run_check_rules(CHECK_RULES, "per_annot", info, pipe_ctx, findings)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Post-loop rules (one-shot checks that need full-corpus aggregation in pipe_ctx).
|
||||||
|
duffle.run_check_rules(CHECK_RULES, "post", nil, pipe_ctx, findings)
|
||||||
|
|
||||||
|
-- scan_source records each marker in scan.debug_skip_markers; this loop validates each record independently and emits at most one error per marker.
|
||||||
|
-- Valid markers stamp `debug_skip = true` on the following atom or component declaration, which downstream consumers read directly.
|
||||||
|
local skip_markers = scan.debug_skip_markers or {} ---@type DebugSkipMarker[]
|
||||||
|
for _, marker in ipairs(skip_markers) do ---@type integer, DebugSkipMarker
|
||||||
|
duffle.run_check_rules(CHECK_RULES, "per_skip_marker", marker, pipe_ctx, findings)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Per-macro rules (TAPE_WORDS vs WORD_COUNT drift).
|
||||||
|
pipe_ctx.word_counts = corpus_pipe_ctx.word_counts
|
||||||
|
for _, m in ipairs(scan.macros) do ---@type integer, MacroEntry
|
||||||
|
duffle.run_check_rules(CHECK_RULES, "per_macro", m, pipe_ctx, findings)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Per-source rules (reg defaults, atom_view layout, compute-register type overrides, Binds_* field uniqueness).
|
||||||
|
-- Each per_source rule sees the full scan payload via pipe_ctx.
|
||||||
|
duffle.run_check_rules(CHECK_RULES, "per_source", src, pipe_ctx, findings)
|
||||||
|
|
||||||
|
-- Information summary (always emitted).
|
||||||
|
findings.info[#findings.info + 1] = {
|
||||||
|
line = 0,
|
||||||
|
msg = string.format("scanned: %d atom(s), %d annotation(s), %d macro-word-decl(s), %d binds struct(s)"
|
||||||
|
, #atoms, #scan.atom_infos, #scan.macros, #scan.binds),
|
||||||
|
}
|
||||||
|
|
||||||
|
return {
|
||||||
|
atoms = atoms,
|
||||||
|
annots = scan.atom_infos,
|
||||||
|
macros = scan.macros,
|
||||||
|
binds = scan.binds,
|
||||||
|
errors = findings.errors,
|
||||||
|
warnings = findings.warnings,
|
||||||
|
info = findings.info,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- M.run — orchestrator entry
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
local M = {} ---@type AnnotationPass
|
||||||
|
|
||||||
|
-- Expose `validate` for downstream passes (e.g. report.lua) that need to re-render the per-source results into a per-MODULE report.
|
||||||
|
M.validate = validate
|
||||||
|
|
||||||
|
--- @param ctx PassCtx
|
||||||
|
--- @return PassResult
|
||||||
|
function M.run(ctx)
|
||||||
|
local outputs = {} ---@type PassOutputEntry[]
|
||||||
|
local errors = {} ---@type Finding[]
|
||||||
|
local warnings = {} ---@type Finding[]
|
||||||
|
|
||||||
|
-- Build the shared pipe_ctx once for this run; every validate() call sees the same cross-source registries.
|
||||||
|
-- The corpus owns the canonical cross-source registries; per-source scans retain body / declaration ownership.
|
||||||
|
local corpus_pipe_ctx = build_corpus_pipe_ctx(ctx) ---@type PassScratch
|
||||||
|
local corpus = ctx.shared.corpus ---@type Corpus
|
||||||
|
|
||||||
|
-- Group `corpus.sources_by_dir` by module, validate every source in each bucket, and emit one errors.h per directory.
|
||||||
|
local by_dir = (corpus and corpus.sources_by_dir) or {} ---@type table<string, SourceFile[]>
|
||||||
|
|
||||||
|
for dir, dir_sources in pairs(by_dir) do ---@type string, SourceFile[]
|
||||||
|
local dir_basename = dir:match("([^/\\]+)$") or dir ---@type string
|
||||||
|
local dir_atoms = 0 ---@type integer
|
||||||
|
local dir_errors = {} ---@type Finding[]
|
||||||
|
local dir_warnings = {} ---@type Finding[]
|
||||||
|
for _, src in ipairs(dir_sources) do ---@type integer, SourceFile
|
||||||
|
local result = validate(ctx, src, corpus_pipe_ctx) ---@type AnnotatedResult
|
||||||
|
result.source = src.path -- tag for downstream rendering
|
||||||
|
dir_atoms = dir_atoms + #result.atoms
|
||||||
|
for _, e in ipairs(result.errors) do ---@type integer, Finding
|
||||||
|
dir_errors[#dir_errors + 1] = { line = e.line, msg = e.msg, source = src.path }
|
||||||
|
errors [#errors + 1] = { line = e.line, msg = e.msg }
|
||||||
|
end
|
||||||
|
for _, w in ipairs(result.warnings) do ---@type integer, Finding
|
||||||
|
dir_warnings[#dir_warnings + 1] = { line = w.line, msg = w.msg }
|
||||||
|
warnings [#warnings + 1] = { line = w.line, msg = w.msg }
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return { outputs = outputs, errors = errors, warnings = warnings }
|
||||||
|
end
|
||||||
|
|
||||||
|
return M
|
||||||
@@ -0,0 +1,626 @@
|
|||||||
|
--- passes/atoms_source_map.lua — Per-.word source-line map emitter for tape atoms.
|
||||||
|
---
|
||||||
|
--- Writer: this pass, given `atom.paths` (the per-atom mutable surface owned by `emission_model`). Readers:
|
||||||
|
--- `passes/dwarf_injection.lua` (synthesizes DW_TAG_inlined_subroutine + per-word line program rows) and
|
||||||
|
--- the gdb-runtime wrapper at `scripts/gdb/gdb_tape_atoms.gdb` (loads the source map via `source <path>`).
|
||||||
|
---
|
||||||
|
--- Inputs from `atom.paths`: the ordered `items` stream, dense `word_events`, `invocations` views. Outputs:
|
||||||
|
--- one `WORD N LINE L TEXT T` line per emitted `.word`, plus the per-word provenance form that DWARF synthesis consumes.
|
||||||
|
---
|
||||||
|
--- Two output forms:
|
||||||
|
--- 1. Markdown form: Handled by `passes/report.lua` (writes `<module>.atoms.md`).
|
||||||
|
--- The render functions `render_source_map` + `render_provenance` are exported for `report.lua` to call directly.
|
||||||
|
--- Compile artifacts (`*.macs.h`, `*.offsets.h`) stay in `<source_dir>/gen/`.
|
||||||
|
--- 2. `gdb_tape_atoms_runtime.gdb`: Post-link opt-in (`ctx.flags.gdb_runtime`),
|
||||||
|
--- so the gdb wrapper script + the generated runtime script share the same canonical location.
|
||||||
|
--- Triggered by `--post-link` or `--gdb-runtime`.
|
||||||
|
---
|
||||||
|
--- Output forma (sourcemap.txt form):
|
||||||
|
--- ```
|
||||||
|
--- # FORMAT_VERSION 1
|
||||||
|
--- # auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT
|
||||||
|
--- ATOM <name> "<abs-source-path>" <total_words>
|
||||||
|
--- WORD 0 LINE 49 TEXT load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||||
|
--- WORD 1 LINE 49 TEXT load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||||
|
--- ... (one WORD line per .word emitted by the atom body) ...
|
||||||
|
--- ENDATOM
|
||||||
|
--- ATOM <next-name> "<abs-source-path>" <total_words>
|
||||||
|
--- ...
|
||||||
|
--- ENDATOM
|
||||||
|
--- ```
|
||||||
|
--- Marker records are zero-width in `atom.paths.items`, so they emit no WORD rows in the dense word view.
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Module-scope requires + package.path setup
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source`
|
||||||
|
-- (works both standalone + when require'd). `duffle_paths.lua` sets package.path then returns `require("duffle")`
|
||||||
|
-- at the bottom, so the dofile value IS the duffle module.
|
||||||
|
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" ---@type string
|
||||||
|
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") ---@type DuffleExport
|
||||||
|
local elf_dwarf = require("elf_dwarf") ---@type ElfDwarfMod
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Constants
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
-- Format version emitted as the first line. Bump + add a migration test if the format changes;
|
||||||
|
-- the gdb runtime loader rejects mismatches (E2).
|
||||||
|
local FORMAT_VERSION = 1 ---@type integer
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Type declarations
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- @class AtomSourceMapCtx
|
||||||
|
--- @field shared PassShared
|
||||||
|
--- @field out_root string
|
||||||
|
--- @field flags PassFlags
|
||||||
|
--- @field project_root string|nil
|
||||||
|
|
||||||
|
--- @class WordMapEntry
|
||||||
|
--- @field pos integer
|
||||||
|
--- @field line integer
|
||||||
|
--- @field text string
|
||||||
|
--- @field body_line integer
|
||||||
|
--- @field gpr_keys string[]|nil
|
||||||
|
--- @field invocation InvocationRecord|nil
|
||||||
|
|
||||||
|
--- @class NmAddr
|
||||||
|
--- @field [1] integer -- st_value
|
||||||
|
--- @field [2] integer -- st_size
|
||||||
|
|
||||||
|
--- @class GdbAtomRecord
|
||||||
|
--- @field idx integer|nil
|
||||||
|
--- @field name string
|
||||||
|
--- @field src_path string
|
||||||
|
--- @field file_base string
|
||||||
|
--- @field addr integer
|
||||||
|
--- @field size_bytes integer
|
||||||
|
--- @field words integer
|
||||||
|
--- @field entries WordMapEntry[]
|
||||||
|
|
||||||
|
--- @class ElfDwarfMod
|
||||||
|
--- @field read_nm fun(elf_path: Path): table<string, NmAddr>
|
||||||
|
|
||||||
|
--- @class AtomSourceMapPass
|
||||||
|
--- @field render_source_map fun(src: SourceFile): string
|
||||||
|
--- @field render_provenance fun(src: SourceFile, wc: WordCounts): string
|
||||||
|
--- @field render_atom_source_map fun(atom: AtomEntry): string
|
||||||
|
--- @field render_atom_provenance fun(atom: AtomEntry, wc: WordCounts, rel_path: string): string
|
||||||
|
--- @field run fun(ctx: PassCtx): PassResult
|
||||||
|
|
||||||
|
--- @class AtomEntry
|
||||||
|
--- @field paths AtomPaths|nil
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Atom-path renderers
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Join word boundaries (from `items`) to per-word call text + source lines (from `word_events`).
|
||||||
|
--- @param atom AtomEntry
|
||||||
|
--- @return WordMapEntry[]
|
||||||
|
--- @return integer
|
||||||
|
local function canonical_word_entries(atom)
|
||||||
|
local paths = atom.paths or {} ---@type AtomPaths
|
||||||
|
local events = paths.word_events or {} ---@type WordEvent[]
|
||||||
|
local word_items = {} ---@type EmissionItem[]
|
||||||
|
for _, item in ipairs(paths.items or {}) do ---@type integer, EmissionItem
|
||||||
|
if item.kind == "word" then word_items[#word_items + 1] = item end
|
||||||
|
end
|
||||||
|
|
||||||
|
local entries = {} ---@type WordMapEntry[]
|
||||||
|
for index, event in ipairs(events) do ---@type integer, WordEvent
|
||||||
|
local item = word_items[index] or {} ---@type EmissionItem
|
||||||
|
entries[#entries + 1] = {
|
||||||
|
pos = event.i or (index - 1),
|
||||||
|
line = event.call_line or item.line or 0,
|
||||||
|
text = event.call_text or item.call_text or "",
|
||||||
|
body_line = event.body_line or item.body_line or item.line or 0,
|
||||||
|
gpr_keys = event.gpr_keys,
|
||||||
|
invocation = (event.outermost_invocation_id
|
||||||
|
and paths.invocations
|
||||||
|
and paths.invocations[event.outermost_invocation_id]) or nil,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
return entries, #events
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Render one atom's provenance stanza. Format 1 line shapes:
|
||||||
|
--- `WORD N CALL <src-path>:<src-line> MACRO <name> "<def-path>:<def-line>" BODY <line>` (component invocation)
|
||||||
|
--- `WORD N CALL <src-path>:<src-line> RAW` (raw `.word` outside any mac_* component)
|
||||||
|
--- Component identity comes from the outermost invocation record; the count-table lookup confirms the component was declared in `corpus.word_counts`
|
||||||
|
--- (populated by word_count_eval + components passes).
|
||||||
|
--- @param src SourceFile
|
||||||
|
--- @param atom AtomEntry
|
||||||
|
--- @param wc WordCounts
|
||||||
|
--- @return string[]
|
||||||
|
--- @return integer
|
||||||
|
local function emit_provenance_stanza(src, atom, wc)
|
||||||
|
local lines = {} ---@type string[]
|
||||||
|
local rel_path = src.path:gsub("\\\\", "/") ---@type string
|
||||||
|
local entries, total = canonical_word_entries(atom) ---@type WordMapEntry[], integer
|
||||||
|
lines[#lines + 1] = string.format('ATOM %s "%s" 0', atom.raw_name or atom.name, rel_path)
|
||||||
|
|
||||||
|
for _, entry in ipairs(entries) do ---@type integer, WordMapEntry
|
||||||
|
local inv = entry.invocation ---@type InvocationRecord|nil
|
||||||
|
local macro_count = inv and wc["mac_" .. inv.component_name] ---@type integer|nil
|
||||||
|
if inv and macro_count ~= nil then
|
||||||
|
lines[#lines + 1] = string.format('WORD %d CALL %s:%d MACRO %s "%s:%d" BODY %d'
|
||||||
|
, entry.pos, rel_path, entry.line, inv.component_name
|
||||||
|
, inv.def_path or "", inv.def_line or 0, entry.body_line)
|
||||||
|
else
|
||||||
|
lines[#lines + 1] = string.format("WORD %d CALL %s:%d RAW", entry.pos, rel_path, entry.line)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
lines[1] = lines[1]:gsub(" 0$", " " .. tostring(total))
|
||||||
|
lines[#lines + 1] = "ENDATOM"
|
||||||
|
return lines, total
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Render the full provenance file content for one source.
|
||||||
|
--- @param src SourceFile
|
||||||
|
--- @param wc WordCounts
|
||||||
|
--- @return string
|
||||||
|
local function render_provenance(src, wc)
|
||||||
|
local lines = {} ---@type string[]
|
||||||
|
lines[#lines + 1] = "# FORMAT_VERSION 1"
|
||||||
|
lines[#lines + 1] = "# auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT"
|
||||||
|
lines[#lines + 1] = "# Per-.word provenance: maps each emitted .word to its call site (atom body"
|
||||||
|
lines[#lines + 1] = "# file:line) and, when the word was emitted by a `mac_X(...)` component invocation,"
|
||||||
|
lines[#lines + 1] = "# the component's definition file:line + the per-word BODY line. Used by"
|
||||||
|
lines[#lines + 1] = "# dwarf_injection to synthesize DW_TAG_inlined_subroutine instances + per-word"
|
||||||
|
lines[#lines + 1] = "# line program rows for native source-level step into component bodies."
|
||||||
|
|
||||||
|
--- @param atom AtomEntry
|
||||||
|
--- @return nil
|
||||||
|
local function append(atom)
|
||||||
|
local stanza = emit_provenance_stanza(src, atom, wc) ---@type string[]
|
||||||
|
for _, line in ipairs(stanza) do lines[#lines + 1] = line end ---@type integer, string
|
||||||
|
end
|
||||||
|
for _, atom in ipairs(src.scan.atoms or {}) do ---@type integer, AtomEntry
|
||||||
|
if atom.paths then append(atom) end
|
||||||
|
end
|
||||||
|
for _, atom in ipairs(src.scan.raw_atoms or {}) do ---@type integer, AtomEntry
|
||||||
|
if atom.paths then append(atom) end
|
||||||
|
end
|
||||||
|
|
||||||
|
return table.concat(lines, "\n") .. "\n"
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Render one atom's stanza for the sourcemap.txt form (ATOM header line, N WORD lines, ENDATOM marker).
|
||||||
|
--- Returns (lines, total_words).
|
||||||
|
--- @param src SourceFile
|
||||||
|
--- @param atom AtomEntry
|
||||||
|
--- @return string[]
|
||||||
|
--- @return integer
|
||||||
|
local function emit_atom_stanza(src, atom)
|
||||||
|
local lines = {} ---@type string[]
|
||||||
|
local rel_path = src.path:gsub("\\\\", "/") ---@type string
|
||||||
|
local entries, total = canonical_word_entries(atom) ---@type WordMapEntry[], integer
|
||||||
|
|
||||||
|
lines[#lines + 1] = string.format('ATOM %s "%s" 0', atom.raw_name or atom.name, rel_path)
|
||||||
|
for _, entry in ipairs(entries) do ---@type integer, WordMapEntry
|
||||||
|
lines[#lines + 1] = string.format("WORD %d LINE %d TEXT %s",
|
||||||
|
entry.pos, entry.line, entry.text)
|
||||||
|
end
|
||||||
|
lines[1] = lines[1]:gsub(" 0$", " " .. tostring(total))
|
||||||
|
lines[#lines + 1] = "ENDATOM"
|
||||||
|
return lines, total
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Render the full source map file content for one source (one .atoms.sourcemap.txt per source). Mirrors offsets.lua's
|
||||||
|
--- `project_atoms` shape: scan.atoms + scan.raw_atoms, no kind filter.
|
||||||
|
--- @param src SourceFile
|
||||||
|
--- @return string
|
||||||
|
local function render_source_map(src)
|
||||||
|
local lines = {} ---@type string[]
|
||||||
|
lines[#lines + 1] = "# FORMAT_VERSION " .. FORMAT_VERSION
|
||||||
|
lines[#lines + 1] = "# auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT"
|
||||||
|
|
||||||
|
--- @param atom AtomEntry
|
||||||
|
--- @return nil
|
||||||
|
local function append(atom)
|
||||||
|
local stanza = emit_atom_stanza(src, atom) ---@type string[]
|
||||||
|
for _, line in ipairs(stanza) do lines[#lines + 1] = line end ---@type integer, string
|
||||||
|
end
|
||||||
|
for _, atom in ipairs(src.scan.atoms or {}) do ---@type integer, AtomEntry
|
||||||
|
if atom.paths then append(atom) end
|
||||||
|
end
|
||||||
|
for _, atom in ipairs(src.scan.raw_atoms or {}) do ---@type integer, AtomEntry
|
||||||
|
if atom.paths then append(atom) end
|
||||||
|
end
|
||||||
|
|
||||||
|
return table.concat(lines, "\n") .. "\n"
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- gdb-runtime emission (post-link, addresses via nm)
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Escape a string for embedding in a gdb `set $var = "..."` literal.
|
||||||
|
--- gdb uses C-style escaping; we escape `\` and `"` (newlines were flattened earlier).
|
||||||
|
--- @param s string
|
||||||
|
--- @return string
|
||||||
|
local function gdb_escape(s)
|
||||||
|
return (s:gsub("\\", "\\\\"):gsub('"', '\\"'))
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Build the list of atoms with addresses + word entries. Shared helper for the gdb-runtime file emission.
|
||||||
|
--- @param ctx PassCtx
|
||||||
|
--- @return GdbAtomRecord[]
|
||||||
|
local function build_atom_table(ctx)
|
||||||
|
local addrs = elf_dwarf.read_nm(ctx.flags.elf_path) ---@type table<string, NmAddr>
|
||||||
|
local corpus = ctx.shared and ctx.shared.corpus ---@type Corpus|nil
|
||||||
|
local matched = {} ---@type GdbAtomRecord[]
|
||||||
|
|
||||||
|
for _, src in ipairs(corpus.source_order or {}) do ---@type integer, SourceFile
|
||||||
|
local file_base = src.path:match("([^/\\\\]+)$") or src.path ---@type string
|
||||||
|
--- @param atom AtomEntry
|
||||||
|
--- @return nil
|
||||||
|
local function append(atom)
|
||||||
|
if not atom.paths then return end
|
||||||
|
local name = atom.raw_name or atom.name ---@type string
|
||||||
|
local info = addrs[name] ---@type NmAddr|nil
|
||||||
|
if not info then return end
|
||||||
|
local entries, total = canonical_word_entries(atom) ---@type WordMapEntry[], integer
|
||||||
|
matched[#matched + 1] = {
|
||||||
|
name = name,
|
||||||
|
src_path = src.path,
|
||||||
|
file_base = file_base,
|
||||||
|
addr = info[1],
|
||||||
|
size_bytes = info[2],
|
||||||
|
words = total,
|
||||||
|
entries = entries,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
for _, atom in ipairs((src.scan or {}).atoms or {}) do append(atom) end ---@type integer, AtomEntry
|
||||||
|
for _, atom in ipairs((src.scan or {}).raw_atoms or {}) do append(atom) end ---@type integer, AtomEntry
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Deterministic order: sort by address (matches `nm` output ordering).
|
||||||
|
--- @param a GdbAtomRecord
|
||||||
|
--- @param b GdbAtomRecord
|
||||||
|
--- @return boolean
|
||||||
|
table.sort(matched, function(a, b) return a.addr < b.addr end)
|
||||||
|
for i, a in ipairs(matched) do a.idx = i - 1 end ---@type integer, GdbAtomRecord
|
||||||
|
return matched
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Append the 9 gdb command definitions to `lines`. Pure gdb scripting — addresses come from `nm`,
|
||||||
|
--- the convenience vars set in `emit_gdb_runtime` provide printf args, and
|
||||||
|
--- each command is a static sequence of `printf` / `tbreak` / `if ... end` blocks.
|
||||||
|
--- The Lua pass emits N atoms' worth of lines; runtime iteration is gdb's job.
|
||||||
|
---
|
||||||
|
--- Why hardcoded per-atom: gdb's `$` substitution doesn't concat inside var names — `$__atom_name_$__i` in a `while`
|
||||||
|
--- loop resolves to one literal identifier, not `name_i`. Compile-time emission is the only path.
|
||||||
|
--- @param lines string[]
|
||||||
|
--- @param matched GdbAtomRecord[]
|
||||||
|
--- @return nil
|
||||||
|
local function append_gdb_commands(lines, matched)
|
||||||
|
-- ── tape_atoms ──
|
||||||
|
-- Hardcoded one printf per atom. No loop.
|
||||||
|
lines[#lines + 1] = "define tape_atoms"
|
||||||
|
for _, a in ipairs(matched) do ---@type integer, GdbAtomRecord
|
||||||
|
-- gdb 12.1 quirk: literals in printf args require an attached target.
|
||||||
|
-- Use the per-atom convenience vars set above as printf args.
|
||||||
|
lines[#lines + 1] = string.format(' printf " %%-32s @ 0x%%08x %%4d words\\n", $__atom_name_%d, $__atom_addr_%d, $__atom_words_%d',
|
||||||
|
a.idx, a.idx, a.idx)
|
||||||
|
end
|
||||||
|
lines[#lines + 1] = "end"
|
||||||
|
lines[#lines + 1] = "document tape_atoms"
|
||||||
|
lines[#lines + 1] = " List every tape atom symbol in the loaded ELF with .rodata addr + word count."
|
||||||
|
lines[#lines + 1] = "end"
|
||||||
|
lines[#lines + 1] = ""
|
||||||
|
|
||||||
|
-- ── break_atom (generic) + per-atom break_atom_X ──
|
||||||
|
lines[#lines + 1] = "define break_atom"
|
||||||
|
lines[#lines + 1] = ' echo "Usage: break_atom_<exact_name> (pick from the list below)"'
|
||||||
|
for _, a in ipairs(matched) do ---@type integer, GdbAtomRecord
|
||||||
|
lines[#lines + 1] = string.format(' printf " break_atom_%%-32s\\n", $__atom_name_%d', a.idx)
|
||||||
|
end
|
||||||
|
lines[#lines + 1] = "end"
|
||||||
|
lines[#lines + 1] = "document break_atom"
|
||||||
|
lines[#lines + 1] = " Generic help: lists the per-atom break_atom_<name> commands."
|
||||||
|
lines[#lines + 1] = "end"
|
||||||
|
lines[#lines + 1] = ""
|
||||||
|
|
||||||
|
for _, a in ipairs(matched) do ---@type integer, GdbAtomRecord
|
||||||
|
lines[#lines + 1] = string.format("define break_atom_%s", a.name)
|
||||||
|
lines[#lines + 1] = string.format(" break *$__atom_addr_%d", a.idx)
|
||||||
|
lines[#lines + 1] = string.format(' printf " Breakpoint set at %s (0x%%08x)\\n", $__atom_addr_%d', a.name, a.idx)
|
||||||
|
lines[#lines + 1] = "end"
|
||||||
|
lines[#lines + 1] = string.format("document break_atom_%s", a.name)
|
||||||
|
lines[#lines + 1] = string.format(" Set a breakpoint at %s.", a.name)
|
||||||
|
lines[#lines + 1] = "end"
|
||||||
|
lines[#lines + 1] = ""
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ── step_atom / next_atom ──
|
||||||
|
-- Hardcoded one tbreak per atom. No loop.
|
||||||
|
lines[#lines + 1] = "define step_atom"
|
||||||
|
for _, a in ipairs(matched) do ---@type integer, GdbAtomRecord
|
||||||
|
lines[#lines + 1] = string.format(" tbreak *$__atom_addr_%d", a.idx)
|
||||||
|
end
|
||||||
|
lines[#lines + 1] = " continue"
|
||||||
|
lines[#lines + 1] = "end"
|
||||||
|
lines[#lines + 1] = "document step_atom"
|
||||||
|
lines[#lines + 1] = " Set one-shot BPs at every atom + continue. Stops at the next atom boundary."
|
||||||
|
lines[#lines + 1] = "end"
|
||||||
|
lines[#lines + 1] = ""
|
||||||
|
|
||||||
|
lines[#lines + 1] = "define next_atom"
|
||||||
|
lines[#lines + 1] = " step_atom"
|
||||||
|
lines[#lines + 1] = "end"
|
||||||
|
lines[#lines + 1] = "document next_atom"
|
||||||
|
lines[#lines + 1] = " Alias for step_atom."
|
||||||
|
lines[#lines + 1] = "end"
|
||||||
|
lines[#lines + 1] = ""
|
||||||
|
|
||||||
|
-- ── where_in_atom ──
|
||||||
|
-- Hardcoded one outer-if per atom; inside, one inner-if per WORD entry.
|
||||||
|
lines[#lines + 1] = "define where_in_atom"
|
||||||
|
lines[#lines + 1] = " set $__pc = (unsigned int)$pc"
|
||||||
|
lines[#lines + 1] = " set $__matched = 0"
|
||||||
|
for _, a in ipairs(matched) do ---@type integer, GdbAtomRecord
|
||||||
|
-- Precompute end_addr (gdb 12.1's expression evaluator chokes on `addr + words*4`).
|
||||||
|
lines[#lines + 1] = string.format(" set $__end_%d = $__atom_addr_%d + $__atom_words_%d * 4", a.idx, a.idx, a.idx)
|
||||||
|
lines[#lines + 1] = string.format(" if $__pc >= $__atom_addr_%d && $__pc < $__end_%d", a.idx, a.idx)
|
||||||
|
lines[#lines + 1] = string.format(' printf "atom: %%s\\n", $__atom_name_%d', a.idx)
|
||||||
|
lines[#lines + 1] = ' printf "addr: 0x%08x\\n", $__pc'
|
||||||
|
lines[#lines + 1] = string.format(" set $__word = ($__pc - $__atom_addr_%d) / 4", a.idx)
|
||||||
|
lines[#lines + 1] = string.format(' printf "word: %%d/%%d\\n", $__word, $__atom_words_%d', a.idx)
|
||||||
|
-- One inner-if per WORD entry. Each word's line + text hardcoded.
|
||||||
|
for _, we in ipairs(a.entries) do ---@type integer, WordMapEntry
|
||||||
|
lines[#lines + 1] = string.format(" if $__word == %d", we.pos)
|
||||||
|
-- Escape TEXT for printf format string.
|
||||||
|
local escaped_text = we.text:gsub("%%", "%%%%"):gsub('"', '\\"') ---@type string
|
||||||
|
lines[#lines + 1] = string.format(' printf "source: %%s:%%d %%s\\n", $__atom_file_%d, %d, "%s"', a.idx, we.line, escaped_text)
|
||||||
|
lines[#lines + 1] = " end"
|
||||||
|
end
|
||||||
|
-- Fallback for words beyond the source map (shouldn't happen if nm matches).
|
||||||
|
local max_word = 0 ---@type integer
|
||||||
|
if #a.entries > 0 then max_word = a.entries[#a.entries].pos end
|
||||||
|
lines[#lines + 1] = string.format(' if $__word > %d', max_word)
|
||||||
|
lines[#lines + 1] = ' printf "source: (no source-map entry for word %%d; map may be stale)\\n", $__word'
|
||||||
|
lines[#lines + 1] = " end"
|
||||||
|
lines[#lines + 1] = " set $__matched = 1"
|
||||||
|
lines[#lines + 1] = " end"
|
||||||
|
end
|
||||||
|
lines[#lines + 1] = " if !$__matched"
|
||||||
|
lines[#lines + 1] = ' echo PC is not inside any known atom (in .text or unmapped region).'
|
||||||
|
lines[#lines + 1] = " end"
|
||||||
|
lines[#lines + 1] = "end"
|
||||||
|
lines[#lines + 1] = "document where_in_atom"
|
||||||
|
lines[#lines + 1] = " Report current atom name, .rodata addr, word offset, and source line."
|
||||||
|
lines[#lines + 1] = "end"
|
||||||
|
lines[#lines + 1] = ""
|
||||||
|
|
||||||
|
-- ── stepi_inside_atom ──
|
||||||
|
-- Hardcoded one if-containment-check per atom (no loop).
|
||||||
|
-- Precompute end_addr in Lua so we don't ask gdb to evaluate `addr + words*4` inside the if condition
|
||||||
|
-- (gdb 12.1's expression evaluator chokes on the `*` and emits a misleading 'function malloc' error in some gdb builds).
|
||||||
|
lines[#lines + 1] = "define stepi_inside_atom"
|
||||||
|
lines[#lines + 1] = " set $__in_atom = 0"
|
||||||
|
lines[#lines + 1] = " set $__did_step = 0"
|
||||||
|
lines[#lines + 1] = " set $__pc = (unsigned int)$pc"
|
||||||
|
for _, a in ipairs(matched) do ---@type integer, GdbAtomRecord
|
||||||
|
-- Precompute end_addr in the convenience var (single expression gdb handles).
|
||||||
|
lines[#lines + 1] = string.format(" set $__end_%d = $__atom_addr_%d + $__atom_words_%d * 4", a.idx, a.idx, a.idx)
|
||||||
|
lines[#lines + 1] = string.format(" if $__pc >= $__atom_addr_%d && $__pc < $__end_%d", a.idx, a.idx)
|
||||||
|
lines[#lines + 1] = " set $__in_atom = 1"
|
||||||
|
lines[#lines + 1] = " stepi"
|
||||||
|
lines[#lines + 1] = " set $__did_step = 1"
|
||||||
|
lines[#lines + 1] = " end"
|
||||||
|
end
|
||||||
|
lines[#lines + 1] = " if !$__did_step"
|
||||||
|
lines[#lines + 1] = ' echo [gdb_tape_atoms] stepi_inside_atom: PC is not inside any atom; refusing to step.'
|
||||||
|
lines[#lines + 1] = " end"
|
||||||
|
lines[#lines + 1] = " where_in_atom"
|
||||||
|
lines[#lines + 1] = "end"
|
||||||
|
lines[#lines + 1] = "document stepi_inside_atom"
|
||||||
|
lines[#lines + 1] = " One MIPS-instruction step, then where_in_atom. The step-and-see-source-line workflow."
|
||||||
|
lines[#lines + 1] = "end"
|
||||||
|
lines[#lines + 1] = ""
|
||||||
|
|
||||||
|
-- ── wave_ctx ──
|
||||||
|
lines[#lines + 1] = "define wave_ctx"
|
||||||
|
lines[#lines + 1] = ' printf "$t4 = R_FaceCursor 0x%08x\\n", $t4'
|
||||||
|
lines[#lines + 1] = ' printf "$t5 = R_VertBase 0x%08x\\n", $t5'
|
||||||
|
lines[#lines + 1] = ' printf "$t6 = R_OtBase 0x%08x\\n", $t6'
|
||||||
|
lines[#lines + 1] = ' printf "$t7 = R_PrimCursor 0x%08x\\n", $t7'
|
||||||
|
lines[#lines + 1] = "end"
|
||||||
|
lines[#lines + 1] = "document wave_ctx"
|
||||||
|
lines[#lines + 1] = " Pretty-print the 4 wave-context GPRs ($t4=R_FaceCursor, $t5=R_VertBase, $t6=R_OtBase, $t7=R_PrimCursor). Requires target attached."
|
||||||
|
lines[#lines + 1] = "end"
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Emit the gdb-runtime file (post-link). Pure gdb scripting — addresses come from `mipsel-none-elf-nm -S`, get embedded
|
||||||
|
--- in `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`, and load via `set $var = ...` + `define ... end` blocks at gdb source-time.
|
||||||
|
--- @param ctx PassCtx
|
||||||
|
--- @return nil
|
||||||
|
local function emit_gdb_runtime(ctx)
|
||||||
|
if not (ctx.flags and ctx.flags.gdb_runtime) then return end
|
||||||
|
local elf_path = ctx.flags.elf_path ---@type string|nil
|
||||||
|
if not elf_path or elf_path == "" then
|
||||||
|
io.stderr:write("[atoms_source_map] --gdb-runtime requires --elf <elf>\n")
|
||||||
|
return
|
||||||
|
end
|
||||||
|
if lfs.attributes(elf_path, "mode") ~= "file" then
|
||||||
|
io.stderr:write(string.format(
|
||||||
|
"[atoms_source_map] --gdb-runtime: ELF not found at %s\n", elf_path))
|
||||||
|
return
|
||||||
|
end
|
||||||
|
|
||||||
|
local matched = build_atom_table(ctx) ---@type GdbAtomRecord[]
|
||||||
|
if #matched == 0 then
|
||||||
|
io.stderr:write("[atoms_source_map] --gdb-runtime: no atoms matched against nm symbols (stale scan?).\n")
|
||||||
|
return
|
||||||
|
end
|
||||||
|
|
||||||
|
local lines = {} ---@type string[]
|
||||||
|
lines[#lines + 1] = "# Auto-generated by ps1_meta.lua (passes/atoms_source_map.lua)"
|
||||||
|
lines[#lines + 1] = "# DO NOT EDIT — re-run ps1_meta.lua --atoms-source-map --gdb-runtime to regenerate"
|
||||||
|
lines[#lines + 1] = "# Sourced by scripts/gdb/gdb_tape_atoms.gdb (the wrapper)."
|
||||||
|
lines[#lines + 1] = "# Pure gdb scripting — no Python, no Tcl, no Guile required."
|
||||||
|
lines[#lines + 1] = "# Commands are FULLY HARDCODED per-atom because gdb doesn't do nested"
|
||||||
|
lines[#lines + 1] = "# `$` substitution in var names (`$foo_$i` is one literal identifier)."
|
||||||
|
lines[#lines + 1] = "# Per-atom convenience vars ($__atom_name_<i> etc.) are set so gdb's"
|
||||||
|
lines[#lines + 1] = "# `printf` has valid expression args (gdb 12.1 quirks: literals in"
|
||||||
|
lines[#lines + 1] = "# printf args require an attached target; convenience-var args do not)."
|
||||||
|
lines[#lines + 1] = string.format("# %d atoms from ELF: %s", #matched, elf_path)
|
||||||
|
lines[#lines + 1] = ""
|
||||||
|
|
||||||
|
-- Format version + count + ELF path (the latter is referenced by the load-line).
|
||||||
|
lines[#lines + 1] = "set $__atom_format_version = " .. FORMAT_VERSION
|
||||||
|
lines[#lines + 1] = string.format("set $__atom_count = %d", #matched)
|
||||||
|
lines[#lines + 1] = string.format('set $__elf_path = "%s"', gdb_escape(elf_path))
|
||||||
|
lines[#lines + 1] = ""
|
||||||
|
|
||||||
|
-- Per-atom convenience vars (used as printf args; literals aren't accepted
|
||||||
|
-- without an attached target on gdb 12.1).
|
||||||
|
for _, a in ipairs(matched) do ---@type integer, GdbAtomRecord
|
||||||
|
lines[#lines + 1] = string.format('set $__atom_name_%d = "%s"', a.idx, gdb_escape(a.name))
|
||||||
|
lines[#lines + 1] = string.format("set $__atom_addr_%d = 0x%x", a.idx, a.addr)
|
||||||
|
lines[#lines + 1] = string.format("set $__atom_words_%d = %d", a.idx, a.words)
|
||||||
|
lines[#lines + 1] = string.format('set $__atom_file_%d = "%s"', a.idx, gdb_escape(a.file_base))
|
||||||
|
end
|
||||||
|
lines[#lines + 1] = ""
|
||||||
|
|
||||||
|
-- The 9 commands (each `define ... end` overrides the wrapper's stub).
|
||||||
|
lines[#lines + 1] = "# ── 9 user commands (overrides wrapper stubs) ──"
|
||||||
|
append_gdb_commands(lines, matched)
|
||||||
|
lines[#lines + 1] = ""
|
||||||
|
|
||||||
|
-- Confirmation line for the source operator.
|
||||||
|
lines[#lines + 1] = 'printf "[gdb_tape_atoms] runtime loaded %d atoms from %s\\n", $__atom_count, $__elf_path'
|
||||||
|
|
||||||
|
local out_path ---@type string
|
||||||
|
-- Move out of `<out_root>/gdb_tape_atoms_runtime.gdb` to `<out_root>/../gdb_tape_atoms_runtime.gdb` when the conventional `<out_root>` is `<build>/gen`
|
||||||
|
-- (any equivalent spelling — relative, absolute backslash, absolute forward-slash, trailing-separator variants).
|
||||||
|
-- This puts the gdb runtime alongside the ELF at `build/` rather than under the report subdir.
|
||||||
|
--- @param p string
|
||||||
|
--- @return boolean
|
||||||
|
local function ends_with_gen_dir(p)
|
||||||
|
if type(p) ~= "string" then return false end
|
||||||
|
return p:match("[/\\]gen[/\\]?$") ~= nil or p == "build/gen" or p == "build\\gen"
|
||||||
|
end
|
||||||
|
if ends_with_gen_dir(ctx.out_root) then
|
||||||
|
-- Strip the trailing `/gen` segment, then write the runtime script under `build/`.
|
||||||
|
-- e.g. "C:/projects/Pikuma/ps1/build/gen" -> "C:/projects/Pikuma/ps1/build".
|
||||||
|
local parent = ctx.out_root:gsub("[/\\]gen[/\\]?$", "") ---@type string
|
||||||
|
out_path = parent .. "/gdb_tape_atoms_runtime.gdb"
|
||||||
|
else
|
||||||
|
out_path = ctx.out_root .. "/gdb_tape_atoms_runtime.gdb"
|
||||||
|
end
|
||||||
|
duffle.ensure_dir(duffle.dirname(out_path))
|
||||||
|
duffle.write_file_lf(out_path, table.concat(lines, "\n") .. "\n")
|
||||||
|
-- io.stderr:write(string.format("[atoms_source_map] wrote %s (%d atoms)\n", out_path, #matched))
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- M — module exports
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
local M = {} ---@type AtomSourceMapPass
|
||||||
|
|
||||||
|
-- Expose the pure render functions so `report.lua` and the focused tests can call them directly without triggering the file-emit path.
|
||||||
|
M.render_source_map = render_source_map
|
||||||
|
M.render_provenance = render_provenance
|
||||||
|
|
||||||
|
--- Render ONE atom's sourcemap stanza.
|
||||||
|
--- @param atom AtomEntry
|
||||||
|
--- @return string
|
||||||
|
function M.render_atom_source_map(atom)
|
||||||
|
assert(type(atom) == "table", "render_atom_source_map: atom must be a table")
|
||||||
|
assert(type(atom.paths) == "table", "render_atom_source_map: atom.paths must be a table")
|
||||||
|
local entries, total = canonical_word_entries(atom) ---@type WordMapEntry[], integer
|
||||||
|
local lines = {} ---@type string[]
|
||||||
|
lines[#lines + 1] = string.format("ATOM %s %d", (atom.raw_name or atom.name), total)
|
||||||
|
for _, entry in ipairs(entries) do ---@type integer, WordMapEntry
|
||||||
|
local word_line = string.format("WORD %d LINE %d TEXT %s", ---@type string
|
||||||
|
entry.pos, entry.line, entry.text)
|
||||||
|
local keys = {} ---@type string[]
|
||||||
|
for pos = 1, 16 do ---@type integer
|
||||||
|
local k = entry.gpr_keys and entry.gpr_keys[pos] ---@type string|nil
|
||||||
|
if type(k) == "string" and k:sub(1, 7) == "reguse:" then
|
||||||
|
keys[#keys + 1] = k
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if #keys > 0 then
|
||||||
|
word_line = word_line .. " KEYS " .. table.concat(keys, ",")
|
||||||
|
end
|
||||||
|
lines[#lines + 1] = word_line
|
||||||
|
end
|
||||||
|
lines[#lines + 1] = "ENDATOM"
|
||||||
|
return table.concat(lines, "\n") .. "\n"
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Render ONE atom's provenance stanza — no per-file format header, no enumeration of other atoms.
|
||||||
|
---
|
||||||
|
--- `rel_path` is the source path (forward-slashes) embedded in every `CALL` line.
|
||||||
|
--- The .md caller (report.lua) is expected to derive this once per `## <source>` heading and pass it down for each atom in that source.
|
||||||
|
--- @param atom AtomEntry
|
||||||
|
--- @param wc WordCounts
|
||||||
|
--- @param rel_path string
|
||||||
|
--- @return string
|
||||||
|
function M.render_atom_provenance(atom, wc, rel_path)
|
||||||
|
assert(type(atom) == "table", "render_atom_provenance: atom must be a table")
|
||||||
|
assert(type(atom.paths) == "table", "render_atom_provenance: atom.paths must be a table")
|
||||||
|
assert(type(rel_path) == "string", "render_atom_provenance: rel_path must be a string")
|
||||||
|
local entries, total = canonical_word_entries(atom) ---@type WordMapEntry[], integer
|
||||||
|
local lines = {} ---@type string[]
|
||||||
|
lines[#lines + 1] = string.format("ATOM %s %d", (atom.raw_name or atom.name), total)
|
||||||
|
for _, entry in ipairs(entries) do ---@type integer, WordMapEntry
|
||||||
|
local inv = entry.invocation ---@type InvocationRecord|nil
|
||||||
|
local macro_count = inv and wc and wc["mac_" .. inv.component_name] ---@type integer|nil
|
||||||
|
if inv and macro_count ~= nil then
|
||||||
|
lines[#lines + 1] = string.format('WORD %d CALL %s:%d MACRO %s "%s:%d" BODY %d'
|
||||||
|
, entry.pos, rel_path, entry.line, inv.component_name, inv.def_path or "", inv.def_line or 0, entry.body_line)
|
||||||
|
else
|
||||||
|
lines[#lines + 1] = string.format(
|
||||||
|
"WORD %d CALL %s:%d RAW", entry.pos, rel_path, entry.line)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return table.concat(lines, "\n") .. "\n"
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Pass entry. For each source that declares at least one tape atom,
|
||||||
|
--- emit two files in `<out_root>/`: `<basename>.atoms.sourcemap.txt` (per-word call-site map) and `<basename>.atoms.provenance.txt`
|
||||||
|
--- (per-word definition + body line, resolved via the outermost `mac_X(...)` invocation).
|
||||||
|
--- When `ctx.flags.gdb_runtime` is true and `ctx.flags.elf_path` exists, also emit the post-link gdb script `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`.
|
||||||
|
--- @param ctx PassCtx
|
||||||
|
--- @return PassResult
|
||||||
|
function M.run(ctx)
|
||||||
|
local outputs = {} ---@type PassOutputEntry[]
|
||||||
|
local errors = {} ---@type Finding[]
|
||||||
|
local warnings = {} ---@type Finding[]
|
||||||
|
|
||||||
|
local corpus = ctx.shared and ctx.shared.corpus ---@type Corpus|nil
|
||||||
|
if type(corpus) ~= "table" or type(corpus.source_order) ~= "table" then
|
||||||
|
error("atoms_source_map.run requires ctx.shared.corpus.source_order (canonical corpus).", 0)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Word counts come from `corpus.word_counts` (populated by word_count_eval + components passes).
|
||||||
|
local wc = corpus.word_counts or {} ---@type WordCounts
|
||||||
|
if not next(wc) then
|
||||||
|
warnings[#warnings + 1] = {
|
||||||
|
line = 0,
|
||||||
|
msg = "atoms_source_map: corpus.word_counts is empty; the word-counts + components passes may not have populated it. Check the PASSES dep edges.",
|
||||||
|
}
|
||||||
|
end
|
||||||
|
|
||||||
|
-- atoms.sourcemap.txt + atoms.provenance.txt content moved to report.lua via `<module>.atoms.md` markdown file.
|
||||||
|
-- This pass emits only the post-link gdb_runtime artifact (see emit_gdb_runtime below).
|
||||||
|
|
||||||
|
-- Optionally emit the gdb-runtime form (post-link, one file per build).
|
||||||
|
if ctx.flags and ctx.flags.gdb_runtime then
|
||||||
|
emit_gdb_runtime(ctx)
|
||||||
|
end
|
||||||
|
|
||||||
|
return { outputs = outputs, errors = errors, warnings = warnings }
|
||||||
|
end
|
||||||
|
|
||||||
|
return M
|
||||||
@@ -0,0 +1,366 @@
|
|||||||
|
--- passes/auto_reg.lua — Per-phase automatic GPR allocator + gen/auto_reg.h emitter.
|
||||||
|
---
|
||||||
|
--- Reads the per-source + corpus-level `atom_auto_regs` + `phase_auto_regs` registries populated by `passes/scan_source.lua`.
|
||||||
|
--- Runs a deterministic first-fit allocator in the `R_T0..R_T7 + R_V0..R_V1` pool (10 physical GPRs).
|
||||||
|
--- Emits one `#define R_<Sym>_Code R_Tn_Code` per marker into per-directory `gen/auto_reg.h`.
|
||||||
|
---
|
||||||
|
--- User-pinned GPRs : The corpus's `register_alias_registry` is consulted to exclude GPRs the user has pinned via
|
||||||
|
--- `atom_reg` + `_Code` defs (e.g. carriers like `R_ResolveScratch = R_T4 atom_reg`).
|
||||||
|
--- These GPRs are unavailable to EVERY atom's source pool.
|
||||||
|
--- Carriers are preserved across atoms by context discipline and must never be reallocated.
|
||||||
|
--- Per-atom body parsing also catches alias references (R_<Alias>) and hardcoded R_Tn references,
|
||||||
|
--- so the user can write either `R_T4` or `R_ResolveScratch` in an atom body and the pass will exclude R_T4 from that atom's pool.
|
||||||
|
---
|
||||||
|
--- Conflict detection: If the user hardcodes `R_Tn` in an atom body that shares a phase with an auto-reg that picked `R_Tn`,
|
||||||
|
--- emit `phase_register_clash` as an info finding (no build stop).
|
||||||
|
--- Should be unreachable after the user-pinning + body-parsing fix above; kept as a defensive safety net.
|
||||||
|
---
|
||||||
|
--- Pool exhaustion: If a phase declares more `R_<Sym>` mappings than the 10-register pool can hold,
|
||||||
|
--- emit `phase_register_pool_exhausted` as a build-stopping error.
|
||||||
|
|
||||||
|
--- @alias GprIdent string
|
||||||
|
|
||||||
|
--- @class GprAllocMap
|
||||||
|
--- @field [string] GprIdent -- bag: auto-reg symbol -> physical GPR
|
||||||
|
|
||||||
|
--- @class AutoRegOutput
|
||||||
|
--- @field auto_reg_h string
|
||||||
|
|
||||||
|
--- @class AutoRegResult
|
||||||
|
--- @field outputs AutoRegOutput[]
|
||||||
|
--- @field errors Finding[]
|
||||||
|
--- @field warnings Finding[]
|
||||||
|
|
||||||
|
--- @class AutoRegPass
|
||||||
|
--- @field run fun(ctx: PassCtx): AutoRegResult
|
||||||
|
--- @field POOL GprIdent[]
|
||||||
|
|
||||||
|
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" ---@type string
|
||||||
|
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") ---@type DuffleExport
|
||||||
|
local isa = require("duffle_isa") ---@type DuffleIsa
|
||||||
|
|
||||||
|
--- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
--- THE GPR ALLOCATION POOL — what's allocatable, and (more importantly) WHY
|
||||||
|
--- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
---
|
||||||
|
--- The auto-reg pass picks physical GPRs for `atom_auto_reg(...)` / `phase_auto_reg(...)` markers.
|
||||||
|
--- The 24-register pool covers R2-R25 (the user/atom allocatable surface):
|
||||||
|
--- R_T0..R_T7, R_V0..R_V1, R_A0..A3, R_S0..S7, R_T8..T9.
|
||||||
|
--- Excluded (and never added to the pool):
|
||||||
|
--- R_0 (code 0) — Hardwired zero. Cannot be written.
|
||||||
|
--- R_AT (code 1) — Assembler temporary. Reserved by the MIPS O32 ABI.
|
||||||
|
--- R_A0..A3 — Explicitly omitted above even though their integer codes map to POOL entries;
|
||||||
|
--- the pool-construction loop below only references the POOL string literals, never the integer codes, so they are NOT auto-allocated by default.
|
||||||
|
--- (A0-A3 become available when the user adds them to POOL or hardcodes an R_A0 reference in the atom body.)
|
||||||
|
--- R_K0/K1 (codes 26-27) — Kernel / interrupt handler reserves. Never touched by user code.
|
||||||
|
--- R_GP/SP/FP/RA (codes 28-31) — R_SP/R_FP/R_RA are tape-runtime carriers between tape_enter and tape_exit; R_GP stays the host global pointer.
|
||||||
|
---
|
||||||
|
local POOL = {} ---@type GprIdent[]
|
||||||
|
for _, row in ipairs(isa.GPR_ROLE) do ---@type integer, GprRole
|
||||||
|
if row.pool then
|
||||||
|
POOL[#POOL + 1] = row.name
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Map from integer MIPS GPR code (the `code` field on AliasEntry) to the physical GPR ident in POOL.
|
||||||
|
-- The standard MIPS O32 ABI register numbering matches mips.h's R_*_Code #defines (mips.h).
|
||||||
|
-- Only the POOL entries matter for auto_reg — non-pool aliases
|
||||||
|
-- (R_AT=1, R_A0..A3=4..7, R_T8=24, R_T9=25, R_K0/K1=26..27, R_GP/SP/FP/RA=28..31)
|
||||||
|
-- are deliberately omitted — see the comment block above for the WHY of each exclusion.
|
||||||
|
local INT_CODE_TO_POOL_GPR = { ---@type table<integer, GprIdent> -- bag: MIPS GPR code -> POOL ident
|
||||||
|
[2] = "R_V0", [3] = "R_V1",
|
||||||
|
[4] = "R_A0", [5] = "R_A1", [6] = "R_A2", [7] = "R_A3",
|
||||||
|
[8] = "R_T0", [9] = "R_T1", [10] = "R_T2", [11] = "R_T3",
|
||||||
|
[12] = "R_T4", [13] = "R_T5", [14] = "R_T6", [15] = "R_T7",
|
||||||
|
[16] = "R_S0", [17] = "R_S1", [18] = "R_S2", [19] = "R_S3",
|
||||||
|
[20] = "R_S4", [21] = "R_S5", [22] = "R_S6", [23] = "R_S7",
|
||||||
|
[24] = "R_T8", [25] = "R_T9",
|
||||||
|
}
|
||||||
|
|
||||||
|
-- Stable sort for deterministic allocation order.
|
||||||
|
--- @param tbl table<string, string> -- bag: key set only; values unused
|
||||||
|
--- @return string[]
|
||||||
|
local function stable_sort_keys(tbl)
|
||||||
|
local keys = {} ---@type string[]
|
||||||
|
for k in pairs(tbl) do keys[#keys + 1] = k end ---@type string
|
||||||
|
table.sort(keys)
|
||||||
|
return keys
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Allocate one phase's auto-reg mappings.
|
||||||
|
-- Returns (allocated_map, errors). On pool exhaustion, errors is populated and the function halts.
|
||||||
|
--- @param phase_label string
|
||||||
|
--- @param decls table<string, string> -- bag: auto-reg symbol -> decl payload
|
||||||
|
--- @return GprAllocMap
|
||||||
|
--- @return Finding[]
|
||||||
|
local function allocate_phase(phase_label, decls)
|
||||||
|
-- Deep-copy POOL into a fresh sequence table. The original `table.unpack and table.unpack(POOL) or { unpack(POOL) }`
|
||||||
|
-- idiom wraps the unpacked values in a single inner table under LuaJIT 5.1 (`table.unpack` is nil; the `or` returns one value),
|
||||||
|
-- which corrupts the pool into `{ {R_T0, R_T1, ...} }` — making `table.remove(pool, 1)` return the inner table on iteration.
|
||||||
|
local pool = {} ---@type GprIdent[]
|
||||||
|
for i = 1, #POOL do pool[i] = POOL[i] end ---@type integer
|
||||||
|
local result = {} ---@type GprAllocMap
|
||||||
|
local errors = {} ---@type Finding[]
|
||||||
|
for _, sym in ipairs(stable_sort_keys(decls)) do ---@type integer, string
|
||||||
|
local next_gpr = table.remove(pool, 1) ---@type GprIdent|nil
|
||||||
|
if not next_gpr then
|
||||||
|
errors[#errors + 1] = {
|
||||||
|
line = 0,
|
||||||
|
msg = string.format("phase_register_pool_exhausted: "
|
||||||
|
.. "phase '%s' requested symbol '%s' but the pool has no remaining registers "
|
||||||
|
.. "(max 24 per phase: R_T0..R_T7 + R_V0..R_V1 + R_A0..R_A3 + R_S0..R_S7 + R_T8..R_T9). Split the phase or use hardcoded GPRs."
|
||||||
|
, phase_label, sym),
|
||||||
|
}
|
||||||
|
return result, errors
|
||||||
|
end
|
||||||
|
result[sym] = next_gpr
|
||||||
|
end
|
||||||
|
return result, errors
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Build two projections from corpus.register_alias_registry:
|
||||||
|
-- user_pinned -- { [physical_gpr_ident] = true } -- GPRs unavailable to auto_reg globally (wave-context carriers, file-scope pinned aliases)
|
||||||
|
-- alias_to_gpr -- { [alias_ident] = physical_gpr_ident } -- for body parsing
|
||||||
|
-- Both projections are derived from the same set of entries: every AliasEntry in register_alias_registry has `has_atom_reg = true`
|
||||||
|
-- (only those entries are added to the registry; see passes/scan_source.lua parse_enum_entry).
|
||||||
|
-- Each entry's `code` is the integer MIPS GPR number (0..31); INT_CODE_TO_POOL_GPR translates it back to the physical GPR ident.
|
||||||
|
-- Aliases whose `code` points to a non-POOL GPR (e.g. R_S0, R_T8, R_K1) are ignored —
|
||||||
|
-- they don't affect the auto_reg pool, and they're already excluded from POOL above.
|
||||||
|
--- @param corpus Corpus
|
||||||
|
--- @return table<GprIdent, boolean>
|
||||||
|
--- @return table<string, GprIdent>
|
||||||
|
local function build_user_pins(corpus)
|
||||||
|
local user_pinned = {} ---@type table<GprIdent, boolean> -- bag: pinned physical GPR -> true
|
||||||
|
local alias_to_gpr = {} ---@type table<string, GprIdent> -- bag: alias ident -> physical GPR
|
||||||
|
if not corpus.register_alias_registry then return user_pinned, alias_to_gpr end
|
||||||
|
for alias_name, alias_entry in pairs(corpus.register_alias_registry) do ---@type string, AliasEntry
|
||||||
|
if alias_entry.has_atom_reg and alias_entry.code then
|
||||||
|
local gpr = INT_CODE_TO_POOL_GPR[alias_entry.code] ---@type GprIdent|nil
|
||||||
|
if gpr then
|
||||||
|
user_pinned[gpr] = true
|
||||||
|
alias_to_gpr[alias_name] = gpr
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return user_pinned, alias_to_gpr
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Find every physical GPR referenced in the atom body, via EITHER:
|
||||||
|
--- (a) A hardcoded physical GPR ident (R_T\d+|R_V\d+|R_A\d+|R_S\d+) — the existing regex;
|
||||||
|
--- (b) An alias ident (R_<Alias>) resolved via alias_to_gpr back to its physical GPR ident.
|
||||||
|
--- Returns { [physical_gpr_ident] = count }.
|
||||||
|
--- Clash-detection and source-pool-exclusion logic only needs the presence of each GPR (boolean test),
|
||||||
|
--- but keeping count preserves the original find_hardcoded_rn shape so callers can switch without churn.
|
||||||
|
--- The alias pattern is sorted lexicographically to keep the regex deterministic.
|
||||||
|
--- @param body_text string
|
||||||
|
--- @param alias_to_gpr table<string, GprIdent> -- bag: alias ident -> physical GPR
|
||||||
|
--- @return table<GprIdent, integer>
|
||||||
|
local function find_used_gprs(body_text, alias_to_gpr)
|
||||||
|
local found = {} ---@type table<GprIdent, integer> -- bag: physical GPR -> hit count
|
||||||
|
-- (a) Hardcoded physical GPRs (R_T0..R_T7, R_V0..R_V1, R_A0..R_A3, R_S0..R_S7).
|
||||||
|
for gpr in body_text:gmatch("(R_T%d+|R_V%d+|R_A%d+|R_S%d+)") do ---@type GprIdent
|
||||||
|
found[gpr] = (found[gpr] or 0) + 1
|
||||||
|
end
|
||||||
|
-- (b) Alias references (R_<Alias>) resolved to physical GPRs via the registry.
|
||||||
|
-- Sorted by name so the regex is byte-stable across runs.
|
||||||
|
if alias_to_gpr and next(alias_to_gpr) then
|
||||||
|
local aliases = {} ---@type string[]
|
||||||
|
for alias_name in pairs(alias_to_gpr) do ---@type string
|
||||||
|
aliases[#aliases + 1] = alias_name
|
||||||
|
end
|
||||||
|
table.sort(aliases)
|
||||||
|
local pattern = "(" .. table.concat(aliases, "|") .. ")" ---@type string
|
||||||
|
for alias_name in body_text:gmatch(pattern) do ---@type string
|
||||||
|
local gpr = alias_to_gpr[alias_name] ---@type GprIdent|nil
|
||||||
|
if gpr and not found[gpr] then
|
||||||
|
found[gpr] = 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return found
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Emit one gen/auto_reg.h header per directory.
|
||||||
|
--- @param out_dir string
|
||||||
|
--- @param dir string
|
||||||
|
--- @param sources SourceFile[]
|
||||||
|
--- @param mappings GprAllocMap
|
||||||
|
--- @return string|nil
|
||||||
|
local function emit_auto_reg_h(out_dir, dir, sources, mappings)
|
||||||
|
if not mappings or next(mappings) == nil then return end
|
||||||
|
local out_path = out_dir .. "/" .. "auto_reg.h" ---@type string
|
||||||
|
duffle.ensure_dir(out_dir)
|
||||||
|
local lines = { ---@type string[]
|
||||||
|
"#ifdef INTELLISENSE_DIRECTIVES",
|
||||||
|
"#pragma once",
|
||||||
|
"#endif",
|
||||||
|
"// Auto-generated by ps1_meta.lua (passes/auto_reg.lua) — DO NOT EDIT",
|
||||||
|
"// Directory: " .. dir:gsub("/", "\\"),
|
||||||
|
}
|
||||||
|
for _, src in ipairs(sources) do ---@type integer, SourceFile
|
||||||
|
lines[#lines + 1] = "// source: " .. src.path
|
||||||
|
end
|
||||||
|
lines[#lines + 1] = "// Per-phase register allocations resolved by the lua pass."
|
||||||
|
lines[#lines + 1] = "// R_<Sym>_Code = <chosen GPR's _Code constant> for every marker in this directory."
|
||||||
|
lines[#lines + 1] = ""
|
||||||
|
for _, sym in ipairs(stable_sort_keys(mappings)) do ---@type integer, string
|
||||||
|
local gpr = mappings[sym] ---@type GprIdent
|
||||||
|
local gpr_code = gpr .. "_Code" ---@type string
|
||||||
|
lines[#lines + 1] = "#define " .. sym .. "_Code " .. gpr_code
|
||||||
|
end
|
||||||
|
lines[#lines + 1] = ""
|
||||||
|
duffle.write_file_lf(out_path, table.concat(lines, "\n") .. "\n")
|
||||||
|
print(" -> " .. out_path)
|
||||||
|
return out_path
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Pass entry
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
local M = {} ---@type AutoRegPass
|
||||||
|
|
||||||
|
--- @param ctx PassCtx
|
||||||
|
--- @return AutoRegResult
|
||||||
|
function M.run(ctx)
|
||||||
|
local outputs = {} ---@type AutoRegOutput[]
|
||||||
|
local errors = {} ---@type Finding[]
|
||||||
|
local warnings = {} ---@type Finding[]
|
||||||
|
|
||||||
|
local corpus = ctx.shared and ctx.shared.corpus ---@type Corpus|nil
|
||||||
|
if type(corpus) ~= "table" then
|
||||||
|
error("auto_reg.run requires ctx.shared.corpus", 0)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- 0. Build the user-pinned GPR exclusion set + alias-to-GPR resolution map.
|
||||||
|
-- Wave-context carriers (e.g. `R_ResolveScratch = R_T4 atom_reg` in hello_camera.atom.c)
|
||||||
|
-- MUST NOT be allocated to any auto-reg marker — they're preserved across atoms by the wave-context discipline.
|
||||||
|
-- The corpus's register_alias_registry is the source of truth for these opt-in pins.
|
||||||
|
-- Body references to those aliases (via alias_to_gpr) are also excluded on a per-atom basis in step 2 below.
|
||||||
|
local user_pinned, alias_to_gpr = build_user_pins(corpus) ---@type table<GprIdent, boolean>, table<string, GprIdent>
|
||||||
|
|
||||||
|
-- 1. Allocate phase pools first (phase declarations take precedence over per-atom declarations).
|
||||||
|
local phase_allocations = {} ---@type table<string, GprAllocMap> -- bag: phase_label -> alloc map
|
||||||
|
for phase_label, decls in pairs(corpus.phase_auto_regs or {}) do ---@type string, table<string, string>
|
||||||
|
local mapping, errs = allocate_phase(phase_label, decls) ---@type GprAllocMap, Finding[]
|
||||||
|
for sym, gpr in pairs(mapping) do ---@type string, GprIdent
|
||||||
|
phase_allocations[phase_label] = phase_allocations[phase_label] or {}
|
||||||
|
phase_allocations[phase_label][sym] = gpr
|
||||||
|
end
|
||||||
|
for _, e in ipairs(errs) do ---@type integer, Finding
|
||||||
|
errors[#errors + 1] = e
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- 2. Allocate per-atom auto-regs. If the atom scope matches a phase, reuse the phase pool.
|
||||||
|
-- Otherwise, allocate a private pool for the atom.
|
||||||
|
-- The phase membership is in `corpus.atom_phases[phase_label].atoms` (an array of atom names declared via `atom_phase(<phase>)`
|
||||||
|
-- in the atom's `atom_info` line). Build a reverse map `atom_name -> phase_label` so the lookup is O(1) per atom scope.
|
||||||
|
local atom_name_to_phase = {} ---@type table<AtomName, string> -- bag: atom name -> phase label
|
||||||
|
for phase_label, entry in pairs(corpus.atom_phases or {}) do ---@type string, AtomPhaseGroup
|
||||||
|
for _, atom_name in ipairs(entry.atoms or {}) do ---@type integer, AtomName
|
||||||
|
atom_name_to_phase[atom_name] = phase_label
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
local atom_allocations = {} ---@type table<AtomName, GprAllocMap> -- bag: atom scope -> alloc map
|
||||||
|
for atom_scope, decls in pairs(corpus.atom_auto_regs or {}) do ---@type AtomName, table<string, string>
|
||||||
|
local phase_label = atom_name_to_phase[atom_scope] ---@type string|nil
|
||||||
|
-- Build the atom's source pool: start with the full POOL, subtract:
|
||||||
|
-- (a) every GPR already committed (phase allocations + prior atom allocations)
|
||||||
|
-- (b) every USER-PINNED GPR (wave-context carriers + file-scope pinned aliases)
|
||||||
|
-- (c) every GPR referenced in the atom's body — either hardcoded R_X or alias R_Xxx
|
||||||
|
-- (the latter resolved via alias_to_gpr; this catches cases where the user wrote R_ResolveScratch instead of R_T4 directly)
|
||||||
|
-- Atoms whose scope matches a phase share the global pool with the phase allocations;
|
||||||
|
-- the original `source_pool = phase_allocations[phase_label]` form used the phase
|
||||||
|
-- allocation MAP as a pool, but that map has no array part, so `table.remove(source_pool, 1)`
|
||||||
|
-- returned nil and every atom-with-phase marker errored with `phase_register_pool_exhausted`.
|
||||||
|
local used = {} ---@type table<GprIdent, boolean> -- bag: committed or body-referenced GPR -> true
|
||||||
|
for _, m in pairs(phase_allocations) do for _, gpr in pairs(m) do used[gpr] = true end end ---@type integer, GprAllocMap
|
||||||
|
for _, m in pairs(atom_allocations) do for _, gpr in pairs(m) do used[gpr] = true end end ---@type integer, GprAllocMap
|
||||||
|
-- (c) Body references — scan the atom body for hardcoded + alias-resolved GPRs.
|
||||||
|
-- Folded into `used` so the source_pool exclusion is a single check.
|
||||||
|
local atom = corpus.atoms_by_name and corpus.atoms_by_name[atom_scope] ---@type AtomEntry|nil
|
||||||
|
if atom and atom.body then
|
||||||
|
local body_used = find_used_gprs(atom.body, alias_to_gpr) ---@type table<GprIdent, integer>
|
||||||
|
for gpr in pairs(body_used) do used[gpr] = true end ---@type GprIdent
|
||||||
|
end
|
||||||
|
local source_pool = {} ---@type GprIdent[]
|
||||||
|
for _, gpr in ipairs(POOL) do ---@type integer, GprIdent
|
||||||
|
-- Exclude (a) prior commitments, (b) USER-PINNED GPRs (wave-context carriers declared via atom_reg + _Code defs, preserved across atoms globally).
|
||||||
|
if not used[gpr] and not user_pinned[gpr] then
|
||||||
|
source_pool[#source_pool + 1] = gpr
|
||||||
|
end
|
||||||
|
end
|
||||||
|
local result = {} ---@type GprAllocMap
|
||||||
|
for _, sym in ipairs(stable_sort_keys(decls)) do ---@type integer, string
|
||||||
|
local next_gpr = table.remove(source_pool, 1) ---@type GprIdent|nil
|
||||||
|
if not next_gpr then
|
||||||
|
errors[#errors + 1] = {
|
||||||
|
line = 0,
|
||||||
|
msg = string.format("phase_register_pool_exhausted: atom '%s' requested symbol '%s' "
|
||||||
|
.. "but no free registers remain in its scope pool."
|
||||||
|
, atom_scope, sym),
|
||||||
|
}
|
||||||
|
else
|
||||||
|
result[sym] = next_gpr
|
||||||
|
end
|
||||||
|
end
|
||||||
|
atom_allocations[atom_scope] = result
|
||||||
|
end
|
||||||
|
|
||||||
|
-- 3. Conflict-with-hardcoded detection (defensive — should be unreachable now).
|
||||||
|
-- The source_pool exclusion in step 2 (b) + (c) already accounts for both user-pinned GPRs
|
||||||
|
-- and body-referenced GPRs (hardcoded R_Tn OR alias R_<Alias>).
|
||||||
|
-- An auto-reg allocation that matched an existing body reference would be impossible by construction.
|
||||||
|
-- This warning is kept as a defensive safety net for cases the body scanner might miss
|
||||||
|
-- (e.g. macros that expand to register references the scanner cannot resolve).
|
||||||
|
-- For each resolved (scope, sym) -> R_Tn mapping, scan the atom body source for used GPRs.
|
||||||
|
for atom_scope, decls in pairs(atom_allocations) do ---@type AtomName, GprAllocMap
|
||||||
|
local atom = corpus.atoms_by_name and corpus.atoms_by_name[atom_scope] ---@type AtomEntry|nil
|
||||||
|
if atom and atom.body then
|
||||||
|
local used_in_body = find_used_gprs(atom.body, alias_to_gpr) ---@type table<GprIdent, integer>
|
||||||
|
for sym, allocated_gpr in pairs(decls) do ---@type string, GprIdent
|
||||||
|
if used_in_body[allocated_gpr] and used_in_body[allocated_gpr] > 0 then
|
||||||
|
warnings[#warnings + 1] = {
|
||||||
|
line = atom.line or 0,
|
||||||
|
msg = string.format("phase_register_clash: atom '%s' has hardcoded '%s' in its body AND an auto-reg marker '%s' "
|
||||||
|
.. "that was allocated to '%s' (same phase). Resolve by removing the hardcoded reference or renaming the auto-reg."
|
||||||
|
, atom_scope, allocated_gpr, sym, allocated_gpr),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- 4. Emit per-directory gen/auto_reg.h.
|
||||||
|
-- For each source directory that has atom_auto_regs or phase_auto_regs entries, emit one header.
|
||||||
|
local sources_by_dir = corpus.sources_by_dir or {} ---@type table<string, SourceFile[]>
|
||||||
|
for dir, sources in pairs(sources_by_dir) do ---@type string, SourceFile[]
|
||||||
|
local per_dir_mappings = {} ---@type GprAllocMap
|
||||||
|
for _, src in ipairs(sources) do ---@type integer, SourceFile
|
||||||
|
-- Collect every (sym -> gpr) entry that originated from a source in this directory.
|
||||||
|
-- `src.scan.atom_auto_regs` is keyed by ATOM SCOPE NAME; `pairs(t)` iterates KEYS so `scope_name` here is the scope ident (e.g. "cube_g4_face").
|
||||||
|
-- The previous `for _, scan_atom_auto` form silently assigned the VALUE (a `{sym = sym}` table) to the variable,
|
||||||
|
-- which made `atom_allocations[scan_atom_auto]` a table-indexed lookup that never resolved.
|
||||||
|
for scope_name in pairs(src.scan and src.scan.atom_auto_regs or {}) do ---@type string
|
||||||
|
for sym, gpr in pairs(atom_allocations[scope_name] or {}) do ---@type string, GprIdent
|
||||||
|
per_dir_mappings[sym] = gpr
|
||||||
|
end
|
||||||
|
end
|
||||||
|
for scope_name in pairs(src.scan and src.scan.phase_auto_regs or {}) do ---@type string
|
||||||
|
for sym, gpr in pairs(phase_allocations[scope_name] or {}) do ---@type string, GprIdent
|
||||||
|
per_dir_mappings[sym] = gpr
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
local out_dir = dir .. "/gen" ---@type string
|
||||||
|
local out_path = emit_auto_reg_h(out_dir, dir, sources, per_dir_mappings) ---@type string|nil
|
||||||
|
if out_path then outputs[#outputs + 1] = { auto_reg_h = out_path } end
|
||||||
|
end
|
||||||
|
return { outputs = outputs, errors = errors, warnings = warnings }
|
||||||
|
end
|
||||||
|
|
||||||
|
M.POOL = POOL
|
||||||
|
|
||||||
|
return M
|
||||||
@@ -0,0 +1,920 @@
|
|||||||
|
--- passes/components.lua — Component-macro header generator.
|
||||||
|
---
|
||||||
|
--- Ownership: `corpus.word_counts` and `corpus.components`.
|
||||||
|
--- Scanner owns `declaration_comment` and `debug_skip` on each declaration record; this pass projects both forward.
|
||||||
|
---
|
||||||
|
--- Reads the pre-scanned SourceScan payload from `duffle.scan_source` for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations (kind="comp_bare" / "comp_proc"),
|
||||||
|
--- then resolves the function-args string from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration via a backward walk.
|
||||||
|
---
|
||||||
|
--- `MipsAtom_Proc_(X, ab, { body })` declarations (kind="atom_proc") are ATOMS, not components, and are deliberately excluded —
|
||||||
|
--- the ELF symbol is the C ident. Raw `MipsCode code_*` is leftover, not the atom rule.
|
||||||
|
---
|
||||||
|
--- Emits one `gen/macs.h` per *immediate source directory* with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
|
||||||
|
--- All sources inside the same directory contribute to the same file (per-directory aggregation).
|
||||||
|
--- The directory itself is the namespace, so the filename does not repeat the module name.
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Module-scope requires + package.path setup
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
-- Bootstrap: same as entry scripts. See `ps1_meta.lua` for the rationale.
|
||||||
|
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
||||||
|
-- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator.
|
||||||
|
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
|
||||||
|
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
||||||
|
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" ---@type string
|
||||||
|
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") ---@type DuffleExport
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Constants
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
-- Atom component declaration identifiers.
|
||||||
|
local ATOM_COMP_PROC = "MipsAtomComp_Proc_" ---@type string
|
||||||
|
local MIPS_ATOM = "Slice_MipsCode" ---@type string -- prefix on the function declaration that wraps an AtomComp_Proc_
|
||||||
|
|
||||||
|
-- Component-name prefixes.
|
||||||
|
local AC_PREFIX = "ac_" ---@type string -- arg to MipsAtomComp_(ac_X); the X is the atom name
|
||||||
|
local AC_PREFIX_LEN = 3 ---@type integer
|
||||||
|
local MAC_PREFIX = "mac_" ---@type string -- prefix on generated macros; the rest is the atom name
|
||||||
|
local MAC_PREFIX_LEN = 4 ---@type integer
|
||||||
|
|
||||||
|
-- ASCII byte values used in tokenization.
|
||||||
|
local BYTE_NEWLINE = 10 ---@type integer
|
||||||
|
local BYTE_SLASH = 47 ---@type integer
|
||||||
|
|
||||||
|
-- Output gen subdirectory + filename (per-directory aggregation; the directory name is the namespace).
|
||||||
|
local GEN_SUBDIR = "gen" ---@type string
|
||||||
|
local MACS_FILENAME = "macs.h" ---@type string
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Type declarations
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
-- SourceFile, PassCtx, PassResult: see ps1_meta.lua
|
||||||
|
-- DuffleExport: see duffle.lua
|
||||||
|
-- SourceScan, AtomEntry, CorpusCollision, CollisionSite: see scan_source.lua
|
||||||
|
-- BodyToken: see emission_model.lua
|
||||||
|
-- WordCounts: see word_count_eval.lua
|
||||||
|
-- InstructionRow, GteCommandRow: see duffle_isa.lua
|
||||||
|
|
||||||
|
--- @class Component
|
||||||
|
--- @field name string -- Atom name (without `ac_` prefix)
|
||||||
|
--- @field body string -- Brace-delimited body (without the braces)
|
||||||
|
--- @field body_off integer|nil -- Byte offset of body[1] in source
|
||||||
|
--- @field body_tokens BodyToken[]|nil
|
||||||
|
--- @field args string|nil -- Function-args string (function form only)
|
||||||
|
--- @field arg_names string[]|nil -- Formal names with leading `ab` dropped
|
||||||
|
--- @field line integer -- Source line of the declaration
|
||||||
|
--- @field comment string|nil -- Scanner-owned `declaration_comment`; the components pass reads it from the scanner record
|
||||||
|
--- @field kind string -- "comp_bare" | "comp_proc" (atom_proc is NOT a component — see `project_components`)
|
||||||
|
--- @field debug_skip boolean -- Mirror of `a.debug_skip` (scanner-owned); true iff a bare `atom_dbg_skip` marker immediately preceded the declaration
|
||||||
|
--- @field path string|nil -- Slash-normalized source path (collision sites)
|
||||||
|
--- @field source string|nil -- Absolute source path (emit)
|
||||||
|
--- @field line_of (fun(pos: integer): integer)|nil
|
||||||
|
--- @field cycle_cost integer|nil -- From metadata[c.name]; nil when the body was not costed
|
||||||
|
--- @field gp0_contrib integer|nil -- From metadata[c.name]; nil when the body was not costed
|
||||||
|
|
||||||
|
--- @class ComponentMeta
|
||||||
|
--- @field cycle_cost integer
|
||||||
|
--- @field gp0_contrib integer
|
||||||
|
|
||||||
|
--- @class ComponentMetaMap
|
||||||
|
--- @field [string] ComponentMeta -- bag: bare component name -> meta
|
||||||
|
|
||||||
|
--- @class MacsOutput
|
||||||
|
--- @field macs_h string
|
||||||
|
|
||||||
|
--- @class ComponentsPass
|
||||||
|
--- @field run fun(ctx: PassCtx): PassResult
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Local helpers (file I/O + path normalization)
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
local M = {} ---@type ComponentsPass
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Back-walk helpers (composed into the entry point below: find_function_args_for)
|
||||||
|
--
|
||||||
|
-- Only the function-args lookup for proc components occurs here.
|
||||||
|
-- The preceding-comment walk occur in `scan_source.lua` — `a.declaration_comment` carries the resolved comment,
|
||||||
|
-- so this file reads it forward rather than re-walking the source.
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Find the args of the function declaration that immediately precedes a `MipsAtomComp_Proc_` invocation.
|
||||||
|
--- Returns the args string (e.g., `"U4 off, U4 code, U1 r, U1 g, U1 b"`) or nil if no function declaration is found.
|
||||||
|
---
|
||||||
|
--- After the `sym` arg was dropped from MipsAtomComp_Proc_, the component name
|
||||||
|
--- and the args both come from the preceding `FI_ Slice_MipsCode ac_X(args)` declaration.
|
||||||
|
--- The shared `duffle.find_function_decl_for` helper does the backward walk; this function returns just the args.
|
||||||
|
---
|
||||||
|
--- @param source string
|
||||||
|
--- @param name string (retained for signature stability; unused — the walk derives the name)
|
||||||
|
--- @param before_pos integer
|
||||||
|
--- @return string|nil
|
||||||
|
local function find_function_args_for(source, name, before_pos)
|
||||||
|
local _, args_inner = duffle.find_function_decl_for(source, before_pos, #MIPS_ATOM) ---@type string|nil, string|nil
|
||||||
|
return args_inner
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Argument-name extraction
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Extract just the parameter NAMES from a function-args string (stripping type annotations). E.g.,
|
||||||
|
--- `"U4 off, U4 code, U1 r, U1 g, U1 b"` -> `{"off", "code", "r", "g", "b"}`
|
||||||
|
--- `"U4 *ptr"` -> `{"ptr"}`
|
||||||
|
--- `""` -> nil
|
||||||
|
--- @param args_str string|nil
|
||||||
|
--- @return string[]|nil
|
||||||
|
local function extract_arg_names(args_str)
|
||||||
|
if not args_str or args_str == "" then return nil end
|
||||||
|
local names = {} ---@type string[]
|
||||||
|
local tokens = duffle.split_top_level_commas(args_str) ---@type string[]
|
||||||
|
for _, tok in ipairs(tokens) do ---@type integer, string
|
||||||
|
local trimmed = duffle.trim(tok) ---@type string
|
||||||
|
if trimmed ~= "" then
|
||||||
|
-- Strip trailing block comment (/* ... */) from the token, if present.
|
||||||
|
-- split_top_level_commas only skips block comments at TOP LEVEL (between commas),
|
||||||
|
-- not block comments embedded WITHIN a token between a parameter and a trailing comma.
|
||||||
|
-- Without this strip, the identifier-walk below stops at the `/` of `*/` and returns
|
||||||
|
-- the wrong name (or nothing). See `test_extract_arg_names_handles_trailing_block_comments`.
|
||||||
|
local trimmed_end = #trimmed ---@type integer
|
||||||
|
if trimmed_end >= 2 and trimmed:sub(trimmed_end - 1, trimmed_end) == "*/" then
|
||||||
|
-- Find the matching `/*` that opens the trailing comment.
|
||||||
|
-- Walk back from the `*/` looking for `/*` (whitespace + `/*`).
|
||||||
|
local close_pos = trimmed_end - 1 ---@type integer -- position of the second-to-last char
|
||||||
|
-- Walk back: skip trailing whitespace, then look for the `/*` opener.
|
||||||
|
while close_pos > 1 do
|
||||||
|
local ch = trimmed:sub(close_pos, close_pos) ---@type string
|
||||||
|
if ch == " " or ch == "\t" or ch == "\n" or ch == "\r" then
|
||||||
|
close_pos = close_pos - 1
|
||||||
|
else
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
-- Now scan back from close_pos for the `/*` opener (slashes are at close_pos-1 and close_pos-2).
|
||||||
|
local opener_pos = nil ---@type integer|nil
|
||||||
|
local scan = close_pos - 3 ---@type integer
|
||||||
|
while scan >= 1 do
|
||||||
|
if trimmed:sub(scan, scan + 1) == "/*" then
|
||||||
|
opener_pos = scan
|
||||||
|
break
|
||||||
|
end
|
||||||
|
scan = scan - 1
|
||||||
|
end
|
||||||
|
if opener_pos then
|
||||||
|
-- Truncate everything from opener_pos onwards.
|
||||||
|
trimmed = duffle.trim(trimmed:sub(1, opener_pos - 1))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if trimmed == "" then goto continue end
|
||||||
|
-- Strip trailing array suffix `[N]` if present.
|
||||||
|
-- Example: `Reg r_data[4]` → identifier is `r_data`, not `4`.
|
||||||
|
trimmed_end = #trimmed
|
||||||
|
if trimmed_end >= 4 and trimmed:sub(trimmed_end, trimmed_end) == "]" then
|
||||||
|
-- Walk back: skip digits, expect `[`.
|
||||||
|
local bracket_pos = trimmed_end - 1 ---@type integer
|
||||||
|
while bracket_pos > 1 do
|
||||||
|
local ch = trimmed:sub(bracket_pos, bracket_pos) ---@type string
|
||||||
|
if ch >= "0" and ch <= "9" then
|
||||||
|
bracket_pos = bracket_pos - 1
|
||||||
|
else
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if bracket_pos >= 1 and trimmed:sub(bracket_pos, bracket_pos) == "[" then
|
||||||
|
trimmed = duffle.trim(trimmed:sub(1, bracket_pos - 1))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if trimmed == "" then goto continue end
|
||||||
|
-- Find the identifier at the end: walk back over trailers (whitespace + `*` + `[]`),
|
||||||
|
-- then walk back over the identifier chars (alnum + `_`).
|
||||||
|
local ident_end = #trimmed ---@type integer
|
||||||
|
while ident_end > 0 do
|
||||||
|
local ch = trimmed:sub(ident_end, ident_end) ---@type string
|
||||||
|
if ch == " " or ch == "\t" or ch == "*" or ch == "]" or ch == "[" then
|
||||||
|
ident_end = ident_end - 1
|
||||||
|
else
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
local ident_start = ident_end ---@type integer
|
||||||
|
while ident_start > 0 do
|
||||||
|
local ch = trimmed:sub(ident_start, ident_start) ---@type string
|
||||||
|
if duffle.is_alnum_byte(string.byte(ch)) or ch == "_" then
|
||||||
|
ident_start = ident_start - 1
|
||||||
|
else
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
ident_start = ident_start + 1
|
||||||
|
local name = trimmed:sub(ident_start, ident_end) ---@type string
|
||||||
|
if name ~= "" then names[#names + 1] = name end
|
||||||
|
::continue::
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if #names == 0 then return nil end
|
||||||
|
return names
|
||||||
|
end
|
||||||
|
|
||||||
|
--- @param args_str string|nil
|
||||||
|
--- @return string[]|nil
|
||||||
|
local function formal_arg_names(args_str)
|
||||||
|
local names = extract_arg_names(args_str) ---@type string[]|nil
|
||||||
|
if not names then return nil end
|
||||||
|
if names[1] == "ab" then table.remove(names, 1) end
|
||||||
|
if #names == 0 then return nil end
|
||||||
|
return names
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Component projection (read from pre-scanned SourceScan)
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Project pre-scanned MipsAtomComp_ / MipsAtomComp_Proc_ entries into Component shape.
|
||||||
|
--- Reads the scanner-owned `declaration_comment` (resolved by scan_source.lua, skipping backward across an associated bare `atom_dbg_skip` marker when present).
|
||||||
|
--- Per-source backward lookups remain in place only for the function `args` of proc components.
|
||||||
|
--- That lookup is unique to components.lua and stays separate from the declaration-comment walk.
|
||||||
|
--- Carries `body_tokens` forward from scan-source so word_count_rec reads from the precomputed table instead of calling duffle.tokenize_body again.
|
||||||
|
--- Carries the scanner-owned `debug_skip` flag forward so the generated projection can emit `/* atom_dbg_skip */`
|
||||||
|
--- before the authored comment and so `update_canonical_components` can mirror the same field onto `corpus.components[name]`.
|
||||||
|
--- @param source string -- the full source text (needed for backward lookups)
|
||||||
|
--- @param scan SourceScan
|
||||||
|
--- @return Component[]
|
||||||
|
local function project_components(source, scan)
|
||||||
|
local out = {} ---@type Component[]
|
||||||
|
for _, a in ipairs(scan.atoms) do ---@type integer, AtomEntry
|
||||||
|
-- Only `MipsAtomComp_(ac_X)` (kind="comp_bare") and `MipsAtomComp_Proc_(ac_X, ...)` (kind="comp_proc")
|
||||||
|
-- are COMPONENTS — they get inlined via `mac_<name>` aliases inside atom bodies.
|
||||||
|
-- `MipsAtom_Proc_` (kind="atom_proc") is an ATOM (ends with `mac_yield()`); it gets emitted via
|
||||||
|
-- `tb_emit` of the C ident, NOT inlined as a macro. Including `atom_proc` here
|
||||||
|
-- would incorrectly emit `mac_<name>` aliases for atoms, polluting `gen/macs.h`.
|
||||||
|
-- See `docs/duffle_dsl_primer.md` §"mac_* aliases" for the contract.
|
||||||
|
if a.kind == "comp_bare" or a.kind == "comp_proc" then
|
||||||
|
-- Function-args lookup is meaningful for `MipsAtomComp_Proc_` components
|
||||||
|
-- (the macro sits inside `FI_ Slice_MipsCode ac_X(...)`); the alias expansion
|
||||||
|
-- discards the `ab` (atom-builder) arg the same way both forms do.
|
||||||
|
local args = find_function_args_for(source, a.raw_name, a.ident_pos) ---@type string|nil
|
||||||
|
-- Comment ownership: scan_source.lua stamps `declaration_comment` on the record by walking backward past any associated bare marker.
|
||||||
|
-- The pass reads `declaration_comment` directly.
|
||||||
|
local comment = a.declaration_comment or "" ---@type string
|
||||||
|
out[#out + 1] = {
|
||||||
|
line = a.line,
|
||||||
|
name = a.name,
|
||||||
|
body = a.body,
|
||||||
|
body_off = a.body_off,
|
||||||
|
body_tokens = a.body_tokens,
|
||||||
|
args = args,
|
||||||
|
arg_names = formal_arg_names(args),
|
||||||
|
comment = comment,
|
||||||
|
kind = a.kind, -- "comp_bare" | "comp_proc"; provenance emitter reads this.
|
||||||
|
debug_skip = a.debug_skip == true,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return out
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Line-comment → block-comment conversion
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
-- Convert `//` line comments to `/* */` block comments in a token.
|
||||||
|
-- C macros use `\` line-continuations; a `//` comment before `\` would consume the continuation,
|
||||||
|
-- breaking the macro. We convert `//` to `/* */` so the multi-line macro structure is preserved.
|
||||||
|
--
|
||||||
|
-- Skips `//` sequences that are inside string or character literals
|
||||||
|
-- (a rough heuristic — sufficient for component bodies which don't have those constructs).
|
||||||
|
--- @param s string
|
||||||
|
--- @return string
|
||||||
|
local function convert_line_comments_to_block(s)
|
||||||
|
local result = s ---@type string
|
||||||
|
local pos = 1 ---@type integer
|
||||||
|
local len = #result ---@type integer
|
||||||
|
while pos <= len do
|
||||||
|
local is_double_slash = result:byte(pos) == BYTE_SLASH ---@type boolean
|
||||||
|
and pos + 1 <= len and result:byte(pos + 1) == BYTE_SLASH
|
||||||
|
if not is_double_slash then
|
||||||
|
pos = pos + 1
|
||||||
|
else
|
||||||
|
-- Find end of line.
|
||||||
|
local eol = pos ---@type integer
|
||||||
|
while eol <= len and result:byte(eol) ~= BYTE_NEWLINE do
|
||||||
|
eol = eol + 1
|
||||||
|
end
|
||||||
|
local before = result:sub(1, pos - 1) ---@type string
|
||||||
|
local comment = result:sub(pos + 2, eol - 1) ---@type string -- skip the `//`
|
||||||
|
local after ---@type string
|
||||||
|
if eol <= len and result:byte(eol) == BYTE_NEWLINE then
|
||||||
|
after = " */" .. result:sub(eol) -- keep the newline
|
||||||
|
else
|
||||||
|
after = " */"
|
||||||
|
end
|
||||||
|
result = before .. "/*" .. comment .. after
|
||||||
|
pos = #before + 2 + #comment + 3 -- skip past converted comment
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return result
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Word-count computation (memoized recursive lookup)
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Strip the `mac_` prefix from a component-call ident so we can look it up against the components-by-name table.
|
||||||
|
--- Returns the ident unchanged if it doesn't start with the prefix
|
||||||
|
--- (so a non-component ident like `mask_upper` falls through to the wc-table branch).
|
||||||
|
--- @param ident string|nil
|
||||||
|
--- @return string|nil
|
||||||
|
local function strip_mac_prefix(ident)
|
||||||
|
if not ident then return nil end
|
||||||
|
if ident:sub(1, MAC_PREFIX_LEN) == MAC_PREFIX then
|
||||||
|
return ident:sub(MAC_PREFIX_LEN + 1)
|
||||||
|
end
|
||||||
|
return ident
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Strip a leading delay marker (`LdSlot_` / `BdSlot_` / `GteDelay_` / `DmaSlot_`)
|
||||||
|
--- plus following whitespace and block comments. Returns the remainder, or ""
|
||||||
|
--- when the token is only the marker.
|
||||||
|
--- `BdSlot_ nop` becomes `nop`. Bare `LdSlot_` becomes "".
|
||||||
|
--- @param tok string
|
||||||
|
--- @return string
|
||||||
|
local function strip_leading_delay_marker(tok)
|
||||||
|
local ident = duffle.read_ident(tok, 1) ---@type string|nil
|
||||||
|
if not ident or not duffle.DELAY_MARKERS[ident] then return tok end
|
||||||
|
local rest = tok:sub(#ident + 1):match("^%s*(.*)$") or "" ---@type string
|
||||||
|
while rest:sub(1, 2) == "/*" do
|
||||||
|
local close = rest:find("*/", 3, true) ---@type integer|nil
|
||||||
|
if not close then return "" end
|
||||||
|
rest = rest:sub(close + 2):match("^%s*(.*)$") or ""
|
||||||
|
end
|
||||||
|
return rest
|
||||||
|
end
|
||||||
|
|
||||||
|
--- (internal) Recursive word-count lookup. `cache` is the memoization table shared across all components
|
||||||
|
--- in a single source's `count_all_components` pass; the in-progress -1 sentinel detects cycles (A -> B -> A).
|
||||||
|
--- @param name string -- the component name (without `mac_`)
|
||||||
|
--- @param comp_by_name table<string, Component>
|
||||||
|
--- @param wc WordCounts
|
||||||
|
--- @param cache table<string, integer> -- bag: name -> count; -1 in-progress sentinel
|
||||||
|
--- @return integer
|
||||||
|
local function word_count_rec(name, comp_by_name, wc, cache)
|
||||||
|
if cache[name] ~= nil then return cache[name] end
|
||||||
|
cache[name] = -1 -- mark in-progress (cycle detection)
|
||||||
|
local cc = comp_by_name[name] ---@type Component|nil
|
||||||
|
local n ---@type integer
|
||||||
|
if cc then
|
||||||
|
n = 0
|
||||||
|
local tokens = cc.body_tokens ---@type BodyToken[]
|
||||||
|
for _, t in ipairs(tokens) do ---@type integer, BodyToken
|
||||||
|
local trimmed = t.tok ---@type string
|
||||||
|
if trimmed ~= "" then
|
||||||
|
local work = trimmed ---@type string
|
||||||
|
while true do
|
||||||
|
local marker = duffle.read_ident(work, 1) ---@type string|nil
|
||||||
|
if marker and duffle.DELAY_MARKERS[marker] then
|
||||||
|
work = strip_leading_delay_marker(work)
|
||||||
|
if work == "" then break end
|
||||||
|
else
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if work ~= "" then
|
||||||
|
local lookup = strip_mac_prefix(duffle.read_ident(work, 1)) ---@type string|nil
|
||||||
|
if lookup == "atom_label" or lookup == "atom_offset" then
|
||||||
|
-- Pure metaprogram anchors; emit zero words.
|
||||||
|
elseif lookup and comp_by_name[lookup] then
|
||||||
|
-- It's a `mac_X(...)` call. Recurse.
|
||||||
|
n = n + word_count_rec(lookup, comp_by_name, wc, cache)
|
||||||
|
elseif lookup and wc and wc[lookup] then
|
||||||
|
-- Encoding macro or pseudo-instruction (e.g. mask_upper = 2, nop2 = 2).
|
||||||
|
n = n + wc[lookup]
|
||||||
|
else
|
||||||
|
-- Unrecognized token. Fall back to 1 word.
|
||||||
|
n = n + 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
else
|
||||||
|
-- Not a known component: assume 1 word (regular instruction).
|
||||||
|
n = 1
|
||||||
|
end
|
||||||
|
cache[name] = n
|
||||||
|
return n
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Compute word counts for every component in `components` in a single pass.
|
||||||
|
--- The name-lookup table + memoization cache are built ONCE (per source) instead of per-component,
|
||||||
|
--- so the cache survives across siblings and a component's recursive `mac_Y(...)`
|
||||||
|
--- references hit memoized values instead of re-walking the body.
|
||||||
|
--- Cycle detection (A -> B -> A) is preserved via the in-progress `-1` sentinel in `cache`.
|
||||||
|
--- @param components Component[]
|
||||||
|
--- @param wc WordCounts
|
||||||
|
--- @return table<string, integer> -- bag: bare component name -> word count
|
||||||
|
local function count_all_components(components, wc)
|
||||||
|
local comp_by_name = {} ---@type table<string, Component>
|
||||||
|
for _, cc in ipairs(components) do comp_by_name[cc.name] = cc end ---@type integer, Component
|
||||||
|
local cache = {} ---@type table<string, integer> -- bag: memo; -1 in-progress sentinel
|
||||||
|
local counts = {} ---@type table<string, integer> -- bag: bare name -> word count
|
||||||
|
for _, c in ipairs(components) do ---@type integer, Component
|
||||||
|
counts[c.name] = word_count_rec(c.name, comp_by_name, wc, cache)
|
||||||
|
end
|
||||||
|
return counts
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ═══════════════════════════════════════════
|
||||||
|
-- Per-component metadata derivation (replaces the hardcoded `M.GP0_MACRO_CONTRIB` + `M.INSTRUCTION_LATENCY[mac_*]` tables that previously lived in `duffle.lua`).
|
||||||
|
--
|
||||||
|
-- Each `MipsAtomComp_(ac_X) { body }` definition in `code/duffle/lottes_tape.h` is the canonical source.
|
||||||
|
-- The `mac_X(...)` macros are GENERATED from these definitions by `emit_component_macros_h` for tape-side composition;
|
||||||
|
-- the metaprogram must NEVER walk the generated variants to derive metadata.
|
||||||
|
-- Always walk the original `MipsAtomComp_` body via `cc.body_tokens`.
|
||||||
|
-- ═══════════════════════════════════════════
|
||||||
|
|
||||||
|
--- (internal) One walk of a component body that fills both `cycle_cost` and `gp0_contrib`.
|
||||||
|
--- Cycle: sum `isa.cycles` / `gte.cycles` / `latency[ident]` / 1 per leaf, recurse `mac_*`.
|
||||||
|
--- `mac_yield` cycle_cost is 0 (runtime cost lands in the next atom's prologue); its gp0 still comes from the token walk.
|
||||||
|
--- GP0: count `gte_sw` and `store_word` / `store_half` / `store_byte` that target `R_PrimCursor` / `O_(Poly_` / `r_prim_cursor` / `r_primitive_cursor` / `r_base`.
|
||||||
|
--- `insert_ot_tag*` gp0_contrib is 0; cycle still comes from the body walk.
|
||||||
|
--- Missing component: cycle 1, gp0 0.
|
||||||
|
--- @param name string -- component bare name (e.g. "yield", "pack_color_word")
|
||||||
|
--- @param comp_by_name table<string, Component>
|
||||||
|
--- @param latency table<string, integer> -- bag: ident -> cycle cost
|
||||||
|
--- @param cache ComponentMetaMap
|
||||||
|
--- @return ComponentMeta
|
||||||
|
local function component_meta_rec(name, comp_by_name, latency, cache)
|
||||||
|
if cache[name] ~= nil then return cache[name] end
|
||||||
|
cache[name] = { cycle_cost = -1, gp0_contrib = -1 }
|
||||||
|
local cc = comp_by_name[name] ---@type Component|nil
|
||||||
|
local cycle_cost ---@type integer
|
||||||
|
local gp0_contrib ---@type integer
|
||||||
|
if cc then
|
||||||
|
local skip_cycle = (name == "yield") ---@type boolean
|
||||||
|
local skip_gp0 = name:match("^insert_ot_tag") ~= nil ---@type boolean
|
||||||
|
cycle_cost = 0
|
||||||
|
gp0_contrib = 0
|
||||||
|
if not skip_cycle or not skip_gp0 then
|
||||||
|
local tokens = cc.body_tokens ---@type BodyToken[]
|
||||||
|
for _, t in ipairs(tokens) do ---@type integer, BodyToken
|
||||||
|
local trimmed = t.tok ---@type string
|
||||||
|
if trimmed ~= "" then
|
||||||
|
local ident = duffle.read_ident(trimmed, 1) ---@type string|nil
|
||||||
|
if ident and ident:sub(1, MAC_PREFIX_LEN) == MAC_PREFIX then
|
||||||
|
local nested = ident:sub(MAC_PREFIX_LEN + 1) ---@type string
|
||||||
|
local nested_meta = component_meta_rec(nested, comp_by_name, latency, cache) ---@type ComponentMeta
|
||||||
|
if not skip_cycle then
|
||||||
|
cycle_cost = cycle_cost + nested_meta.cycle_cost
|
||||||
|
end
|
||||||
|
if not skip_gp0 then
|
||||||
|
gp0_contrib = gp0_contrib + nested_meta.gp0_contrib
|
||||||
|
end
|
||||||
|
else
|
||||||
|
if not skip_cycle then
|
||||||
|
local isa = duffle.instr(ident) ---@type InstructionRow|nil
|
||||||
|
local gte = duffle.gte(ident) ---@type GteCommandRow|nil
|
||||||
|
cycle_cost = cycle_cost + ((isa and isa.cycles) or (gte and gte.cycles) or latency[ident] or 1)
|
||||||
|
end
|
||||||
|
if not skip_gp0 then
|
||||||
|
if ident == "gte_sw" then
|
||||||
|
gp0_contrib = gp0_contrib + 1
|
||||||
|
elseif ident == "store_word" or ident == "store_half" or ident == "store_byte" then
|
||||||
|
if trimmed:find("R_PrimCursor", 1, true)
|
||||||
|
or trimmed:find("O_(Poly_", 1, true)
|
||||||
|
or trimmed:find("r_prim_cursor", 1, true)
|
||||||
|
or trimmed:find("r_primitive_cursor", 1, true)
|
||||||
|
or trimmed:find("r_base", 1, true)
|
||||||
|
then
|
||||||
|
gp0_contrib = gp0_contrib + 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
else
|
||||||
|
cycle_cost = 1
|
||||||
|
gp0_contrib = 0
|
||||||
|
end
|
||||||
|
cache[name] = { cycle_cost = cycle_cost, gp0_contrib = gp0_contrib }
|
||||||
|
return cache[name]
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Compute `cycle_cost` + `gp0_contrib` for every component in `components` in a single pass.
|
||||||
|
--- One memoization cache; a nested `mac_Y` inside a `mac_X` body computes both fields once.
|
||||||
|
--- @param components Component[]
|
||||||
|
--- @param latency table<string, integer> -- bag: ident -> cycle cost
|
||||||
|
--- @return ComponentMetaMap
|
||||||
|
local function compute_components_metadata(components, latency)
|
||||||
|
local comp_by_name = {} ---@type table<string, Component>
|
||||||
|
for _, cc in ipairs(components) do comp_by_name[cc.name] = cc end ---@type integer, Component
|
||||||
|
local cache = {} ---@type ComponentMetaMap
|
||||||
|
local out = {} ---@type ComponentMetaMap
|
||||||
|
for _, c in ipairs(components) do ---@type integer, Component
|
||||||
|
out[c.name] = component_meta_rec(c.name, comp_by_name, latency, cache)
|
||||||
|
end
|
||||||
|
return out
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Per-component emit logic
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Split a (possibly multi-line) comment into per-line entries.
|
||||||
|
--- Hand-rolled (no regex patterns used).
|
||||||
|
--- @param s string
|
||||||
|
--- @return string[]
|
||||||
|
local function split_comment_lines(s)
|
||||||
|
local out = {} ---@type string[]
|
||||||
|
local pos = 1 ---@type integer
|
||||||
|
local s_len = #s ---@type integer
|
||||||
|
while pos <= s_len do
|
||||||
|
local nl = s:find("\n", pos, true) ---@type integer|nil
|
||||||
|
if not nl then
|
||||||
|
out[#out + 1] = s:sub(pos)
|
||||||
|
break
|
||||||
|
end
|
||||||
|
out[#out + 1] = s:sub(pos, nl - 1)
|
||||||
|
pos = nl + 1
|
||||||
|
end
|
||||||
|
return out
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Determine the macro signature: function-args list (function form) or variadic-ignored (bare form).
|
||||||
|
--- For `MipsAtomComp_Proc_` components, the leading `ab` (atom-builder) arg is dropped:
|
||||||
|
--- the generated `mac_<name>` macros are inline-expansion aliases for baked atoms; their bodies don't reference `ab`
|
||||||
|
--- (the builder is only consumed by the procedural `atombuilder_unroll` line that `MipsAtomComp_Proc_` appends after the body).
|
||||||
|
--- Inline callers therefore don't need to thread a builder context.
|
||||||
|
--- @param args_str string|nil
|
||||||
|
--- @return string
|
||||||
|
local function signature_from_args(args_str)
|
||||||
|
local names = formal_arg_names(args_str) ---@type string[]|nil
|
||||||
|
if names then
|
||||||
|
return table.concat(names, ", ")
|
||||||
|
end
|
||||||
|
return "..."
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Strip the trailing `" \"` (space + backslash) line continuation from the last body line.
|
||||||
|
--- The last 2 chars are always that pair.
|
||||||
|
--- @param lines string[]
|
||||||
|
--- @return nil
|
||||||
|
local function strip_trailing_continuation(lines)
|
||||||
|
local last = lines[#lines] ---@type string
|
||||||
|
if last:sub(-2) == " \\" then
|
||||||
|
lines[#lines] = last:sub(1, -3)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Classify a token as a "pure delay marker token" (a delay-marker identifier with no following instruction — only whitespace and/or block comments).
|
||||||
|
--- Examples that match:
|
||||||
|
--- * `GteDelay_` → marker alone
|
||||||
|
--- * `GteDelay_ /* RT diagonal: D1 = a.x... */` → marker + block comment
|
||||||
|
--- * `GteDelay_ /* RT diagonal: ... */\n\t` → marker + comment + trailing whitespace
|
||||||
|
--- Examples that DO NOT match (these contain a real instruction after the marker and must be preserved verbatim so the instruction still gets emitted):
|
||||||
|
--- * `GteDelay_ nop2`
|
||||||
|
--- * `GteDelay_ add_si(r.dst_ptr, r.scratch, dst_offset)`
|
||||||
|
---
|
||||||
|
--- Why this classification matters: the metaprogram emits tokens separated by `,` and joins them with `\<newline>` line continuations. After C preprocessor
|
||||||
|
--- phase 2 (line splicing), the macro body collapses to a single logical line.
|
||||||
|
--- Each delay-marker identifier expands to empty (its definition `#define GteDelay_ // ...` consumes the `//` line comment during preprocessing
|
||||||
|
--- of the definition itself, leaving an empty replacement list).
|
||||||
|
--- When a token is purely a delay marker with only a trailing comment, the `,` the metaprogram normally adds before
|
||||||
|
--- each token-after-the-first brackets empty content and produces the syntax error `,,` (`expected expression before ',' token`) at C compile.
|
||||||
|
--- The metaprogram therefore emits such tokens WITHOUT the leading `,` (see `token_skips_leading_comma`) —
|
||||||
|
--- but the marker + trailing comment are still emitted verbatim so the annotation is preserved in `gen/macs.h`.
|
||||||
|
--- @param tok string -- a single token from split_top_level_commas (already trimmed at the start, may contain trailing whitespace + block comment)
|
||||||
|
--- @return boolean
|
||||||
|
local function is_pure_delay_marker_token(tok)
|
||||||
|
local markers = duffle.DELAY_MARKERS ---@type table<string, boolean> -- bag: delay-marker ident -> true
|
||||||
|
if type(markers) ~= "table" then return false end
|
||||||
|
|
||||||
|
-- Identify a leading delay-marker identifier (e.g. `GteDelay_`).
|
||||||
|
local ident_end = 1 ---@type integer
|
||||||
|
while ident_end <= #tok do
|
||||||
|
local ch = tok:sub(ident_end, ident_end) ---@type string
|
||||||
|
if ch:match("[%w_]") then
|
||||||
|
ident_end = ident_end + 1
|
||||||
|
else
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
local ident = tok:sub(1, ident_end - 1) ---@type string
|
||||||
|
if not markers[ident] then return false end
|
||||||
|
|
||||||
|
-- Walk the remainder: only whitespace and block comments are allowed.
|
||||||
|
local scan = ident_end ---@type integer
|
||||||
|
while scan <= #tok do
|
||||||
|
local ch = tok:sub(scan, scan) ---@type string
|
||||||
|
if ch:match("%s") then
|
||||||
|
scan = scan + 1
|
||||||
|
elseif ch == "/" and tok:sub(scan + 1, scan + 1) == "*" then
|
||||||
|
local close = tok:find("*/", scan + 2, true) ---@type integer|nil
|
||||||
|
if not close then return false end
|
||||||
|
scan = close + 2
|
||||||
|
else
|
||||||
|
-- Non-whitespace, non-block-comment content: a real instruction
|
||||||
|
-- follows the marker (e.g. `GteDelay_ nop2`); keep this token intact.
|
||||||
|
return false
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return true
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Classify a token's "leading comma requirement".
|
||||||
|
--- Pure delay-marker tokens (`GteDelay_` / `LdSlot_` / `BdSlot_` / `DmaSlot_`
|
||||||
|
--- followed by whitespace + optional block comment and NOTHING ELSE) expand
|
||||||
|
--- to empty at C preprocessor time. Emitting them WITHOUT the leading `,`
|
||||||
|
--- separator that the metaprogram normally adds before each token after the
|
||||||
|
--- first keeps exactly one `,` between the surrounding real expressions in the spliced macro body:
|
||||||
|
--- * before this rule: `<tok1> ,\t<gdelay> ,\t<tok3>` → after expansion
|
||||||
|
--- `<tok1> , /* comment */ , <tok3>` → `,,` syntax error.
|
||||||
|
--- * after this rule: `<tok1> \t<gdelay> ,\t<tok3>` → after expansion
|
||||||
|
--- `<tok1> /* comment */ , <tok3>` → `<tok1>, <tok3>` — valid.
|
||||||
|
---
|
||||||
|
--- Tokens like `GteDelay_ nop2` keep the leading `,`
|
||||||
|
--- (the marker is followed by a real instruction, so the marker + instruction together need the separator on the LEFT to land between two real expressions).
|
||||||
|
--- @param tok string
|
||||||
|
--- @return boolean -- true if the token needs NO leading `,` separator.
|
||||||
|
local function token_skips_leading_comma(tok)
|
||||||
|
return is_pure_delay_marker_token(tok)
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Emit the `#define mac_X(sig) \<newline>\t<tok1> \<newline>,\t<tok2> ...` block.
|
||||||
|
--- Converts `//` line comments to `/* */` block comments in each token so they don't break the C macro `\` line continuations.
|
||||||
|
---
|
||||||
|
--- Pure delay-marker tokens (`GteDelay_` / `LdSlot_` / `BdSlot_` / `DmaSlot_` with only a trailing block comment, no real instruction) are emitted WITHOUT a leading `,` separator;
|
||||||
|
--- the annotation IS preserved in the generated header
|
||||||
|
--- (so the comment + marker remain visible to anyone reading `gen/macs.h`), but the C preprocessor expands the marker to empty, so leaving the `,`
|
||||||
|
--- separator out is what stops the `,,` syntax error. See `token_skips_leading_comma` for the contract.
|
||||||
|
--- @param lines string[]
|
||||||
|
--- @param c Component
|
||||||
|
--- @param sig string
|
||||||
|
--- @param tokens string[]
|
||||||
|
--- @return nil
|
||||||
|
local function emit_macro_body(lines, c, sig, tokens)
|
||||||
|
for tok_idx = 1, #tokens do ---@type integer
|
||||||
|
tokens[tok_idx] = convert_line_comments_to_block(tokens[tok_idx])
|
||||||
|
end
|
||||||
|
if #tokens == 0 then return end
|
||||||
|
lines[#lines + 1] = "#define mac_" .. c.name .. "(" .. sig .. ") \\"
|
||||||
|
lines[#lines + 1] = "\t" .. tokens[1] .. " \\"
|
||||||
|
for tok_idx = 2, #tokens do ---@type integer
|
||||||
|
local sep = token_skips_leading_comma(tokens[tok_idx]) and "\t" or ",\t" ---@type string
|
||||||
|
lines[#lines + 1] = sep .. tokens[tok_idx] .. " \\"
|
||||||
|
end
|
||||||
|
strip_trailing_continuation(lines)
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Build the list of lines for one component
|
||||||
|
--- (signature comment, `#define mac_X(...)` line with backslash-continued tokens, then `WORD_COUNT(mac_X, N)` entry).
|
||||||
|
--- For skipped components, a `/* atom_dbg_skip */` marker comment is emitted immediately before the authored comment block.
|
||||||
|
--- The marker is a single line, the comment comes next, and the `#define` line follows. The `debug_skip` stamp is scanner-owned
|
||||||
|
--- (`a.debug_skip == true` on the declaration record); the components pass projects it directly.
|
||||||
|
--- @param c Component
|
||||||
|
--- @param counts table<string, integer> -- bag: bare component name -> word count
|
||||||
|
--- @return string[] -- list of lines for this component
|
||||||
|
local function build_component_lines(c, counts)
|
||||||
|
local lines = {} ---@type string[]
|
||||||
|
|
||||||
|
-- Marker comment: emitted once for every skipped component.
|
||||||
|
-- The marker is scanner-owned (declared by `atom_dbg_skip` immediately before the declaration in the source);
|
||||||
|
-- This pass projects `c.debug_skip` and emits the marker as a generated comment.
|
||||||
|
if c.debug_skip then
|
||||||
|
lines[#lines + 1] = "/* atom_dbg_skip */"
|
||||||
|
end
|
||||||
|
|
||||||
|
if c.comment and c.comment ~= "" then
|
||||||
|
for _, line in ipairs(split_comment_lines(c.comment)) do ---@type integer, string
|
||||||
|
lines[#lines + 1] = line
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
local tokens = duffle.split_top_level_commas(c.body) ---@type string[]
|
||||||
|
for i = 1, #tokens do tokens[i] = duffle.trim(tokens[i]) end ---@type integer
|
||||||
|
local sig = signature_from_args(c.args) ---@type string
|
||||||
|
-- Direct lookup against the per-source precomputed `counts` table (built once by count_all_components).
|
||||||
|
local n = counts[c.name] ---@type integer
|
||||||
|
|
||||||
|
if n > 0 then
|
||||||
|
emit_macro_body(lines, c, sig, tokens)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Emit the WORD_COUNT(mac_<X>, N) entry.
|
||||||
|
lines[#lines + 1] = "WORD_COUNT(mac_" .. c.name .. ", " .. n .. ")"
|
||||||
|
lines[#lines + 1] = ""
|
||||||
|
|
||||||
|
return lines
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Per-source emit logic
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Build the boilerplate header lines (the `#ifdef INTELLISENSE_DIRECTIVES` block,
|
||||||
|
--- the `// Auto-generated` comment, the `// Source:` line, and the self-contained `WORD_COUNT` macro definition).
|
||||||
|
--- @param dir string -- Absolute source directory
|
||||||
|
--- @param sources SourceFile[] -- Sources contributing to this directory (for the header comment)
|
||||||
|
--- @return string[]
|
||||||
|
local function header_boilerplate(dir, sources)
|
||||||
|
local source_lines = { "// Directory: " .. duffle.to_absolute_path(dir) .. "/" } ---@type string[]
|
||||||
|
for _, src in ipairs(sources) do ---@type integer, SourceFile
|
||||||
|
source_lines[#source_lines + 1] = "// source: " .. duffle.to_absolute_path(src.path)
|
||||||
|
end
|
||||||
|
local source_blob = table.concat(source_lines, "\n") ---@type string
|
||||||
|
return {
|
||||||
|
-- #pragma once wrapped in #ifdef INTELLISENSE_DIRECTIVES, matching the convention in lottes_tape.h.
|
||||||
|
-- The build does manual unity includes (the user controls include order), so the pragma is only active for IDE/tooling.
|
||||||
|
"#ifdef INTELLISENSE_DIRECTIVES",
|
||||||
|
"#pragma once",
|
||||||
|
"#endif",
|
||||||
|
"// Auto-generated by ps1_meta.lua — DO NOT EDIT",
|
||||||
|
source_blob,
|
||||||
|
"// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)",
|
||||||
|
"",
|
||||||
|
-- Self-contained: define WORD_COUNT if not already defined.
|
||||||
|
-- We use the same definition here so the auto-generated entries below expand
|
||||||
|
-- to compile-time constants whether the metadata file is included first or not.
|
||||||
|
"#ifndef WORD_COUNT",
|
||||||
|
"#define WORD_COUNT(name, count) enum { words_##name = (count) };",
|
||||||
|
"#endif",
|
||||||
|
"",
|
||||||
|
}
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Compute the per-directory output path for `.macs.h`.
|
||||||
|
--- e.g. any source in `code/duffle/` produces `code/duffle/gen/macs.h` regardless of source filename.
|
||||||
|
--- The directory name is the namespace; the filename does not repeat it.
|
||||||
|
--- @param dir string -- Absolute source directory
|
||||||
|
--- @return string -- Output directory
|
||||||
|
--- @return string -- Full output path
|
||||||
|
local function compute_macs_h_path(dir)
|
||||||
|
local out_dir = dir .. "/" .. GEN_SUBDIR ---@type string
|
||||||
|
local out_path = out_dir .. "/" .. MACS_FILENAME ---@type string
|
||||||
|
return out_dir, out_path
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Emit a per-directory `.macs.h` header with the aggregated `mac_X` macros + `WORD_COUNT` entries.
|
||||||
|
--- Writes in BINARY mode so LF line endings are preserved (the git blob is LF; Windows text-mode would emit CRLF and break the byte-identical diff).
|
||||||
|
--- @param ctx PassCtx
|
||||||
|
--- @param dir string -- Absolute source directory
|
||||||
|
--- @param sources SourceFile[] -- Sources contributing to this directory (for the header comment)
|
||||||
|
--- @param components Component[] -- Aggregated components from all sources in this directory
|
||||||
|
--- @param counts table<string, integer> -- bag: bare component name -> word count
|
||||||
|
--- @return string|nil -- Path to the written file (nil if no components)
|
||||||
|
local function emit_component_macros_h(ctx, dir, sources, components, counts)
|
||||||
|
if #components == 0 then return nil end
|
||||||
|
local out_dir, out_path = compute_macs_h_path(dir) ---@type string, string
|
||||||
|
local lines = header_boilerplate(dir, sources) ---@type string[]
|
||||||
|
|
||||||
|
for _, c in ipairs(components) do ---@type integer, Component
|
||||||
|
for _, l in ipairs(build_component_lines(c, counts)) do ---@type integer, string
|
||||||
|
lines[#lines + 1] = l
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
local content = table.concat(lines, "\n") .. "\n" ---@type string
|
||||||
|
duffle.ensure_dir(out_dir)
|
||||||
|
duffle.write_file_lf(out_path, content)
|
||||||
|
print(string.format(" -> %s", out_path))
|
||||||
|
return out_path
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Pass entry
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- (internal) Extend `corpus.word_counts` with this source's component macros so offsets sees them without re-reading the file.
|
||||||
|
--- First declaration wins: a later caller's count is dropped (the existing entry from the first source is preserved).
|
||||||
|
--- @param corpus Corpus
|
||||||
|
--- @param components Component[]
|
||||||
|
--- @param counts table<string, integer> -- bag: bare component name -> word count
|
||||||
|
--- @return nil
|
||||||
|
local function update_canonical_word_counts(corpus, components, counts)
|
||||||
|
local wc = corpus.word_counts ---@type WordCounts
|
||||||
|
for _, c in ipairs(components) do ---@type integer, Component
|
||||||
|
local key = "mac_" .. c.name ---@type string
|
||||||
|
if wc[key] == nil then
|
||||||
|
wc[key] = counts[c.name]
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
--- (internal) Populate `corpus.components` with this source's one component row per bare name.
|
||||||
|
--- First declaration wins; later declarations of the same bare name are dropped and recorded as a collision via `corpus.collisions` (kind = "component").
|
||||||
|
--- The pass does NOT write to `ctx.shared.components`.
|
||||||
|
--- No parallel skip map is built here; consumers that need the per-component skip state read `corpus.components[name].debug_skip` directly.
|
||||||
|
--- The `cycle_cost` + `gp0_contrib` fields are populated from `metadata[c.name]` (computed by `compute_components_metadata` against the original `MipsAtomComp_` body).
|
||||||
|
--- @param corpus Corpus
|
||||||
|
--- @param src SourceFile
|
||||||
|
--- @param components Component[]
|
||||||
|
--- @param metadata ComponentMetaMap
|
||||||
|
--- @param scan SourceScan
|
||||||
|
--- @return nil
|
||||||
|
local function update_canonical_components(corpus, src, components, metadata, scan)
|
||||||
|
local rel_path = src.path:gsub("\\", "/") ---@type string
|
||||||
|
local line_of = scan and scan.line_of ---@type (fun(pos: integer): integer)|nil
|
||||||
|
for _, c in ipairs(components) do ---@type integer, Component
|
||||||
|
-- Keyed by bare name (e.g. `yield`, `load_tri_indices`).
|
||||||
|
-- The atoms_source_map pass looks up components by bare name from the corpus;
|
||||||
|
-- `mac_` prefix lives at the call-site identifier and is stripped before lookup.
|
||||||
|
local m = metadata and metadata[c.name] or nil ---@type ComponentMeta|nil
|
||||||
|
if corpus.components[c.name] == nil then
|
||||||
|
c.path = rel_path
|
||||||
|
c.source = src.path
|
||||||
|
c.line_of = line_of
|
||||||
|
c.kind = c.kind or "comp_bare"
|
||||||
|
c.debug_skip = c.debug_skip == true
|
||||||
|
c.cycle_cost = m and m.cycle_cost or nil
|
||||||
|
c.gp0_contrib = m and m.gp0_contrib or nil
|
||||||
|
corpus.components[c.name] = c
|
||||||
|
else
|
||||||
|
-- A second declaration of the same bare name: record a typed collision so static-analysis + the report can surface it.
|
||||||
|
-- Identical-shape declarations (same path + line) reuse the first-wins entry without a collision record.
|
||||||
|
local existing = corpus.components[c.name] ---@type Component
|
||||||
|
if existing.path ~= rel_path or existing.line ~= c.line then
|
||||||
|
local kind = c.kind or "comp_bare" ---@type string
|
||||||
|
local first_kind = existing.kind or "comp_bare" ---@type string
|
||||||
|
corpus.collisions[#corpus.collisions + 1] = {
|
||||||
|
kind = "component",
|
||||||
|
name = c.name,
|
||||||
|
first_site = { path = existing.path, line = existing.line },
|
||||||
|
conflicting_site = { path = rel_path, line = c.line },
|
||||||
|
first_shape = "kind=" .. first_kind,
|
||||||
|
conflicting_shape = "kind=" .. kind,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
--- @param ctx PassCtx
|
||||||
|
--- @return PassResult
|
||||||
|
function M.run(ctx)
|
||||||
|
local outputs = {} ---@type MacsOutput[]
|
||||||
|
local errors = {} ---@type Finding[]
|
||||||
|
local warnings = {} ---@type Finding[]
|
||||||
|
|
||||||
|
-- Corpus ownership gate.
|
||||||
|
local corpus = ctx.shared and ctx.shared.corpus ---@type Corpus|nil
|
||||||
|
if type(corpus) ~= "table" then
|
||||||
|
error("components.run requires ctx.shared.corpus.", 0)
|
||||||
|
end
|
||||||
|
if type(corpus.source_order) ~= "table" then
|
||||||
|
error("components.run requires ctx.shared.corpus.source_order.", 0)
|
||||||
|
end
|
||||||
|
if type(corpus.word_counts) ~= "table" then
|
||||||
|
error("components.run requires ctx.shared.corpus.word_counts; "
|
||||||
|
.. "word_count_eval.run must run before components.run "
|
||||||
|
.. "(see PASSES deps).", 0)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Projection ownership:
|
||||||
|
-- * `corpus.word_counts["mac_"..name]` — current component count
|
||||||
|
-- * `corpus.components[name]` — one row: body, line_of, source, cost
|
||||||
|
-- The pass writes to the corpus only; consumers read from the corpus directly.
|
||||||
|
|
||||||
|
-- Per-directory aggregation: every source in the same directory contributes to one `gen/macs.h`.
|
||||||
|
-- The directory itself is the namespace. `corpus.sources_by_dir` preserves source-order within each bucket (matches `corpus.source_order`).
|
||||||
|
local sources_by_dir = corpus.sources_by_dir or duffle.group_sources_by_dir(corpus.source_order) ---@type table<string, SourceFile[]>
|
||||||
|
for dir, sources in pairs(sources_by_dir) do ---@type string, SourceFile[]
|
||||||
|
-- Aggregate components from every source in this directory.
|
||||||
|
-- `project_components` returns nil for sources with no `MipsAtomComp_` declarations; we skip those.
|
||||||
|
local aggregated_components = {} ---@type Component[]
|
||||||
|
local metadata_per_source = {} ---@type table<SourceFile, ComponentMetaMap>
|
||||||
|
for _, src in ipairs(sources) do ---@type integer, SourceFile
|
||||||
|
local per_source = project_components(src.text, src.scan) or {} ---@type Component[]
|
||||||
|
for _, c in ipairs(per_source) do ---@type integer, Component
|
||||||
|
aggregated_components[#aggregated_components + 1] = c
|
||||||
|
end
|
||||||
|
if #per_source > 0 then
|
||||||
|
metadata_per_source[src] = compute_components_metadata(per_source, {})
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if #aggregated_components > 0 then
|
||||||
|
-- Compute word counts across the aggregated set. `corpus.word_counts` carries the
|
||||||
|
-- same-source + prior-directory entries so the recursive lookup sees both.
|
||||||
|
local counts = count_all_components(aggregated_components, corpus.word_counts) ---@type table<string, integer> -- bag: bare name -> word count
|
||||||
|
local macs_path = emit_component_macros_h(ctx, dir, sources, aggregated_components, counts) ---@type string|nil
|
||||||
|
if macs_path then
|
||||||
|
outputs[#outputs + 1] = { macs_h = macs_path }
|
||||||
|
-- Populate the projections AFTER disk emission (byte-identical `.macs.h` contract).
|
||||||
|
update_canonical_word_counts(corpus, aggregated_components, counts)
|
||||||
|
for _, src in ipairs(sources) do ---@type integer, SourceFile
|
||||||
|
local per_source = project_components(src.text, src.scan) or {} ---@type Component[]
|
||||||
|
if #per_source > 0 then
|
||||||
|
update_canonical_components(corpus, src, per_source, metadata_per_source[src], src.scan)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return { outputs = outputs, errors = errors, warnings = warnings }
|
||||||
|
end
|
||||||
|
|
||||||
|
return M
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,366 @@
|
|||||||
|
--- passes/emission_model.lua: Per-atom emission projection.
|
||||||
|
---
|
||||||
|
--- The `emission-model` pass owns `atom.paths`, the canonical per-atom mutable surface for atoms and raw atoms with bodies in `ctx.shared.corpus.source_order`.
|
||||||
|
--- For each atom, the pass invokes `duffle.project_emission(body_text, components, word_counts, components)`.
|
||||||
|
--- It stores the ordered `items` stream plus the dense `word_events` / `markers` / `invocations` views on `atom.paths`.
|
||||||
|
---
|
||||||
|
--- Public boundary:
|
||||||
|
--- * `M.run(ctx)` is the only entry point.
|
||||||
|
--- * The pass returns `{outputs = {}, errors = ..., warnings = ...}`.
|
||||||
|
--- Pass kind = `validation`. Findings record on the result; the orchestrator does not exit non-zero.
|
||||||
|
---
|
||||||
|
--- Source-order discipline:
|
||||||
|
--- * `corpus.source_order` sets the source-record order.
|
||||||
|
--- * Within each source, the pass visits `src.scan.atoms` and `src.scan.raw_atoms` in declaration order.
|
||||||
|
---
|
||||||
|
--- Per-atom projection fields on `atom.paths`:
|
||||||
|
--- `tokens`, `line_in_body`, `items`, `word_events`, `markers`, `invocations`, `errors`, `warnings`.
|
||||||
|
--- The construction walk appends `items` and derives each dense view from that ordered stream.
|
||||||
|
---
|
||||||
|
--- Component expansion and construction validation:
|
||||||
|
--- * known `mac_X(...)` calls recursively expand component bodies;
|
||||||
|
--- * invocation records retain monotonic IDs, parent IDs, immediate call text, and the immutable outermost root call text;
|
||||||
|
--- * invocation construction stamps `debug_skip` from `corpus.components[name].debug_skip` at the construction site (no second pass, no source parse, no parallel lookup);
|
||||||
|
--- * component cycles close balanced invocation boundaries and emit a `cycle` construction error at the recursive edge;
|
||||||
|
--- * declared-vs-measured component word counts emit `count_mismatch` construction errors; opaque uncounted macros emit warnings.
|
||||||
|
---
|
||||||
|
--- `passes.scan_source` strips its private `_code_macros` / `_code_macro_bodies` tables before this pass runs.
|
||||||
|
|
||||||
|
--- @class BodyToken
|
||||||
|
--- @field tok string
|
||||||
|
--- @field rel integer
|
||||||
|
|
||||||
|
--- @class EmissionItem
|
||||||
|
--- @field kind string
|
||||||
|
--- @field encoder string|nil
|
||||||
|
--- @field args string[]|nil
|
||||||
|
--- @field i integer|nil
|
||||||
|
--- @field word_count integer|nil
|
||||||
|
--- @field line integer|nil
|
||||||
|
--- @field call_text string|nil
|
||||||
|
--- @field root_call_text string|nil
|
||||||
|
--- @field invocation_ids integer[]|nil
|
||||||
|
--- @field outermost_invocation_id integer|nil
|
||||||
|
--- @field gpr_keys string[]|nil
|
||||||
|
--- @field ident string|nil
|
||||||
|
--- @field isa_kind string|nil
|
||||||
|
--- @field nop_words integer|nil
|
||||||
|
--- @field is_yield boolean|nil
|
||||||
|
--- @field is_load boolean|nil
|
||||||
|
--- @field is_branch boolean|nil
|
||||||
|
--- @field is_unconditional_jump boolean|nil
|
||||||
|
--- @field is_terminal_jump boolean|nil
|
||||||
|
--- @field gp0_shape string|nil
|
||||||
|
--- @field name string|nil
|
||||||
|
--- @field target string|nil
|
||||||
|
--- @field word_index integer|nil
|
||||||
|
--- @field consuming_encoder string|nil
|
||||||
|
--- @field consuming_arg_pos integer|nil
|
||||||
|
--- @field invocation_id integer|nil
|
||||||
|
|
||||||
|
--- @class WordEvent
|
||||||
|
--- @field i integer
|
||||||
|
--- @field encoder string
|
||||||
|
--- @field args string[]
|
||||||
|
--- @field def_path string
|
||||||
|
--- @field def_line integer
|
||||||
|
--- @field call_text string|nil
|
||||||
|
--- @field root_call_text string|nil
|
||||||
|
--- @field invocation_ids integer[]
|
||||||
|
--- @field outermost_invocation_id integer
|
||||||
|
--- @field word_count integer
|
||||||
|
--- @field gpr_keys string[]|nil
|
||||||
|
--- @field ident string
|
||||||
|
--- @field kind string
|
||||||
|
--- @field nop_words integer
|
||||||
|
--- @field is_yield boolean
|
||||||
|
--- @field is_load boolean
|
||||||
|
--- @field is_branch boolean
|
||||||
|
--- @field is_unconditional_jump boolean
|
||||||
|
--- @field is_terminal_jump boolean
|
||||||
|
--- @field gp0_shape string|nil
|
||||||
|
--- @field body_line integer|nil
|
||||||
|
--- @field call_line integer|nil
|
||||||
|
--- @field call_path string|nil
|
||||||
|
|
||||||
|
--- @class EmissionMarker
|
||||||
|
--- @field kind string
|
||||||
|
--- @field name string
|
||||||
|
--- @field line integer
|
||||||
|
--- @field word_index integer
|
||||||
|
--- @field target string|nil
|
||||||
|
--- @field consuming_encoder string|nil
|
||||||
|
--- @field consuming_arg_pos integer|nil
|
||||||
|
|
||||||
|
-- Finding: see ps1_meta.lua
|
||||||
|
|
||||||
|
--- @class AtomPaths
|
||||||
|
--- @field tokens BodyToken[]
|
||||||
|
--- @field line_in_body table<integer, integer> -- bag: body byte offset -> 1-based line
|
||||||
|
--- @field items EmissionItem[]
|
||||||
|
--- @field word_events WordEvent[]
|
||||||
|
--- @field markers EmissionMarker[]
|
||||||
|
--- @field invocations InvocationRecord[]
|
||||||
|
--- @field errors Finding[]
|
||||||
|
--- @field warnings Finding[]
|
||||||
|
|
||||||
|
--- @class EmissionModelPass
|
||||||
|
--- @field run fun(ctx: PassCtx): PassResult
|
||||||
|
|
||||||
|
local M = {} ---@type EmissionModelPass
|
||||||
|
|
||||||
|
-- ─────────────────────────────────────────────────────────────────────────
|
||||||
|
-- Bootstrap: load `duffle_paths.lua` via debug.getinfo so the module works standalone (run as `luajit passes/emission_model.lua`) and when require'd from the orchestrator.
|
||||||
|
-- ─────────────────────────────────────────────────────────────────────────
|
||||||
|
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" ---@type string
|
||||||
|
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") ---@type DuffleExport
|
||||||
|
|
||||||
|
-- ─────────────────────────────────────────────────────────────────────────
|
||||||
|
-- Helpers
|
||||||
|
-- ─────────────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
-- Convert the recursive walk's body-relative line numbers into physical source lines once.
|
||||||
|
-- The walker builds `line_of` from `body_text` and stamps body-relative line numbers (1..N) into `item.line` and `invocation.call_line`.
|
||||||
|
-- This function converts those values to physical source lines at the close site with the forwarded source `line_of` closure.
|
||||||
|
-- `call_line` discipline:
|
||||||
|
-- * ROOT invocations (`inv.parent_id == 0`) receive body-relative `call_line` values directly from `M.LineIndex(body_text)` in the walker.
|
||||||
|
-- The source `line_of` closure supplies physical lines at the close site, so this function converts each root value exactly once.
|
||||||
|
-- * INNER invocations (`inv.parent_id ~= 0`) receive physical `call_line` values directly from the COMPONENT's `line_of` in the walker.
|
||||||
|
-- Recursive descent forwards that closure through `corpus.components[name].line_of`; those values arrive physical and remain unchanged.
|
||||||
|
--
|
||||||
|
-- After this function, every `inv.call_line` is physical. DWARF and provenance output read it directly.
|
||||||
|
-- The word-event loop forwards the already-physical `outer_inv.call_line` into `we.call_line` for words inside an invocation.
|
||||||
|
--- @param projection EmissionProjection
|
||||||
|
--- @param atom_record AtomEntry
|
||||||
|
--- @param src SourceFile
|
||||||
|
--- @param corpus Corpus
|
||||||
|
--- @return nil
|
||||||
|
local function stamp_root_provenance(projection, atom_record, src, corpus)
|
||||||
|
local root_line_of = src.scan and src.scan.line_of ---@type (fun(pos: integer): integer)|nil
|
||||||
|
assert(type(root_line_of) == "function"
|
||||||
|
, "emission_model: src.scan.line_of is required (canonical LineIndex closure over the source text) to stamp physical provenance")
|
||||||
|
assert(type(atom_record.body_off) == "number"
|
||||||
|
, "emission_model: atom_record.body_off (byte offset of the body's first byte in source) is required to derive `root_body_line`. The scanner must populate body_off for every atom record.")
|
||||||
|
-- `root_body_line` is the physical source line of the ATOM HEADER byte containing the opening `{`; that byte is one byte BEFORE `atom_record.body_off`.
|
||||||
|
-- The walker assigns line 2 to the body's first content line because line 1 is the trailing `\n` after `{`. Body-text line k therefore maps to `root_body_line + (k - 1)`.
|
||||||
|
-- `body_off - 1` points at the opening `{`, whose line index identifies the header line. `body_off` points after `{` and would shift every word row forward by one line.
|
||||||
|
local root_body_line = root_line_of(atom_record.body_off - 1) or atom_record.line or 0 ---@type integer
|
||||||
|
local components = corpus.components or {} ---@type table<string, Component>
|
||||||
|
local word_items = {} ---@type EmissionItem[]
|
||||||
|
|
||||||
|
for _, item in ipairs(projection.items) do ---@type integer, EmissionItem
|
||||||
|
if item.kind == "word" then word_items[#word_items + 1] = item end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Resolve one word's physical body line, where the byte containing that word appears in source.
|
||||||
|
-- * Component expansions carry `invocation_ids`; the component's full-file `line_of` leaves `item.line` physical.
|
||||||
|
-- * Raw tokens in the root atom body carry an empty `invocation_ids` list and a body-relative `item.line`; convert them here.
|
||||||
|
--- @param event WordEvent
|
||||||
|
--- @param item EmissionItem
|
||||||
|
--- @return integer
|
||||||
|
local function body_line_for(event, item)
|
||||||
|
local ids = event.invocation_ids or {} ---@type integer[]
|
||||||
|
-- The innermost open invocation identifies which line index the walker used.
|
||||||
|
-- A component `line_of` makes `item.line` physical; the atom's `body_text` line index makes it body-relative.
|
||||||
|
if ids and #ids > 0 then
|
||||||
|
local inner_id = ids[#ids] ---@type integer
|
||||||
|
local inner_inv = inner_id and projection.invocations[inner_id] ---@type InvocationRecord|nil
|
||||||
|
if inner_inv then
|
||||||
|
local component = components[inner_inv.component_name] ---@type Component|nil
|
||||||
|
if component and component.line_of then
|
||||||
|
-- Walker used `comp.line_of`, which is the source's physical LineIndex. item.line is already physical.
|
||||||
|
return item.line or 0
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
-- RAW root-body word: item.line is body-text's 1-based line number (the first content line is line 2 because line 1 is the trailing `\n` after `{`).
|
||||||
|
-- Convert body-text-relative → physical using `root_body_line + (item.line - 1)`.
|
||||||
|
return (root_body_line or 0) + (item.line or 1) - 1
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Stamp the root source path onto invocation records whose `call_path` the walker left empty.
|
||||||
|
-- The walker passes `body_entry.source` to `emit_invoke_begin`; `M.project_emission` creates the root `body_entry` with source `""`, leaving its `call_path` empty.
|
||||||
|
-- This stamp gives every invocation a physical `call_path` matching `passes/atoms_source_map.lua`'s in-memory provenance projection.
|
||||||
|
local root_path = src.path or "" ---@type string
|
||||||
|
for _, inv in ipairs(projection.invocations) do ---@type integer, InvocationRecord
|
||||||
|
if inv.call_path == nil or inv.call_path == "" then
|
||||||
|
inv.call_path = root_path
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Normalize `inv.call_line` to a physical source line.
|
||||||
|
-- * ROOT invocations (`parent_id == 0`) carry body-relative `call_line` values from `M.LineIndex(body_text)`; convert them once with `root_body_line`.
|
||||||
|
-- * INNER invocations (`parent_id ~= 0`) carry physical `call_line` values from the component's `line_of`; retain them unchanged.
|
||||||
|
for _, inv in ipairs(projection.invocations) do ---@type integer, InvocationRecord
|
||||||
|
if inv.parent_id == 0 then
|
||||||
|
inv.call_line = (root_body_line or 0) + (inv.call_line or 1) - 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Build `body_lines` for each invocation.
|
||||||
|
-- `atoms_source_map` and `dwarf_injection` read `inv.body_lines[k]` directly from the invocation record created here.
|
||||||
|
-- Component words already carry physical `item.line` values from the walker's COMPONENT line index, so `body_line_for` returns them unchanged.
|
||||||
|
for _, inv in ipairs(projection.invocations) do ---@type integer, InvocationRecord
|
||||||
|
local sw = inv.start_word ---@type integer
|
||||||
|
local ew = inv.end_word ---@type integer
|
||||||
|
local bls = {} ---@type integer[]
|
||||||
|
for i = sw, ew do ---@type integer
|
||||||
|
local it = projection.items and projection.items[i] ---@type EmissionItem|nil
|
||||||
|
if it and it.kind == "word" then
|
||||||
|
local fake_event = { invocation_ids = { inv.id } } ---@type WordEvent
|
||||||
|
bls[#bls + 1] = body_line_for(fake_event, it) or 0
|
||||||
|
end
|
||||||
|
end
|
||||||
|
inv.body_lines = bls
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Resolve each `word_event`'s physical `body_line` and `call_line`.
|
||||||
|
-- For words inside an invocation, `we.call_line` identifies the OUTER atom source line containing the `mac_X(...)` token that triggered expansion.
|
||||||
|
-- The root-invocation conversion above makes every `inv.call_line` physical; forward it directly and use each raw word's `body_line` as the fallback.
|
||||||
|
for index, we in ipairs(projection.word_events) do ---@type integer, WordEvent
|
||||||
|
local item = word_items[index] or {} ---@type EmissionItem
|
||||||
|
local body_line = body_line_for(we, item) ---@type integer
|
||||||
|
item.line = body_line
|
||||||
|
we.body_line = body_line
|
||||||
|
|
||||||
|
local call_line = body_line ---@type integer
|
||||||
|
local outer_id = we.outermost_invocation_id or 0 ---@type integer
|
||||||
|
local outer_inv = projection.invocations[outer_id] ---@type InvocationRecord|nil
|
||||||
|
if outer_inv then
|
||||||
|
-- `outer_inv.call_line` is physical after the conversion loop above, so use it directly.
|
||||||
|
call_line = outer_inv.call_line
|
||||||
|
end
|
||||||
|
we.call_line = call_line
|
||||||
|
|
||||||
|
if we.def_path == nil or we.def_path == "" then we.def_path = src.path or "" end
|
||||||
|
if we.def_line == nil or we.def_line == 0 then we.def_line = atom_record.line or 0 end
|
||||||
|
if we.call_path == nil or we.call_path == "" then we.call_path = src.path or "" end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Project one atom record into `atom.paths`.
|
||||||
|
-- Mutates the atom record in-place and returns the projection (for pass-level error/warning accumulation).
|
||||||
|
--- @param atom_record AtomEntry
|
||||||
|
--- @param src SourceFile
|
||||||
|
--- @param corpus Corpus
|
||||||
|
--- @return EmissionProjection
|
||||||
|
local function project_atom(atom_record, src, corpus)
|
||||||
|
local body = atom_record.body or "" ---@type string
|
||||||
|
local wc = corpus.word_counts or {} ---@type WordCounts
|
||||||
|
local comps = corpus.components or {} ---@type table<string, Component>
|
||||||
|
local schema = nil ---@type RegUseSchema|nil
|
||||||
|
if atom_record.reg_use_schema_name then
|
||||||
|
schema = corpus.reg_use_schemas and corpus.reg_use_schemas[atom_record.reg_use_schema_name]
|
||||||
|
end
|
||||||
|
-- That construction site stamps `invocation.debug_skip` while appending each record to `proj.invocations`.
|
||||||
|
local proj = duffle.project_emission(body, comps, wc, comps, { ---@type EmissionProjection
|
||||||
|
reg_use_schema = schema,
|
||||||
|
reg_use_param = atom_record.reg_use_param_name,
|
||||||
|
atom_name = atom_record.name,
|
||||||
|
schema_name = atom_record.reg_use_schema_name,
|
||||||
|
})
|
||||||
|
if atom_record.reg_use_schema_name and not schema then
|
||||||
|
proj.errors[#proj.errors + 1] = {
|
||||||
|
kind = "error",
|
||||||
|
check = "reguse_missing_schema",
|
||||||
|
msg = string.format("RegUse schema %q is missing", atom_record.reg_use_schema_name),
|
||||||
|
schema_name = atom_record.reg_use_schema_name,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
for _, err in ipairs(corpus.reg_use_errors or {}) do ---@type integer, RegUseError
|
||||||
|
if err.schema_name == atom_record.reg_use_schema_name then
|
||||||
|
proj.errors[#proj.errors + 1] = {
|
||||||
|
kind = "error",
|
||||||
|
check = err.kind,
|
||||||
|
line = err.line or err.source_line or 0,
|
||||||
|
msg = err.msg or "",
|
||||||
|
source = err.source or err.source_file,
|
||||||
|
schema_name = err.schema_name,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
local paths = { ---@type AtomPaths
|
||||||
|
tokens = atom_record.body_tokens or {},
|
||||||
|
line_in_body = duffle.build_body_line_index(body),
|
||||||
|
items = proj.items,
|
||||||
|
word_events = proj.word_events,
|
||||||
|
markers = proj.markers,
|
||||||
|
invocations = proj.invocations,
|
||||||
|
errors = proj.errors,
|
||||||
|
warnings = proj.warnings,
|
||||||
|
}
|
||||||
|
stamp_root_provenance(proj, atom_record, src, corpus)
|
||||||
|
atom_record.paths = paths
|
||||||
|
return proj
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ─────────────────────────────────────────────────────────────────────────
|
||||||
|
-- Run the emission-model pass.
|
||||||
|
-- ─────────────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
--- @param ctx PassCtx -- { shared = { corpus = ... }, out_root, ... }
|
||||||
|
--- @return PassResult
|
||||||
|
function M.run(ctx)
|
||||||
|
local outputs = {} ---@type PassOutputEntry[]
|
||||||
|
local errors = {} ---@type Finding[]
|
||||||
|
local warnings = {} ---@type Finding[]
|
||||||
|
|
||||||
|
local corpus = ctx and ctx.shared and ctx.shared.corpus ---@type Corpus|nil
|
||||||
|
if type(corpus) ~= "table" then error("emission_model: ctx.shared.corpus is required (canonical projection)", 0) end
|
||||||
|
if type(corpus.source_order) ~= "table" then error("emission_model: ctx.shared.corpus.source_order is required", 0) end
|
||||||
|
|
||||||
|
-- Project once, collect errors + warnings for one atom.
|
||||||
|
-- Kind must be one of: atom | atom_proc | raw_atom | comp_bare | comp_proc.
|
||||||
|
--- @param atom AtomEntry
|
||||||
|
--- @param src SourceFile
|
||||||
|
--- @return nil
|
||||||
|
local function process_atom(atom, src)
|
||||||
|
if not (atom and atom.body) then return end
|
||||||
|
local kind = atom.kind ---@type string
|
||||||
|
if kind ~= "atom" and kind ~= "atom_proc" and kind ~= "raw_atom" and kind ~= "comp_bare" and kind ~= "comp_proc" then
|
||||||
|
return
|
||||||
|
end
|
||||||
|
local proj = project_atom(atom, src, corpus) ---@type EmissionProjection
|
||||||
|
for _, e in ipairs(proj.errors) do ---@type integer, Finding
|
||||||
|
-- Finding.kind is severity. Finding.check holds the diagnostic code
|
||||||
|
-- (cycle / count_mismatch / unbalanced / reguse_*).
|
||||||
|
errors[#errors + 1] = {
|
||||||
|
kind = "error",
|
||||||
|
check = e.check,
|
||||||
|
line = e.line,
|
||||||
|
msg = e.msg,
|
||||||
|
source = e.source or src.path,
|
||||||
|
schema_name = e.schema_name,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
for _, w in ipairs(proj.warnings) do ---@type integer, Finding
|
||||||
|
warnings[#warnings + 1] = {
|
||||||
|
kind = "warning",
|
||||||
|
check = w.check,
|
||||||
|
line = w.line,
|
||||||
|
msg = w.msg,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Walk `corpus.source_order`; within each source, visit atoms followed by raw_atoms.
|
||||||
|
-- Recognized kinds (atom | atom_proc | raw_atom | comp_bare | comp_proc) each receive the atom.paths projection via duffle.project_emission.
|
||||||
|
-- Components are macros inlined into atom bodies; focused tests and isolated component analyses consume atom.paths directly.
|
||||||
|
for _, src in ipairs(corpus.source_order) do ---@type integer, SourceFile
|
||||||
|
local scan = src.scan or {} ---@type SourceScan
|
||||||
|
for _, atom in ipairs(scan.atoms or {}) do ---@type integer, AtomEntry
|
||||||
|
process_atom(atom, src)
|
||||||
|
end
|
||||||
|
for _, atom in ipairs(scan.raw_atoms or {}) do ---@type integer, AtomEntry
|
||||||
|
process_atom(atom, src)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return {
|
||||||
|
outputs = outputs,
|
||||||
|
errors = errors,
|
||||||
|
warnings = warnings,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
|
||||||
|
return M
|
||||||
@@ -0,0 +1,323 @@
|
|||||||
|
--- passes/offsets.lua — Branch-offset generator.
|
||||||
|
---
|
||||||
|
--- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`)
|
||||||
|
--- for `MipsAtom_(name)` and leftover `MipsCode code_*` declarations, computes the word offset
|
||||||
|
--- (ELF symbol is the C ident; raw `code_*` is leftover, not the atom rule)
|
||||||
|
--- from each `atom_offset(F, T)` marker to its target `atom_label(T)` declaration, and emits
|
||||||
|
--- `gen/offsets.h` with one `#define _atom_offset_F_T = N` per branch.
|
||||||
|
---
|
||||||
|
--- Per-directory aggregation: every source in the same directory contributes to the same `gen/offsets.h`.
|
||||||
|
--- The directory itself is the namespace; the filename does not repeat the module name.
|
||||||
|
---
|
||||||
|
--- The offset is `target_word - branch_word - 1` (the standard MIPS branch-immediate encoding: branch_offset = relative_pc_in_words - 1).
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Module-scope requires + package.path setup
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
-- Bootstrap: same as entry scripts. See `ps1_meta.lua` for the rationale.
|
||||||
|
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
||||||
|
-- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator.
|
||||||
|
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
|
||||||
|
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
||||||
|
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" ---@type string
|
||||||
|
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") ---@type DuffleExport
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Constants
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
-- Offset macro/enum naming prefixes (the emitted header uses these).
|
||||||
|
local OFFSET_MACRO_PREFIX = "_atom_offset_" ---@type string
|
||||||
|
local OFFSET_ENUM_PREFIX = "atom_offset_" ---@type string
|
||||||
|
|
||||||
|
-- Column width for the `#define _atom_offset_F_T = N` alignment.
|
||||||
|
local OFFSET_MACRO_COL = 44 ---@type integer
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Type declarations
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
-- SourceFile, PassCtx, PassResult: see ps1_meta.lua
|
||||||
|
|
||||||
|
--- @class BranchOffset
|
||||||
|
--- @field tag string -- Marker tag (e.g. "F" in `atom_offset(F, T)`)
|
||||||
|
--- @field target string -- Target label name (e.g. "T" in `atom_offset(F, T)`)
|
||||||
|
--- @field branch_word integer -- Branch word position within the atom body
|
||||||
|
--- @field offset integer -- Computed per consuming instruction (see `compute_offsets`)
|
||||||
|
--- @field consuming_encoder string|nil -- Instruction consuming the offset (e.g. "branch_le_zero", "jump", "call_addr")
|
||||||
|
--- @field consuming_arg_pos integer|nil -- 1-based arg position within the consuming instruction's arg list
|
||||||
|
|
||||||
|
--- @class AtomData
|
||||||
|
--- @field name string -- Atom name
|
||||||
|
--- @field total_words integer -- Total word count of the atom body
|
||||||
|
--- @field offsets BranchOffset[] -- Per-branch offset list
|
||||||
|
|
||||||
|
--- @class OffsetBranch
|
||||||
|
--- @field tag string
|
||||||
|
--- @field target string
|
||||||
|
--- @field branch_word integer
|
||||||
|
--- @field consuming_encoder string|nil
|
||||||
|
--- @field consuming_arg_pos integer|nil
|
||||||
|
--- @field line integer|nil
|
||||||
|
|
||||||
|
--- @class MarkerProjectState
|
||||||
|
--- @field labels table<string, integer> -- bag: label name -> word index
|
||||||
|
--- @field branches OffsetBranch[]
|
||||||
|
|
||||||
|
--- @class OffsetConst
|
||||||
|
--- @field macro_name string
|
||||||
|
--- @field enum_name string
|
||||||
|
--- @field value integer
|
||||||
|
|
||||||
|
--- @class OffsetOutput
|
||||||
|
--- @field offsets_h string
|
||||||
|
|
||||||
|
--- @class OffsetsPass
|
||||||
|
--- @field run fun(ctx: PassCtx): PassResult
|
||||||
|
|
||||||
|
--- @class AtomEntry
|
||||||
|
--- @field paths AtomPaths|nil
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Canonical marker projection
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
-- MARKER_PROJECTORS is the marker-kind data table.
|
||||||
|
-- The emission-model pass already records marker word positions + consuming-instruction context;
|
||||||
|
-- this pass only projects those records into the label/branch lookup shape needed by offset computation.
|
||||||
|
local MARKER_PROJECTORS = { ---@type table<string, fun(state: MarkerProjectState, marker: EmissionMarker): nil>
|
||||||
|
--- @param state MarkerProjectState
|
||||||
|
--- @param marker EmissionMarker
|
||||||
|
--- @return nil
|
||||||
|
label = function(state, marker)
|
||||||
|
state.labels[marker.name] = marker.word_index
|
||||||
|
end,
|
||||||
|
--- @param state MarkerProjectState
|
||||||
|
--- @param marker EmissionMarker
|
||||||
|
--- @return nil
|
||||||
|
offset = function(state, marker)
|
||||||
|
state.branches[#state.branches + 1] = {
|
||||||
|
tag = marker.name,
|
||||||
|
target = marker.target,
|
||||||
|
branch_word = marker.word_index,
|
||||||
|
consuming_encoder = marker.consuming_encoder,
|
||||||
|
consuming_arg_pos = marker.consuming_arg_pos,
|
||||||
|
}
|
||||||
|
end,
|
||||||
|
}
|
||||||
|
|
||||||
|
--- Project canonical marker records into the two lookup tables used by the offset renderer.
|
||||||
|
--- No source text, body text, or body token is inspected.
|
||||||
|
--- @param markers EmissionMarker[]
|
||||||
|
--- @return table<string, integer>
|
||||||
|
--- @return OffsetBranch[]
|
||||||
|
local function project_markers(markers)
|
||||||
|
local state = { labels = {}, branches = {} } ---@type MarkerProjectState
|
||||||
|
for _, marker in ipairs(markers or {}) do ---@type integer, EmissionMarker
|
||||||
|
local project = MARKER_PROJECTORS[marker.kind] ---@type (fun(state: MarkerProjectState, marker: EmissionMarker): nil)|nil
|
||||||
|
if project then project(state, marker) end
|
||||||
|
end
|
||||||
|
return state.labels, state.branches
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Offset computation + header generation
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Compute branch offsets per consuming instruction.
|
||||||
|
--- Disposition table:
|
||||||
|
--- `branch_*` -> relative offset: `target_word - branch_word - 1` (MIPS branch-immediate encoding).
|
||||||
|
--- `jump` / `call_addr` -> same value as `branch_*` (a relative word offset).
|
||||||
|
--- The duffle headers' `enc_i` macro truncates the value to the immediate-field width (16 bits for branches, 26 bits for jumps).
|
||||||
|
--- For tape-atom bodies within a single module, this works for `j`/`jal` because the linker's symbol resolution produces the correct 26-bit absolute target via standard `j` relocations.
|
||||||
|
--- For cross-module `j`/`jal` (atom body in one module, target in another), the linker emits a `R_MIPS_26` relocation against the lower 26 bits; the upper 4 bits come from the PC of the delay slot following the `j`.
|
||||||
|
--- The metaprogram doesn't know either at compile time, so the emitted value is the relative word offset that the duffle `enc_i` macro places in the immediate field; the toolchain handles the rest.
|
||||||
|
--- `jump_reg` / `call_reg` / `jump_link` -> ERROR. Register-form jumps have no offset field; `atom_offset` is invalid.
|
||||||
|
--- missing `consuming_encoder` -> ERROR. A lone top-level `atom_offset` is not a branch.
|
||||||
|
--- @param labels table<string, integer>
|
||||||
|
--- @param branches OffsetBranch[]
|
||||||
|
--- @param errors Finding[]
|
||||||
|
--- @return BranchOffset[]
|
||||||
|
local function compute_offsets(labels, branches, errors)
|
||||||
|
local results = {} ---@type BranchOffset[]
|
||||||
|
for _, br in ipairs(branches) do ---@type integer, OffsetBranch
|
||||||
|
local target = labels[br.target] ---@type integer|nil
|
||||||
|
if not target then
|
||||||
|
errors[#errors + 1] = {
|
||||||
|
line = br.line or 0,
|
||||||
|
msg = "Branch target '" .. br.target .. "' has no atom_label (at word " .. br.branch_word .. ")",
|
||||||
|
}
|
||||||
|
else
|
||||||
|
local consuming = br.consuming_encoder ---@type string|nil
|
||||||
|
if consuming == nil or consuming == "" then
|
||||||
|
errors[#errors + 1] = {
|
||||||
|
line = br.line or 0,
|
||||||
|
msg = "atom_offset requires a consuming encoder (branch_*, jump, call_addr); top-level atom_offset is invalid; at word " .. br.branch_word,
|
||||||
|
}
|
||||||
|
elseif consuming == "jump_reg" or consuming == "call_reg" or consuming == "jump_link" then
|
||||||
|
errors[#errors + 1] = {
|
||||||
|
line = br.line or 0,
|
||||||
|
msg = "atom_offset cannot be used with " .. consuming
|
||||||
|
.. " (register-form jumps have no offset field); at word " .. br.branch_word,
|
||||||
|
}
|
||||||
|
else
|
||||||
|
-- Consuming instructions with an offset field (`branch_*`, `jump`, `call_addr`) use the same relative offset value.
|
||||||
|
-- The MIPS encoding differs per opcode but the duffle `enc_i` macro handles the truncation to the immediate-field width.
|
||||||
|
results[#results + 1] = {
|
||||||
|
target = br.target,
|
||||||
|
tag = br.tag,
|
||||||
|
branch_word = br.branch_word,
|
||||||
|
offset = target - br.branch_word - 1,
|
||||||
|
consuming_encoder = br.consuming_encoder,
|
||||||
|
consuming_arg_pos = br.consuming_arg_pos,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return results
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Right-pad `s` with spaces to width `w`. If `s` is already `w` or wider, no padding is added.
|
||||||
|
--- @param s string
|
||||||
|
--- @param w integer
|
||||||
|
--- @return string
|
||||||
|
local function pad_right(s, w)
|
||||||
|
return s .. string.rep(" ", math.max(0, w - #s))
|
||||||
|
end
|
||||||
|
|
||||||
|
--- (internal) Build a constant-table entry `{macro_name, enum_name, value}` from a BranchOffset.
|
||||||
|
--- @param bo BranchOffset
|
||||||
|
--- @return OffsetConst
|
||||||
|
local function make_offset_const(bo)
|
||||||
|
return {
|
||||||
|
macro_name = OFFSET_MACRO_PREFIX .. bo.tag .. "_" .. bo.target,
|
||||||
|
enum_name = OFFSET_ENUM_PREFIX .. bo.tag .. "_" .. bo.target,
|
||||||
|
value = bo.offset,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
|
||||||
|
--- (internal) Emit one atom's offset constants + enum into the lines buffer.
|
||||||
|
--- @param add fun(s: string)
|
||||||
|
--- @param atom AtomData
|
||||||
|
--- @return nil
|
||||||
|
local function emit_atom_offsets(add, atom)
|
||||||
|
if #atom.offsets == 0 then return end
|
||||||
|
add("// --- atom: " .. atom.name .. " (" .. atom.total_words .. " words) ---")
|
||||||
|
add("")
|
||||||
|
local consts = {} ---@type OffsetConst[]
|
||||||
|
for _, r in ipairs(atom.offsets) do ---@type integer, BranchOffset
|
||||||
|
consts[#consts + 1] = make_offset_const(r)
|
||||||
|
end
|
||||||
|
for _, c in ipairs(consts) do ---@type integer, OffsetConst
|
||||||
|
add("#define " .. pad_right(c.macro_name, OFFSET_MACRO_COL) .. " " .. c.value)
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
add("enum {")
|
||||||
|
for _, c in ipairs(consts) do ---@type integer, OffsetConst
|
||||||
|
add(" " .. c.enum_name .. " = " .. c.macro_name .. ",")
|
||||||
|
end
|
||||||
|
add("};")
|
||||||
|
add("")
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Generate the per-directory .offsets.h header.
|
||||||
|
--- @param dir string
|
||||||
|
--- @param sources SourceFile[]
|
||||||
|
--- @param atoms_data AtomData[]
|
||||||
|
--- @return string
|
||||||
|
local function generate_header(dir, sources, atoms_data)
|
||||||
|
local dir_basename = duffle.basename_no_ext(dir) ---@type string
|
||||||
|
|
||||||
|
local lines = {} ---@type string[]
|
||||||
|
--- @param s string
|
||||||
|
--- @return nil
|
||||||
|
local function add(s) lines[#lines + 1] = s end
|
||||||
|
|
||||||
|
add("// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT")
|
||||||
|
add("// Directory: " .. dir:gsub("/", "\\") .. "\\")
|
||||||
|
for _, src in ipairs(sources) do ---@type integer, SourceFile
|
||||||
|
add("// source: " .. src.path:gsub("/", "\\"))
|
||||||
|
end
|
||||||
|
add("#pragma once")
|
||||||
|
add("")
|
||||||
|
add("#pragma region " .. dir_basename)
|
||||||
|
add("")
|
||||||
|
add("")
|
||||||
|
for _, atom in ipairs(atoms_data) do ---@type integer, AtomData
|
||||||
|
emit_atom_offsets(add, atom)
|
||||||
|
end
|
||||||
|
add("#pragma endregion " .. dir_basename)
|
||||||
|
add("")
|
||||||
|
return table.concat(lines, "\n") .. "\n"
|
||||||
|
end
|
||||||
|
|
||||||
|
local M = {} ---@type OffsetsPass
|
||||||
|
|
||||||
|
--- (internal) Aggregate atoms from every source in one directory, render the per-directory `offsets.h`.
|
||||||
|
--- Returns the offsets_h path if a header was written, or nil.
|
||||||
|
--- @param ctx PassCtx
|
||||||
|
--- @param dir string
|
||||||
|
--- @param sources SourceFile[]
|
||||||
|
--- @param errors Finding[]
|
||||||
|
--- @return string|nil
|
||||||
|
local function process_directory(ctx, dir, sources, errors)
|
||||||
|
local atoms_data = {} ---@type AtomData[]
|
||||||
|
|
||||||
|
--- @param atom AtomEntry
|
||||||
|
--- @return nil
|
||||||
|
local function append_atom(atom)
|
||||||
|
local paths = atom and atom.paths ---@type AtomPaths|nil
|
||||||
|
if not paths then return end
|
||||||
|
local labels, branches = project_markers(paths.markers) ---@type table<string, integer>, OffsetBranch[]
|
||||||
|
atoms_data[#atoms_data + 1] = {
|
||||||
|
name = atom.raw_name or atom.name,
|
||||||
|
total_words = #(paths.word_events or {}),
|
||||||
|
offsets = compute_offsets(labels, branches, errors),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
|
||||||
|
for _, src in ipairs(sources) do ---@type integer, SourceFile
|
||||||
|
local scan = src.scan or {} ---@type SourceScan
|
||||||
|
for _, atom in ipairs(scan.atoms or {}) do append_atom(atom) end ---@type integer, AtomEntry
|
||||||
|
for _, atom in ipairs(scan.raw_atoms or {}) do append_atom(atom) end ---@type integer, AtomEntry
|
||||||
|
end
|
||||||
|
if #atoms_data == 0 then return nil end
|
||||||
|
|
||||||
|
local out_path = dir .. "/gen/offsets.h" ---@type string
|
||||||
|
duffle.ensure_dir(duffle.dirname(out_path))
|
||||||
|
duffle.write_file(out_path, generate_header(dir, sources, atoms_data))
|
||||||
|
return out_path
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Run the offsets pass.
|
||||||
|
--- For each canonical source-directory, emits a per-directory `gen/offsets.h`
|
||||||
|
--- containing constants for every marker recorded in atom.paths across every source in that directory.
|
||||||
|
--- @param ctx PassCtx
|
||||||
|
--- @return PassResult
|
||||||
|
function M.run(ctx)
|
||||||
|
local outputs = {} ---@type OffsetOutput[]
|
||||||
|
local errors = {} ---@type Finding[]
|
||||||
|
local warnings = {} ---@type Finding[]
|
||||||
|
|
||||||
|
local corpus = ctx.shared and ctx.shared.corpus ---@type Corpus|nil
|
||||||
|
if type(corpus) ~= "table" then
|
||||||
|
error("offsets.run requires ctx.shared.corpus", 0)
|
||||||
|
end
|
||||||
|
if type(corpus.source_order) ~= "table" then
|
||||||
|
error("offsets.run requires ctx.shared.corpus.source_order.", 0)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Per-directory aggregation: every source in the same directory contributes to one `gen/offsets.h`.
|
||||||
|
local sources_by_dir = corpus.sources_by_dir or duffle.group_sources_by_dir(corpus.source_order) ---@type table<string, SourceFile[]>
|
||||||
|
for dir, sources in pairs(sources_by_dir) do ---@type string, SourceFile[]
|
||||||
|
local out_path = process_directory(ctx, dir, sources, errors) ---@type string|nil
|
||||||
|
if out_path then
|
||||||
|
outputs[#outputs + 1] = { offsets_h = out_path }
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return { outputs = outputs, errors = errors, warnings = warnings }
|
||||||
|
end
|
||||||
|
|
||||||
|
return M
|
||||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,112 @@
|
|||||||
|
--- word_count_eval.lua — Word-counting logic for the tape-atom metaprogram pipeline.
|
||||||
|
---
|
||||||
|
--- Two responsibilities:
|
||||||
|
--- 1. **Public utility** `M.count_token_words(token, wc)`: Used by `passes/offsets.lua`, `passes/annotation.lua`, and other passes.
|
||||||
|
--- 2. **Pass entry** `M.run(ctx)`: Loads the authored `word_count.metadata.h` into `ctx.shared.corpus.word_counts` for downstream passes.
|
||||||
|
--- The generated `.macs.h` files are OUTPUT artifacts and are NOT inputs to this pass;
|
||||||
|
--- Current component counts are owned by `passes/components.lua` (which populates `corpus.word_counts` and `corpus.components`
|
||||||
|
--- AFTER computing each current count from the just-built body + `corpus.word_counts`).
|
||||||
|
---
|
||||||
|
--- **Canonical contract**:
|
||||||
|
--- * `ctx.shared.corpus.word_counts` is the count table.
|
||||||
|
--- * `corpus.word_counts` is the sole count table. Consumers read `corpus.word_counts` directly.
|
||||||
|
--- * `ctx.shared.components` is NOT created by this pass (projections only).
|
||||||
|
--- * No `.macs.h` recursive discovery (no `scan_dir`, no scan cache, no `_invalidate_scan_cache`).
|
||||||
|
---
|
||||||
|
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
|
||||||
|
--- Lua 5.3 compatible.
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Module-scope requires + package.path setup
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
||||||
|
-- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator.
|
||||||
|
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
||||||
|
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" ---@type string
|
||||||
|
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") ---@type DuffleExport
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Type declarations
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- @class WordCounts
|
||||||
|
--- @field [string] integer -- bag: macro name -> word count
|
||||||
|
|
||||||
|
--- @class WordCountEval
|
||||||
|
--- @field count_token_words fun(token: string, wc: WordCounts): integer
|
||||||
|
--- @field run fun(ctx: PassCtx): PassResult
|
||||||
|
|
||||||
|
-- SourceFile, PassCtx, PassResult: see ps1_meta.lua
|
||||||
|
-- DuffleExport: see duffle.lua (facade returned by duffle_paths.lua)
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Module exports
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
local M = {} ---@type WordCountEval
|
||||||
|
|
||||||
|
-- ┌────────────────────────────────────────────────────────────────────┐
|
||||||
|
-- │ Shared utility: count_token_words │
|
||||||
|
-- └────────────────────────────────────────────────────────────────────┘
|
||||||
|
|
||||||
|
--- Count words emitted by a single comma-separated token inside an atom body.
|
||||||
|
--- For most tokens (regular MIPS instructions) this returns 1.
|
||||||
|
--- For `mac_X(...)` calls, this returns the resolved word count from `wc` (recursively if needed). For `nop2` etc., returns wc[name].
|
||||||
|
--- For unknown macros, returns 1 and (optionally) warns.
|
||||||
|
--- @param token string -- a single token from split_top_level_commas
|
||||||
|
--- @param wc WordCounts -- the shared word-count table
|
||||||
|
--- @return integer
|
||||||
|
function M.count_token_words(token, wc)
|
||||||
|
local s = duffle.trim(token) ---@type string
|
||||||
|
if s == "" then return 0 end
|
||||||
|
local name, after = duffle.read_ident(s, 1) ---@type string|nil, integer
|
||||||
|
if not name then return 1 end
|
||||||
|
if wc[name] then return wc[name] end
|
||||||
|
local paren_pos = duffle.skip_ws_and_cmt(s, after) ---@type integer
|
||||||
|
if s:sub(paren_pos, paren_pos) == "(" then
|
||||||
|
io.stderr:write(" warning: unknown macro '" .. name .. "', assuming 1 word\n")
|
||||||
|
end
|
||||||
|
return 1
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ┌────────────────────────────────────────────────────────────────────┐
|
||||||
|
-- │ Pass entry: M.run(ctx) — "word-counts" pass │
|
||||||
|
-- └────────────────────────────────────────────────────────────────────┘
|
||||||
|
|
||||||
|
--- Load the authored `word_count.metadata.h` into `ctx.shared.corpus.word_counts`.
|
||||||
|
--- Generated `.macs.h` files are OUTPUT artifacts and are NOT scanned as inputs.
|
||||||
|
--- Current component counts are computed and inserted by `passes/components.lua`
|
||||||
|
--- after the components pass iterates `corpus.source_order` and writes each source-directory's `gen/macs.h` file.
|
||||||
|
---
|
||||||
|
--- Contract:
|
||||||
|
--- * `ctx.shared.corpus` MUST exist (canonical corpus ownership).
|
||||||
|
--- * `ctx.metadata_path` MUST be a readable file path to the authored `word_count.metadata.h`.
|
||||||
|
--- * The pass assigns exactly one table to `corpus.word_counts`.
|
||||||
|
--- Consumers read the corpus-owned table directly.
|
||||||
|
--- Consumers must read `corpus.word_counts` directly.
|
||||||
|
--- @param ctx PassCtx
|
||||||
|
--- @return PassResult
|
||||||
|
function M.run(ctx)
|
||||||
|
-- 1. Canonical-corpus ownership gate.
|
||||||
|
local corpus = ctx.shared and ctx.shared.corpus ---@type Corpus|nil
|
||||||
|
if type(corpus) ~= "table" then
|
||||||
|
error("word_count_eval.run requires ctx.shared.corpus (canonical corpus). The fixture must install the corpus before running this pass.", 0)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- 2. metadata_path gate.
|
||||||
|
if type(ctx.metadata_path) ~= "string" or ctx.metadata_path == "" then
|
||||||
|
error("word_count_eval.run requires ctx.metadata_path (path to the authored word_count.metadata.h).", 0)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- 3. Load authored metadata. Generated .macs.h files are NOT scanned
|
||||||
|
-- (the pass computes their counts from the just-built bodies after disk emission; see passes/components.lua).
|
||||||
|
local wc = duffle.load_word_counts(ctx.metadata_path) ---@type WordCounts
|
||||||
|
|
||||||
|
-- 4. Assign the count table. ONE assignment, no copy. The assignment creates no secondary alias.
|
||||||
|
corpus.word_counts = wc
|
||||||
|
|
||||||
|
return { outputs = {}, errors = {}, warnings = {} }
|
||||||
|
end
|
||||||
|
|
||||||
|
return M
|
||||||
Binary file not shown.
@@ -0,0 +1,51 @@
|
|||||||
|
-- autoexec.lua - pcsx_debug_helper plugin entry point.
|
||||||
|
-- Packaged in scripts/pcsx_debug_helper.zip. Loaded by pcsx-redux via the -archive CLI flag (see scripts/launch_pcsx_debug.ps1).
|
||||||
|
--
|
||||||
|
-- Registers two web handlers for external CLI tools:
|
||||||
|
-- /api/v1/lua/gte - full GTE state (32 data + 32 control regs + PC)
|
||||||
|
-- /api/v1/lua/gp - GP state summary (screenshot endpoint + VRAM endpoint refs)
|
||||||
|
--
|
||||||
|
-- The GTE handler reads COP2 regs via PCSX.getRegisters().CP2D/CP2C.
|
||||||
|
-- The pcsx-redux gdb stub doesn't expose COP2, so this is the only way for external tools to see GTE state.
|
||||||
|
--
|
||||||
|
-- The GP handler is a thin pointer:
|
||||||
|
-- pcsx-redux's Lua API exposes only PCSX.GPU.takeScreenShot() (no GPUSTAT, no GP0/GP1 command log, no display state). For richer GP state, the existing web endpoints are the practical path:
|
||||||
|
-- /api/v1/state/still - PNG screenshot
|
||||||
|
-- /api/v1/gpu/vram/raw - VRAM raw bytes (1MB)
|
||||||
|
--
|
||||||
|
-- Companion: scripts/gdb/gdb_tape_atoms.gdb (covers GPRs + atom-aware stepping).
|
||||||
|
|
||||||
|
local function register_handlers()
|
||||||
|
if not PCSX.WebServer then PCSX.WebServer = {} end
|
||||||
|
if not PCSX.WebServer.Handlers then PCSX.WebServer.Handlers = {} end
|
||||||
|
|
||||||
|
-- ── GTE state ──
|
||||||
|
PCSX.WebServer.Handlers.gte = function(req)
|
||||||
|
local r = PCSX.getRegisters()
|
||||||
|
local out = { "pc=0x" .. string.format("%x", r.pc) }
|
||||||
|
for i = 0, 31 do
|
||||||
|
out[#out + 1] = string.format("D[%d]=0x%08x C[%d]=0x%08x",
|
||||||
|
i, r.CP2D.r[i], i, r.CP2C.r[i])
|
||||||
|
end
|
||||||
|
return table.concat(out, "\n")
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ── GP state (pointer to existing endpoints) ──
|
||||||
|
-- pcsx-redux's Lua GPU API exposes only takeScreenShot(); no GPUSTAT / GP0 / GP1 command log / display state.
|
||||||
|
-- We point to the existing web endpoints that DO expose those (when the emulator is actually rendering. Paused-at-BP frames won't have a fresh frame).
|
||||||
|
PCSX.WebServer.Handlers.gp = function(req)
|
||||||
|
local out = {
|
||||||
|
"gpu_screenshot_png=http://localhost:8080/api/v1/state/still",
|
||||||
|
"vram_raw=http://localhost:8080/api/v1/gpu/vram/raw (1MB VRAM)",
|
||||||
|
"gpustat=NOT_AVAILABLE_VIA_LUA",
|
||||||
|
"gp_command_log=NOT_AVAILABLE_VIA_LUA (use pcsx-redux Debug > GPU Logger)",
|
||||||
|
"hint_run_emulator_unpaused_for_screenshot",
|
||||||
|
}
|
||||||
|
return table.concat(out, "\n")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
local ok, err = pcall(register_handlers)
|
||||||
|
if ok then print("[pcsx_debug_helper] handlers registered: gte, gp")
|
||||||
|
else print("[pcsx_debug_helper] registration failed: " .. tostring(err))
|
||||||
|
end
|
||||||
@@ -0,0 +1,847 @@
|
|||||||
|
--- ps1_meta.lua — Orchestrator entry point for the tape-atom metaprogram.
|
||||||
|
---
|
||||||
|
--- Dispatches to pass modules under `scripts/passes/`, resolving dependencies topologically (Kahn's algorithm + cycle detection).
|
||||||
|
---
|
||||||
|
--- Architecture:
|
||||||
|
--- - PASSES table: Declarative dep graph (data, not code).
|
||||||
|
--- - FLAG_HANDLERS table: Maps CLI flags to handlers.
|
||||||
|
--- - parse_args → build_ctx (resolves unity/direct includes or exact sources) → topo_sort → dispatch_passes.
|
||||||
|
--- - The first pass in the dep graph is `scan-source` (see `passes/scan_source.lua`).
|
||||||
|
--- It calls `duffle.scan_source` once per source to produce the fat `SourceScan` payload, which is attached to each `src.scan`.
|
||||||
|
--- Every other pass that reads source structure depends on `scan-source` and consumes `src.scan` as a read-only.
|
||||||
|
---
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Module-scope requires + package.path setup
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
-- Bootstrap: load `duffle_paths.lua` via this script's own path.
|
||||||
|
-- Use `arg[0]` when this file is the entry script (`arg[0]` ends in "ps1_meta.lua");
|
||||||
|
-- fall back to `debug.getinfo(1, "S").source` when this file is being dofile()'d or require()'d (in which case `arg[0]` is the *caller's* path).
|
||||||
|
-- That single statement: (a) sets `package.path` + `package.cpath`, (b) at the bottom returns `require("duffle")`.
|
||||||
|
-- So the dofile's return value is the duffle module.
|
||||||
|
local _is_entry_script = arg and arg[0] and arg[0]:match("ps1_meta%.lua$") ~= nil ---@type boolean
|
||||||
|
local _bootstrap_src ---@type string
|
||||||
|
if _is_entry_script then
|
||||||
|
_bootstrap_src = arg[0]
|
||||||
|
else
|
||||||
|
-- debug.getinfo(1, "S").source returns "@<path>" for the current chunk;
|
||||||
|
-- strip the leading "@" so the directory match works in both cases.
|
||||||
|
_bootstrap_src = debug.getinfo(1, "S").source:sub(2)
|
||||||
|
end
|
||||||
|
local duffle = dofile((_bootstrap_src:match("(.*[/\\])") or "./") .. "duffle_paths.lua") ---@type DuffleExport
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Constants
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
-- Exit codes (per the --help text and the post-build summary convention).
|
||||||
|
local EXIT_OK = 0 ---@type integer
|
||||||
|
local EXIT_VALIDATION_ERRORS = 1 ---@type integer
|
||||||
|
local EXIT_INTERNAL_ERROR = 2 ---@type integer
|
||||||
|
|
||||||
|
-- Default --out-root value if not provided.
|
||||||
|
local DEFAULT_OUT_ROOT = "build/gen" ---@type string
|
||||||
|
|
||||||
|
-- Sentinel for "all passes" in `PASS_FLAG_TO_NAME`. Distinguishes `--all` from the per-pass flags (which map to individual pass names).
|
||||||
|
local ALL_PASSES_SENTINEL = "__all__" ---@type string
|
||||||
|
|
||||||
|
-- Sentinel key for the pass-flag dispatcher in `FLAG_HANDLERS`.
|
||||||
|
-- The actual pass names are looked up via `PASS_FLAG_TO_NAME`, not direct dispatch, so this key never matches a real flag.
|
||||||
|
local PASS_FLAG_DISPATCH_KEY = "__pass__" ---@type string
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Type declarations
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- @class PassDescriptor
|
||||||
|
--- @field module string -- Module name passed to require()
|
||||||
|
--- @field kind string -- "shared" | "header-output" | "validation" | "diagnostic" | "report"
|
||||||
|
--- -- Report severity is independent from process exit policy (see PASS_KIND_STOP_ON_ERROR).
|
||||||
|
--- @field deps string[] -- Names of upstream passes
|
||||||
|
--- @field groups string[]? -- OPTIONAL build-phase groups this pass is a root of (e.g. { "pre-link" }, { "post-link" }); absent ⇒ dependency-only
|
||||||
|
|
||||||
|
--- @class Corpus
|
||||||
|
--- @field unity_root string|nil
|
||||||
|
--- @field project_root string
|
||||||
|
--- @field code_root string
|
||||||
|
--- @field source_order SourceFile[]
|
||||||
|
--- @field sources_by_path table<Path, SourceFile>
|
||||||
|
--- @field sources_by_dir table<string, SourceFile[]>
|
||||||
|
--- @field atoms_by_name table<AtomName, AtomEntry>
|
||||||
|
--- @field binds_by_name table<string, BindsEntry>
|
||||||
|
--- @field atom_infos AtomInfoEntry[]
|
||||||
|
--- @field register_alias_registry table<string, AliasEntry>
|
||||||
|
--- @field type_name_registry table<string, TypeNameEntry>
|
||||||
|
--- @field atom_views table<AtomName, AtomViewEntry>
|
||||||
|
--- @field atom_ctxs table<AtomName, AtomCtxEntry>
|
||||||
|
--- @field atom_phases table<string, AtomPhaseGroup>
|
||||||
|
--- @field word_counts WordCounts
|
||||||
|
--- @field components table<string, Component>
|
||||||
|
--- @field atom_bundles table<string, AtomBundle>|nil
|
||||||
|
--- @field tape_emits TapeEmit[]|nil
|
||||||
|
--- @field collisions CorpusCollision[]
|
||||||
|
--- @field resolver SourceResolver
|
||||||
|
--- @field component_atom_infos AtomInfoEntry[]|nil
|
||||||
|
--- @field atom_auto_regs table<AtomName, table<string, string>>|nil
|
||||||
|
--- @field phase_auto_regs table<string, table<string, string>>|nil
|
||||||
|
--- @field reg_use_schemas table<string, RegUseSchema>|nil
|
||||||
|
--- @field reg_use_errors RegUseError[]|nil
|
||||||
|
--- @field static_analysis_results table<string, AtomAnalysis>|nil
|
||||||
|
--- @field tape_chains table<string, TapeChain>|nil
|
||||||
|
|
||||||
|
--- @class PassShared
|
||||||
|
--- @field corpus Corpus
|
||||||
|
|
||||||
|
--- @class PassFlags
|
||||||
|
--- @field gdb_runtime boolean|nil
|
||||||
|
--- @field dwarf_injection boolean|nil
|
||||||
|
--- @field elf_path string|nil
|
||||||
|
|
||||||
|
--- @class PassCtx
|
||||||
|
--- @field metadata_path string -- Path to word_count.metadata.h
|
||||||
|
--- @field shared PassShared -- Cross-pass shared state
|
||||||
|
--- @field out_root string -- Output root (e.g. "build/gen")
|
||||||
|
--- @field project_root string -- PS1 repository root
|
||||||
|
--- @field flags PassFlags -- CLI flags + per-pass stash
|
||||||
|
--- @field verbose boolean -- If true, log diagnostic info
|
||||||
|
|
||||||
|
--- CheckName: see static_analysis.lua. AtomName: see duffle.lua.
|
||||||
|
--- @class Finding
|
||||||
|
--- @field line integer
|
||||||
|
--- @field msg string
|
||||||
|
--- @field kind string|nil -- error | warning | info
|
||||||
|
--- @field atom AtomName|nil
|
||||||
|
--- @field check CheckName|nil
|
||||||
|
--- @field source string|nil -- optional; emit/reguse path
|
||||||
|
--- @field schema_name string|nil -- optional; emit/reguse
|
||||||
|
|
||||||
|
--- @class PassScratch
|
||||||
|
--- @field corpus Corpus|nil
|
||||||
|
--- @field info_by_atom table<string, AtomInfoEntry>|nil
|
||||||
|
--- @field binds_index table<string, BindsEntry>|nil
|
||||||
|
--- @field atom_index table<string, AtomEntry>|nil
|
||||||
|
--- @field annot_counts table<string, integer>|nil -- bag
|
||||||
|
--- @field types table<string, RegTypeDefault>|nil
|
||||||
|
--- @field atom_views table<string, AtomViewEntry>|nil
|
||||||
|
--- @field seen_defaults table<string, integer>|nil -- bag
|
||||||
|
--- @field seen_field table<string, integer>|nil -- bag
|
||||||
|
--- @field _scan SourceScan|nil
|
||||||
|
--- @field word_counts WordCounts|nil
|
||||||
|
--- @field register_alias_registry table<string, AliasEntry>|nil
|
||||||
|
--- @field type_name_registry table<string, TypeNameEntry>|nil
|
||||||
|
--- @field type_occurrences RegTypeOccurrence[]|nil
|
||||||
|
--- @field atom_infos_list AtomInfoEntry[]|nil
|
||||||
|
--- @field binds_list BindsEntry[]|nil
|
||||||
|
--- @field unknown_seen table<string, integer>|nil -- bag
|
||||||
|
--- @field atoms AtomEntry[]|nil
|
||||||
|
--- @field components_by_name table<string, Component>|nil
|
||||||
|
--- @field atoms_by_name table<string, AtomEntry>|nil
|
||||||
|
--- @field tape_chains table<string, string[]>|nil
|
||||||
|
--- @field source_order SourceFile[]|nil
|
||||||
|
--- @field component_atom_infos AtomInfoEntry[]|nil
|
||||||
|
--- @field atom_infos_all AtomInfoEntry[]|nil
|
||||||
|
--- @field gte_cr_alias_groups GteCrAliasGroup[]|nil
|
||||||
|
--- @field line_for_word_event (fun(ev: WordEvent): integer)|nil
|
||||||
|
|
||||||
|
--- @class PassOutputEntry
|
||||||
|
--- @field kind string
|
||||||
|
--- @field path string
|
||||||
|
|
||||||
|
--- @class PassResult
|
||||||
|
--- @field outputs PassOutputEntry[]
|
||||||
|
--- @field errors Finding[] -- Build-stops (per-pass kind policy)
|
||||||
|
--- @field warnings Finding[] -- Informational
|
||||||
|
--- @field info Finding[]|nil -- static_analysis only
|
||||||
|
|
||||||
|
--- @class ParsedArgs
|
||||||
|
--- @field requested_set string[] -- Pass names to run (explicit --all expanded)
|
||||||
|
--- @field sources string[] -- Exact --source values, retained in CLI order
|
||||||
|
--- @field unity_root string|nil -- --unity-root value; mutually exclusive with sources
|
||||||
|
--- @field metadata string -- --metadata value
|
||||||
|
--- @field out_root string -- --out-root value (default "build/gen")
|
||||||
|
--- @field project_root string -- PS1 repository root (derived from metadata by default)
|
||||||
|
--- @field verbose boolean -- If true, log diagnostic info
|
||||||
|
--- @field flags PassFlags|nil -- Per-pass stash; copied onto PassCtx.flags
|
||||||
|
|
||||||
|
--- @alias FlagHandler fun(args: ParsedArgs, argv: string[]|nil, arg_idx: integer|nil): integer|nil
|
||||||
|
|
||||||
|
--- @class PassModule
|
||||||
|
--- @field run fun(ctx: PassCtx): PassResult
|
||||||
|
|
||||||
|
--- @class Ps1MetaMod
|
||||||
|
--- @field PASSES table<string, PassDescriptor>
|
||||||
|
--- @field PASS_KIND_STOP_ON_ERROR table<string, boolean>
|
||||||
|
--- @field parse_args fun(argv: string[]): ParsedArgs
|
||||||
|
--- @field build_ctx fun(args: ParsedArgs): PassCtx
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- PASSES Table
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
-- Build-phase groups: Each PASSES row may declare membership in one or more named groups via `groups = { ... }`.
|
||||||
|
-- The CLI flags --pre-link and --post-link request the *roots* of their group; topo_sort then closes transitive dependencies from those roots,
|
||||||
|
-- and dispatch_passes runs every pass in the resulting closure without phase-filtering.
|
||||||
|
--
|
||||||
|
-- A row without a `groups` entry is dependency-only: it runs only when a transitive dep requests it,
|
||||||
|
-- but it remains directly requestable through its explicit CLI flag (e.g. --atoms-source-map, --scan-source).
|
||||||
|
|
||||||
|
local PASSES = { ---@type table<string, PassDescriptor>
|
||||||
|
["scan-source"] = {
|
||||||
|
module = "passes.scan_source",
|
||||||
|
kind = "shared", deps = {},
|
||||||
|
},
|
||||||
|
["word-counts"] = {
|
||||||
|
module = "passes.word_count_eval",
|
||||||
|
kind = "shared", deps = {},
|
||||||
|
},
|
||||||
|
components = {
|
||||||
|
module = "passes.components",
|
||||||
|
kind = "header-output",
|
||||||
|
deps = {"scan-source", "word-counts"},
|
||||||
|
},
|
||||||
|
auto_reg = {
|
||||||
|
module = "passes.auto_reg",
|
||||||
|
kind = "header-output",
|
||||||
|
deps = {"components"},
|
||||||
|
groups = { "pre-link" },
|
||||||
|
},
|
||||||
|
["emission-model"] = {
|
||||||
|
module = "passes.emission_model",
|
||||||
|
kind = "validation",
|
||||||
|
deps = {"components"},
|
||||||
|
},
|
||||||
|
annotation = {
|
||||||
|
module = "passes.annotation",
|
||||||
|
kind = "validation",
|
||||||
|
deps = {"scan-source", "word-counts"},
|
||||||
|
},
|
||||||
|
offsets = {
|
||||||
|
module = "passes.offsets",
|
||||||
|
kind = "header-output",
|
||||||
|
deps = {"scan-source", "word-counts", "components", "emission-model"},
|
||||||
|
groups = { "pre-link" },
|
||||||
|
},
|
||||||
|
["static-analysis"] = {
|
||||||
|
module = "passes.static_analysis",
|
||||||
|
-- "diagnostic" — every `error`/`warning` finding is written to the report file.
|
||||||
|
-- Report severity is independent from process exit policy.
|
||||||
|
kind = "diagnostic",
|
||||||
|
deps = {"scan-source", "word-counts", "components", "emission-model"},
|
||||||
|
},
|
||||||
|
["atoms-source-map"] = {
|
||||||
|
module = "passes.atoms_source_map",
|
||||||
|
kind = "header-output",
|
||||||
|
deps = {"word-counts", "components", "emission-model"},
|
||||||
|
},
|
||||||
|
["dwarf-injection"] = {
|
||||||
|
module = "passes.dwarf_injection",
|
||||||
|
kind = "shared",
|
||||||
|
deps = {"scan-source", "atoms-source-map"},
|
||||||
|
groups = { "post-link" },
|
||||||
|
},
|
||||||
|
report = {
|
||||||
|
module = "passes.report",
|
||||||
|
kind = "report",
|
||||||
|
deps = {"annotation", "static-analysis", "atoms-source-map"}, -- +atoms-source-map (consolidated-report-files refactor, 2026-07-26)
|
||||||
|
groups = { "pre-link" },
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
-- ────────────────────────────────────────────────────────────────────────────
|
||||||
|
-- Phase-root selection: Derive the sorted set of roots belonging to a named build-phase group, then append them to `args.requested_set`.
|
||||||
|
-- topo_sort closes the transitive deps from there; dispatch_passes runs every resolved pass without phase-filtering.
|
||||||
|
-- ────────────────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
--- @param group_name string -- Build-phase group ("pre-link" | "post-link")
|
||||||
|
--- @return string[] -- Sorted root pass names belonging to that group
|
||||||
|
local function roots_for_group(group_name)
|
||||||
|
local names = {} ---@type string[]
|
||||||
|
for name, pass in pairs(PASSES) do ---@type string, PassDescriptor
|
||||||
|
if pass.groups then
|
||||||
|
for _, g in ipairs(pass.groups) do ---@type integer, string
|
||||||
|
if g == group_name then
|
||||||
|
names[#names + 1] = name
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
table.sort(names)
|
||||||
|
return names
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Append every root belonging to `group_name` to `args.requested_set`.
|
||||||
|
--- Errors loudly if no PASSES row declares the group, so a typo'd or future-removed group name
|
||||||
|
--- cannot silently fall through to pre-link (or any other default) and dispatch nothing.
|
||||||
|
--- @param args ParsedArgs
|
||||||
|
--- @param group_name string
|
||||||
|
--- @return nil
|
||||||
|
local function request_roots_for_group(args, group_name)
|
||||||
|
local roots = roots_for_group(group_name) ---@type string[]
|
||||||
|
if #roots == 0 then
|
||||||
|
error(string.format("ps1_meta: build-phase group %q has zero roots in PASSES; check PASSES rows for a `groups = { %q }` field"
|
||||||
|
, group_name, group_name))
|
||||||
|
end
|
||||||
|
for _, name in ipairs(roots) do ---@type integer, string
|
||||||
|
args.requested_set[#args.requested_set + 1] = name
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Pass-kind taxonomy: findings always print. No pass kind stops the build.
|
||||||
|
-- Report severity is independent from process exit policy.
|
||||||
|
-- Adding a new pass kind requires listing it here explicitly; an unknown kind must not silently fall back to "true".
|
||||||
|
local PASS_KIND_STOP_ON_ERROR = { ---@type table<string, boolean> -- bag: pass kind -> stop-on-error
|
||||||
|
["shared"] = false,
|
||||||
|
["header-output"] = false,
|
||||||
|
["validation"] = false,
|
||||||
|
["diagnostic"] = false,
|
||||||
|
["report"] = false,
|
||||||
|
}
|
||||||
|
|
||||||
|
-- Closed set of CLI flags -> pass names.
|
||||||
|
-- Per-pass flags (e.g. --word-counts); phase flags (--pre-link, --post-link, --all) are within FLAG_HANDLERS because they own side effects or invoke group-derivation logic.
|
||||||
|
-- dwarf-injection is *also* a per-pass opt-in flag, but its selection + opt-in state are both owned by the explicit FLAG_HANDLERS entry below
|
||||||
|
-- (it sets args.flags.dwarf_injection and appends "dwarf-injection" to requested_set), so it is intentionally absent from this table.
|
||||||
|
local PASS_FLAG_TO_NAME = { ---@type table<string, string> -- bag: CLI flag -> pass name or ALL_PASSES_SENTINEL
|
||||||
|
["--word-counts"] = "word-counts",
|
||||||
|
["--components"] = "components",
|
||||||
|
["--validate"] = "annotation",
|
||||||
|
["--offsets"] = "offsets",
|
||||||
|
["--static-analysis"] = "static-analysis",
|
||||||
|
["--atoms-source-map"] = "atoms-source-map",
|
||||||
|
["--report"] = "report",
|
||||||
|
["--scan-source"] = "scan-source",
|
||||||
|
["--all"] = ALL_PASSES_SENTINEL,
|
||||||
|
}
|
||||||
|
|
||||||
|
--- Append every pass name to args.requested_set.
|
||||||
|
--- Names are derived from PASSES (no parallel name list); used by --all and by any caller that wants the full closure.
|
||||||
|
--- @param args ParsedArgs
|
||||||
|
--- @return nil
|
||||||
|
local function request_all_passes(args)
|
||||||
|
local names = {} ---@type string[]
|
||||||
|
for name in pairs(PASSES) do names[#names + 1] = name end ---@type string
|
||||||
|
table.sort(names)
|
||||||
|
for _, n in ipairs(names) do ---@type integer, string
|
||||||
|
args.requested_set[#args.requested_set + 1] = n
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Per-flag handlers. Each handler takes (args, argv, arg_idx) and returns the new arg_idx (so multi-arg flags like --source FILE advance it).
|
||||||
|
-- Returning nil + os.exit() handles termination flags (--help).
|
||||||
|
local FLAG_HANDLERS = {} ---@type table<string, FlagHandler>
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- CLI parsing
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Print the CLI usage to stdout and exit 0.
|
||||||
|
--- @return nil
|
||||||
|
local function print_help()
|
||||||
|
io.write([[
|
||||||
|
ps1_meta.lua - Tape-atom metaprogram orchestrator
|
||||||
|
|
||||||
|
USAGE:
|
||||||
|
ps1_meta.lua [PASS_FLAGS] [COMMON_FLAGS]
|
||||||
|
|
||||||
|
PASS_FLAGS:
|
||||||
|
Pick a phase or one-or-more individual passes:
|
||||||
|
--pre-link [phase; default] Run the pre-link group + transitive deps.
|
||||||
|
The root set is data-driven from each PASSES row's groups` field; no parallel name list is maintained.
|
||||||
|
--post-link [phase] Run the post-link group + transitive deps.
|
||||||
|
Requires --elf. Sets --gdb-runtime and --dwarf-injection opt-in flags as well.
|
||||||
|
--all Select every row of the PASSES table. Pass-local opt-in guards remain active, so --dwarf-injection still requires
|
||||||
|
--elf and --gdb-runtime still requires a runtime emission.
|
||||||
|
Or pick any subset:
|
||||||
|
--scan-source Scan sources into the fat SourceScan payload
|
||||||
|
--word-counts Load metadata.h + scan for existing .macs.h
|
||||||
|
--components Generate <srcdir>/gen/macs.h (per-directory aggregation)
|
||||||
|
--validate Run atom annotation DSL validation
|
||||||
|
--offsets Generate <srcdir>/gen/offsets.h (per-directory aggregation)
|
||||||
|
--atoms-source-map Generate <basename>.atoms.sourcemap.txt per source
|
||||||
|
--dwarf-injection [opt-in] Select the post-link dwarf-injection pass + set the opt-in flag. Requires --elf.
|
||||||
|
--static-analysis Static analysis: GTE pipeline-fill, mac_yield, ABI handoff, cycle budget
|
||||||
|
--report Render per-project summary
|
||||||
|
|
||||||
|
COMMON_FLAGS:
|
||||||
|
--unity-root FILE Unity source root: load root + direct quoted authored includes only. Mutually exclusive with --source.
|
||||||
|
--source FILE Exact source file to process (repeatable, never expands includes). Mutually exclusive with --unity-root.
|
||||||
|
--metadata PATH Path to metadata.h (required)
|
||||||
|
--out-root DIR Output root for reports (default: build/gen)
|
||||||
|
--project-root DIR PS1 repository root (default: derived from <repo>/code/duffle/word_count.metadata.h)
|
||||||
|
--gdb-runtime Also emit <out_root>/gdb_tape_atoms_runtime.gdb (post-link, requires --elf)
|
||||||
|
--elf PATH Path to linked .elf (for --gdb-runtime / --dwarf-injection)
|
||||||
|
--verbose Print per-pass debug output
|
||||||
|
--help Show this help and exit
|
||||||
|
|
||||||
|
EXIT CODES:
|
||||||
|
0 Ran. Findings print on stderr and in the report; they do not fail the process.
|
||||||
|
2 Metaprogram internal error
|
||||||
|
|
||||||
|
EXAMPLES:
|
||||||
|
ps1_meta.lua --pre-link --metadata code/duffle/word_count.metadata.h --unity-root code/gte_hello/hello_gte.c
|
||||||
|
ps1_meta.lua --post-link --metadata code/duffle/word_count.metadata.h --unity-root code/gte_hello/hello_gte.c --elf build/hello_gte.elf
|
||||||
|
ps1_meta.lua --all --metadata metadata.h --source code/foo.c --source code/bar.c
|
||||||
|
]])
|
||||||
|
end
|
||||||
|
|
||||||
|
local FLAG_VALUE_NAMES = { ---@type table<string, string> -- bag: flag -> value metavar
|
||||||
|
["--source"] = "FILE",
|
||||||
|
["--unity-root"] = "FILE",
|
||||||
|
["--metadata"] = "PATH",
|
||||||
|
["--out-root"] = "DIR",
|
||||||
|
["--project-root"] = "DIR",
|
||||||
|
["--elf"] = "PATH",
|
||||||
|
}
|
||||||
|
|
||||||
|
--- @param argv string[]
|
||||||
|
--- @param arg_idx integer
|
||||||
|
--- @param flag string
|
||||||
|
--- @return string
|
||||||
|
--- @return integer
|
||||||
|
local function require_flag_value(argv, arg_idx, flag)
|
||||||
|
local value = argv[arg_idx + 1] ---@type string|nil
|
||||||
|
local next_known = type(value) == "string" ---@type boolean
|
||||||
|
and (FLAG_HANDLERS[value] ~= nil or PASS_FLAG_TO_NAME[value] ~= nil)
|
||||||
|
if value == nil or next_known then
|
||||||
|
io.stderr:write("ps1_meta: " .. flag .. " requires " .. FLAG_VALUE_NAMES[flag] .. "\n")
|
||||||
|
os.exit(EXIT_INTERNAL_ERROR)
|
||||||
|
end
|
||||||
|
return value, arg_idx + 1
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Per-flag handlers. Each takes (args, argv, arg_idx) and returns the new arg_idx (so multi-arg flags like --source FILE advance it).
|
||||||
|
-- Termination flags like --help call os.exit() instead.
|
||||||
|
-- Populated AFTER print_help so the --help handler can reference it as an upvalue (Lua resolves locals at closure-call time,
|
||||||
|
-- but if the closure is defined before the local, it falls back to _G).
|
||||||
|
|
||||||
|
--- @param args ParsedArgs
|
||||||
|
--- @return nil
|
||||||
|
FLAG_HANDLERS["--help"] = function(args) print_help(); os.exit(0) end
|
||||||
|
--- @param args ParsedArgs
|
||||||
|
--- @return nil
|
||||||
|
FLAG_HANDLERS["--verbose"] = function(args) args.verbose = true end
|
||||||
|
|
||||||
|
--- @param args ParsedArgs
|
||||||
|
--- @param argv string[]
|
||||||
|
--- @param arg_idx integer
|
||||||
|
--- @return integer
|
||||||
|
FLAG_HANDLERS["--source"] = function(args, argv, arg_idx)
|
||||||
|
local value, value_idx = require_flag_value(argv, arg_idx, "--source") ---@type string, integer
|
||||||
|
args.sources[#args.sources + 1] = value
|
||||||
|
return value_idx
|
||||||
|
end
|
||||||
|
--- @param args ParsedArgs
|
||||||
|
--- @param argv string[]
|
||||||
|
--- @param arg_idx integer
|
||||||
|
--- @return integer
|
||||||
|
FLAG_HANDLERS["--unity-root"] = function(args, argv, arg_idx)
|
||||||
|
local value, value_idx = require_flag_value(argv, arg_idx, "--unity-root") ---@type string, integer
|
||||||
|
args.unity_root = value
|
||||||
|
return value_idx
|
||||||
|
end
|
||||||
|
--- @param args ParsedArgs
|
||||||
|
--- @param argv string[]
|
||||||
|
--- @param arg_idx integer
|
||||||
|
--- @return integer
|
||||||
|
FLAG_HANDLERS["--metadata"] = function(args, argv, arg_idx)
|
||||||
|
local value, value_idx = require_flag_value(argv, arg_idx, "--metadata") ---@type string, integer
|
||||||
|
args.metadata = value
|
||||||
|
return value_idx
|
||||||
|
end
|
||||||
|
--- @param args ParsedArgs
|
||||||
|
--- @param argv string[]
|
||||||
|
--- @param arg_idx integer
|
||||||
|
--- @return integer
|
||||||
|
FLAG_HANDLERS["--out-root"] = function(args, argv, arg_idx)
|
||||||
|
local value, value_idx = require_flag_value(argv, arg_idx, "--out-root") ---@type string, integer
|
||||||
|
args.out_root = value
|
||||||
|
return value_idx
|
||||||
|
end
|
||||||
|
--- @param args ParsedArgs
|
||||||
|
--- @param argv string[]
|
||||||
|
--- @param arg_idx integer
|
||||||
|
--- @return integer
|
||||||
|
FLAG_HANDLERS["--project-root"] = function(args, argv, arg_idx)
|
||||||
|
local value, value_idx = require_flag_value(argv, arg_idx, "--project-root") ---@type string, integer
|
||||||
|
args.project_root = value
|
||||||
|
return value_idx
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Per-pass stash flags. Read by `passes/atoms_source_map.lua` to opt into the post-link gdb-runtime emission.
|
||||||
|
-- Same shape as the existing per-flag handlers. mutates `args.flags` (which propagates into `ctx.flags`).
|
||||||
|
--- @param args ParsedArgs
|
||||||
|
--- @return nil
|
||||||
|
FLAG_HANDLERS["--gdb-runtime"] = function(args)
|
||||||
|
args.flags = args.flags or {}
|
||||||
|
args.flags.gdb_runtime = true
|
||||||
|
end
|
||||||
|
--- @param args ParsedArgs
|
||||||
|
--- @param argv string[]
|
||||||
|
--- @param arg_idx integer
|
||||||
|
--- @return integer
|
||||||
|
FLAG_HANDLERS["--elf"] = function(args, argv, arg_idx)
|
||||||
|
local value, value_idx = require_flag_value(argv, arg_idx, "--elf") ---@type string, integer
|
||||||
|
args.flags = args.flags or {}
|
||||||
|
args.flags.elf_path = value
|
||||||
|
return value_idx
|
||||||
|
end
|
||||||
|
-- Enable DWARF injection (default OFF). Opts in to the post-link pass and sets the flag in one shot.
|
||||||
|
-- The explicit handler below owns both selection and opt-in state, so --dwarf-injection is intentionally absent from PASS_FLAG_TO_NAME.
|
||||||
|
--- @param args ParsedArgs
|
||||||
|
--- @return nil
|
||||||
|
FLAG_HANDLERS["--dwarf-injection"] = function(args)
|
||||||
|
args.flags = args.flags or {}
|
||||||
|
args.flags.dwarf_injection = true
|
||||||
|
args.requested_set[#args.requested_set + 1] = "dwarf-injection"
|
||||||
|
end
|
||||||
|
-- Build-phase flags: --pre-link and --post-link request the roots of their declared groups (see roots_for_group).
|
||||||
|
-- topo_sort closes transitive deps from those roots; dispatch_passes runs every pass in the resolved closure without phase-filtering.
|
||||||
|
--- @param args ParsedArgs
|
||||||
|
--- @return nil
|
||||||
|
FLAG_HANDLERS["--pre-link"] = function(args)
|
||||||
|
request_roots_for_group(args, "pre-link")
|
||||||
|
end
|
||||||
|
-- Batch post-link phase: gdb-runtime + dwarf-injection in one luajit cold start.
|
||||||
|
-- Sets the same opt-in flags as --gdb-runtime + --dwarf-injection and selects the post-link build-phase group.
|
||||||
|
-- elf is required; parse_args enforces it after all flags are parsed.
|
||||||
|
--- @param args ParsedArgs
|
||||||
|
--- @return nil
|
||||||
|
FLAG_HANDLERS["--post-link"] = function(args)
|
||||||
|
args.flags = args.flags or {}
|
||||||
|
args.flags.gdb_runtime = true
|
||||||
|
args.flags.dwarf_injection = true
|
||||||
|
request_roots_for_group(args, "post-link")
|
||||||
|
end
|
||||||
|
|
||||||
|
-- `--dwarf-injection` also emits atom-local debug data.
|
||||||
|
|
||||||
|
-- Pass-flag handler. Reads the closed-set table, expands --all, appends to requested_set.
|
||||||
|
--- @param args ParsedArgs
|
||||||
|
--- @param a string
|
||||||
|
--- @return nil
|
||||||
|
FLAG_HANDLERS[PASS_FLAG_DISPATCH_KEY] = function(args, a)
|
||||||
|
local name = PASS_FLAG_TO_NAME[a] ---@type string|nil
|
||||||
|
if name == ALL_PASSES_SENTINEL then
|
||||||
|
request_all_passes(args)
|
||||||
|
return
|
||||||
|
end
|
||||||
|
args.requested_set[#args.requested_set + 1] = name
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Parse argv into a structured table. Validates against a closed enum.
|
||||||
|
--- @param argv string[]
|
||||||
|
--- @return ParsedArgs
|
||||||
|
local function parse_args(argv)
|
||||||
|
local args = { ---@type ParsedArgs
|
||||||
|
requested_set = {},
|
||||||
|
sources = {},
|
||||||
|
unity_root = nil,
|
||||||
|
metadata = nil,
|
||||||
|
out_root = DEFAULT_OUT_ROOT,
|
||||||
|
project_root = nil,
|
||||||
|
verbose = false,
|
||||||
|
}
|
||||||
|
|
||||||
|
local pos = 1 ---@type integer
|
||||||
|
while pos <= #argv do
|
||||||
|
local a = argv[pos] ---@type string
|
||||||
|
local handler = FLAG_HANDLERS[a] ---@type FlagHandler|nil
|
||||||
|
if handler then
|
||||||
|
pos = handler(args, argv, pos) or pos
|
||||||
|
elseif PASS_FLAG_TO_NAME[a] then
|
||||||
|
FLAG_HANDLERS[PASS_FLAG_DISPATCH_KEY](args, a)
|
||||||
|
else
|
||||||
|
io.stderr:write("ps1_meta: unknown flag '" .. a .. "'\n")
|
||||||
|
io.stderr:write("Run with --help for usage.\n")
|
||||||
|
os.exit(EXIT_INTERNAL_ERROR)
|
||||||
|
end
|
||||||
|
pos = pos + 1
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Default: --pre-link if no explicit pass flags were given.
|
||||||
|
-- The first invocation of a build is always pre-link, so this avoids silently also invoking post-link work in builds without an ELF artifact.
|
||||||
|
if #args.requested_set == 0 then request_roots_for_group(args, "pre-link") end
|
||||||
|
|
||||||
|
if not args.metadata then
|
||||||
|
io.stderr:write("ps1_meta: --metadata PATH is required\n")
|
||||||
|
os.exit(EXIT_INTERNAL_ERROR)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- `<repo>/code/duffle/word_count.metadata.h` is the canonical metadata location.
|
||||||
|
-- `project_root` names `<repo>`; the resolver derives `<project_root>/code` separately.
|
||||||
|
if not args.project_root then
|
||||||
|
local metadata_dir = duffle.dirname(duffle.normalize_path(args.metadata)) ---@type string
|
||||||
|
local code_root = duffle.dirname(metadata_dir) ---@type string
|
||||||
|
args.project_root = duffle.dirname(code_root)
|
||||||
|
else
|
||||||
|
args.project_root = duffle.normalize_path(args.project_root)
|
||||||
|
end
|
||||||
|
|
||||||
|
local has_unity = type(args.unity_root) == "string" and args.unity_root ~= "" ---@type boolean
|
||||||
|
if has_unity and #args.sources > 0 then
|
||||||
|
io.stderr:write("ps1_meta: --unity-root FILE and --source FILE are mutually exclusive\n")
|
||||||
|
os.exit(EXIT_INTERNAL_ERROR)
|
||||||
|
end
|
||||||
|
if not has_unity and #args.sources == 0 then
|
||||||
|
io.stderr:write("ps1_meta: either --unity-root FILE or at least one --source FILE is required\n")
|
||||||
|
os.exit(EXIT_INTERNAL_ERROR)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Post-link opt-ins (--gdb-runtime, --dwarf-injection) write output that depends on the linked ELF.
|
||||||
|
-- Without --elf the metaprogram can't satisfy those requests, so refuse loud and early.
|
||||||
|
-- This covers the explicit --post-link batch, --dwarf-injection by itself, and --gdb-runtime by itself.
|
||||||
|
local flags = args.flags or {} ---@type PassFlags
|
||||||
|
local elf_path = flags.elf_path ---@type string|nil
|
||||||
|
local has_elf = type(elf_path) == "string" and #elf_path > 0 ---@type boolean
|
||||||
|
local post_links = flags.gdb_runtime or flags.dwarf_injection ---@type boolean
|
||||||
|
if post_links and not has_elf then
|
||||||
|
io.stderr:write("ps1_meta: --elf PATH is required for post-link output\n")
|
||||||
|
os.exit(EXIT_INTERNAL_ERROR)
|
||||||
|
end
|
||||||
|
|
||||||
|
return args
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Build ctx from parsed args
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Build the PassCtx from parsed args. Exact mode opens only the repeated `--source` inputs;
|
||||||
|
--- unity mode delegates direct-include resolution to duffle.resolve_source_corpus`.
|
||||||
|
--- Scanning remains pass-owned (`src.scan`).
|
||||||
|
--- @param args ParsedArgs
|
||||||
|
--- @return PassCtx
|
||||||
|
local function build_ctx(args)
|
||||||
|
local normalized_project_root = duffle.normalize_path(args.project_root) ---@type string
|
||||||
|
local project_root = normalized_project_root ---@type string
|
||||||
|
local project_root_is_absolute = normalized_project_root:match("^%a:/") ---@type boolean
|
||||||
|
or normalized_project_root:sub(1, 2) == "//"
|
||||||
|
or normalized_project_root:sub(1, 1) == "/"
|
||||||
|
if not project_root_is_absolute then
|
||||||
|
-- canonical_path_key validates ordinary relative paths and rejects drive-relative paths before the absolute-path rewrite is performed.
|
||||||
|
duffle.canonical_path_key(normalized_project_root)
|
||||||
|
project_root = duffle.normalize_path(duffle.to_absolute_path(normalized_project_root))
|
||||||
|
else
|
||||||
|
-- Do not route POSIX/UNC/drive-absolute paths through to_absolute_path.
|
||||||
|
duffle.canonical_path_key(project_root)
|
||||||
|
end
|
||||||
|
local resolution ---@type Corpus
|
||||||
|
if args.unity_root then
|
||||||
|
local ok_resolve, resolved = pcall(duffle.resolve_source_corpus, { ---@type boolean, Corpus|string
|
||||||
|
unity_root = args.unity_root,
|
||||||
|
project_root = project_root,
|
||||||
|
})
|
||||||
|
if not ok_resolve then
|
||||||
|
io.stderr:write("ps1_meta: cannot resolve --unity-root " .. tostring(args.unity_root) .. ": " .. tostring(resolved) .. "\n")
|
||||||
|
os.exit(EXIT_INTERNAL_ERROR)
|
||||||
|
end
|
||||||
|
resolution = resolved
|
||||||
|
else
|
||||||
|
local ok_exact, exact = pcall(duffle.resolve_exact_sources, { ---@type boolean, Corpus|string
|
||||||
|
sources = args.sources,
|
||||||
|
project_root = project_root,
|
||||||
|
})
|
||||||
|
if not ok_exact then
|
||||||
|
io.stderr:write("ps1_meta: cannot resolve --source: " .. tostring(exact) .. "\n")
|
||||||
|
os.exit(EXIT_INTERNAL_ERROR)
|
||||||
|
end
|
||||||
|
resolution = exact
|
||||||
|
end
|
||||||
|
|
||||||
|
local corpus = { ---@type Corpus
|
||||||
|
unity_root = resolution.unity_root,
|
||||||
|
project_root = resolution.project_root,
|
||||||
|
code_root = resolution.code_root,
|
||||||
|
source_order = resolution.source_order,
|
||||||
|
sources_by_path = resolution.sources_by_path,
|
||||||
|
sources_by_dir = resolution.sources_by_dir,
|
||||||
|
atoms_by_name = {},
|
||||||
|
binds_by_name = {},
|
||||||
|
atom_infos = {},
|
||||||
|
register_alias_registry = {},
|
||||||
|
type_name_registry = {},
|
||||||
|
atom_views = {},
|
||||||
|
atom_ctxs = {},
|
||||||
|
atom_phases = {},
|
||||||
|
word_counts = {},
|
||||||
|
components = {},
|
||||||
|
atom_bundles = {},
|
||||||
|
tape_emits = {},
|
||||||
|
collisions = {},
|
||||||
|
resolver = resolution.resolver,
|
||||||
|
}
|
||||||
|
local ctx = { ---@type PassCtx
|
||||||
|
metadata_path = args.metadata,
|
||||||
|
shared = { corpus = corpus },
|
||||||
|
out_root = args.out_root,
|
||||||
|
project_root = corpus.project_root,
|
||||||
|
flags = args.flags or {},
|
||||||
|
verbose = args.verbose,
|
||||||
|
}
|
||||||
|
|
||||||
|
-- Source records and directory buckets are owned by the corpus.
|
||||||
|
-- Consumers read `corpus.source_order` and `corpus.sources_by_dir` directly.
|
||||||
|
-- The corpus is the sole source of truth for source records and module grouping; `ctx` only holds per-pass execution state.
|
||||||
|
return ctx
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Topological sort (Kahn's algorithm + cycle detection)
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Topologically sort the requested pass set, augmented with all transitive deps.
|
||||||
|
--- Detects cycles and errors out with details.
|
||||||
|
--- @param passes table<string, PassDescriptor>
|
||||||
|
--- @param requested_set string[]
|
||||||
|
--- @return string[] -- execution order
|
||||||
|
---
|
||||||
|
--- Dependency closure, in-degree calculation, queue seeding, and sorting are local blocks.
|
||||||
|
--- Keeping these blocks local makes the topological sort self-contained.
|
||||||
|
local function topo_sort(passes, requested_set)
|
||||||
|
-- Dependency closure: include every pass transitively required by `requested_set`.
|
||||||
|
local needed = {} ---@type table<string, boolean> -- bag: pass name -> needed
|
||||||
|
for _, name in ipairs(requested_set) do needed[name] = true end ---@type integer, string
|
||||||
|
local changed = true ---@type boolean
|
||||||
|
while changed do
|
||||||
|
changed = false
|
||||||
|
for name, _ in pairs(needed) do ---@type string, boolean
|
||||||
|
local pass = passes[name] ---@type PassDescriptor
|
||||||
|
if not pass then error("unknown pass '" .. name .. "' requested") end
|
||||||
|
for _, dep in ipairs(pass.deps) do ---@type integer, string
|
||||||
|
if not needed[dep] then
|
||||||
|
needed[dep] = true
|
||||||
|
changed = true
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- In-degree calculation: count each needed pass's needed dependencies.
|
||||||
|
local in_degree = {} ---@type table<string, integer> -- bag: pass name -> in-degree
|
||||||
|
for name, _ in pairs(needed) do in_degree[name] = 0 end ---@type string, boolean
|
||||||
|
for name, _ in pairs(needed) do ---@type string, boolean
|
||||||
|
for _, dep in ipairs(passes[name].deps) do ---@type integer, string
|
||||||
|
if needed[dep] then
|
||||||
|
in_degree[name] = in_degree[name] + 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Ready-queue seeding: add zero-in-degree passes in deterministic order.
|
||||||
|
local ready = {} ---@type string[]
|
||||||
|
for name, deg in pairs(in_degree) do ---@type string, integer
|
||||||
|
if deg == 0 then ready[#ready + 1] = name end
|
||||||
|
end
|
||||||
|
table.sort(ready)
|
||||||
|
|
||||||
|
-- Ready-queue drain: decrement dependents when each pass is emitted.
|
||||||
|
-- Newly-zero-degree passes are inserted back into the ready queue (kept sorted).
|
||||||
|
local order = {} ---@type string[]
|
||||||
|
while #ready > 0 do
|
||||||
|
local just_finished = table.remove(ready, 1) ---@type string
|
||||||
|
order[#order + 1] = just_finished
|
||||||
|
for name, _ in pairs(needed) do ---@type string, boolean
|
||||||
|
if name ~= just_finished then
|
||||||
|
for _, dep in ipairs(passes[name].deps) do ---@type integer, string
|
||||||
|
if dep == just_finished then
|
||||||
|
in_degree[name] = in_degree[name] - 1
|
||||||
|
if in_degree[name] == 0 then
|
||||||
|
ready[#ready + 1] = name
|
||||||
|
table.sort(ready)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Cycle detection: if `order` doesn't include all needed passes, some are stuck with in_degree > 0
|
||||||
|
-- (the cycle closed on itself before Kahn could process them).
|
||||||
|
-- Without this check, a fully-closed cycle (e.g. A -> B -> A) would silently return an empty order list, leaving the orchestrator to dispatch nothing.
|
||||||
|
local needed_count = 0 ---@type integer
|
||||||
|
for _ in pairs(needed) do needed_count = needed_count + 1 end ---@type string -- count hash entries; Lua's #t doesn't work
|
||||||
|
if #order ~= needed_count then
|
||||||
|
for name, deg in pairs(in_degree) do ---@type string, integer
|
||||||
|
if deg > 0 then
|
||||||
|
error("dependency cycle detected involving pass '" .. name .. "'")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return order
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Main Orchestrator
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- (internal) Write every pass error to stderr.
|
||||||
|
--- Returns true only when the pass kind still stops the build.
|
||||||
|
--- @param pass_name string
|
||||||
|
--- @param pass PassDescriptor
|
||||||
|
--- @param result PassResult
|
||||||
|
--- @return boolean
|
||||||
|
local function report_validation_errors(pass_name, pass, result)
|
||||||
|
local has_errors = result.errors and #result.errors > 0 ---@type boolean
|
||||||
|
if not has_errors then return false end
|
||||||
|
for _, e in ipairs(result.errors) do ---@type integer, Finding
|
||||||
|
io.stderr:write(string.format("[%s] line %d: %s\n", pass_name, e.line or 0, e.msg or ""))
|
||||||
|
end
|
||||||
|
return PASS_KIND_STOP_ON_ERROR[pass.kind] == true
|
||||||
|
end
|
||||||
|
|
||||||
|
--- (internal) Run each pass in `order` in topological sequence.
|
||||||
|
--- @param ctx PassCtx
|
||||||
|
--- @param order string[]
|
||||||
|
--- @return boolean -- true if any validation errors were reported
|
||||||
|
local function dispatch_passes(ctx, order)
|
||||||
|
local had_errors = false ---@type boolean
|
||||||
|
for _, pass_name in ipairs(order) do ---@type integer, string
|
||||||
|
local pass = PASSES[pass_name] ---@type PassDescriptor
|
||||||
|
local mod = require(pass.module) ---@type PassModule
|
||||||
|
local result = mod.run(ctx) ---@type PassResult
|
||||||
|
if report_validation_errors(pass_name, pass, result) then
|
||||||
|
had_errors = true
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return had_errors
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Main entry point. Runs the requested passes in dep-topological order.
|
||||||
|
--- @param argv string[]
|
||||||
|
--- @return nil
|
||||||
|
local function main(argv)
|
||||||
|
local ok, err = pcall(function() ---@type boolean, string|nil
|
||||||
|
local args = parse_args(argv) ---@type ParsedArgs
|
||||||
|
local ctx = build_ctx(args) ---@type PassCtx
|
||||||
|
|
||||||
|
local requested = args.requested_set ---@type string[]
|
||||||
|
local closed = topo_sort(PASSES, requested) ---@type string[]
|
||||||
|
|
||||||
|
dispatch_passes(ctx, closed)
|
||||||
|
end)
|
||||||
|
|
||||||
|
if not ok then
|
||||||
|
io.stderr:write("[ps1_meta] internal error: " .. tostring(err) .. "\n")
|
||||||
|
os.exit(EXIT_INTERNAL_ERROR)
|
||||||
|
end
|
||||||
|
|
||||||
|
os.exit(EXIT_OK)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Module export for in-process consumers (tests that dofile this script).
|
||||||
|
-- The conditional `main(...)` call below only fires when this file is invoked as the entry script (arg[0] ends in "ps1_meta.lua");
|
||||||
|
-- in dofile() mode (test's arg[0] does not match), main() is skipped and the chunk returns `_M` to the caller.
|
||||||
|
local _M = { ---@type Ps1MetaMod
|
||||||
|
PASSES = PASSES,
|
||||||
|
PASS_KIND_STOP_ON_ERROR = PASS_KIND_STOP_ON_ERROR,
|
||||||
|
parse_args = parse_args,
|
||||||
|
build_ctx = build_ctx,
|
||||||
|
}
|
||||||
|
|
||||||
|
if arg and arg[0] and arg[0]:match("ps1_meta%.lua$") then
|
||||||
|
main({...})
|
||||||
|
end
|
||||||
|
|
||||||
|
return _M
|
||||||
@@ -1,587 +0,0 @@
|
|||||||
#!/usr/bin/env lua
|
|
||||||
-- tape_atom_offset_gen.lua
|
|
||||||
--
|
|
||||||
-- Finds every `MipsAtom_(name) { ... }` declaration in the given sources,
|
|
||||||
-- counts the words in each body using the WORD_COUNT manifest, computes
|
|
||||||
-- branch offsets for atom_label(name) / atom_offset(tag, name) markers,
|
|
||||||
-- and writes one header per source into <source_dir>/gen/<basename>.offsets.h
|
|
||||||
--
|
|
||||||
-- Generated header layout (per source):
|
|
||||||
-- #pragma region <basename>
|
|
||||||
-- #undef atom_offset
|
|
||||||
-- #define atom_offset(tag, name) atom_offset_##tag##_##name
|
|
||||||
-- // --- atom: <name> (<n> words) ---
|
|
||||||
-- #define atom_offset_<tag>_<target> (N) // preprocessor form
|
|
||||||
-- #undef atom_offset_<tag>_<target> // (so enum can reuse)
|
|
||||||
-- enum {
|
|
||||||
-- atom_offset_<tag>_<target> = N, // C enum form
|
|
||||||
-- };
|
|
||||||
-- #define atom_offset_<tag>_<target> (N) // re-define for preprocessor
|
|
||||||
-- #pragma endregion <basename>
|
|
||||||
--
|
|
||||||
-- Usage:
|
|
||||||
-- lua gen_atom_offsets.lua <metadata.h> <source1> [source2 ...]
|
|
||||||
|
|
||||||
-- ============================================================
|
|
||||||
-- Character classification
|
|
||||||
-- ============================================================
|
|
||||||
|
|
||||||
local function is_space(c) return c == " " or c == "\t" or c == "\n" or c == "\r" or c == "\v" or c == "\f" end
|
|
||||||
local function is_alpha(c)
|
|
||||||
if not c or #c == 0 then return false end
|
|
||||||
if c >= "a" and c <= "z" then return true end
|
|
||||||
if c >= "A" and c <= "Z" then return true end
|
|
||||||
return c == "_"
|
|
||||||
end
|
|
||||||
|
|
||||||
local function is_digit(c) return c and c >= "0" and c <= "9" end
|
|
||||||
local function is_alnum(c) return is_alpha(c) or is_digit(c) end
|
|
||||||
|
|
||||||
-- ============================================================
|
|
||||||
-- I/O
|
|
||||||
-- ============================================================
|
|
||||||
|
|
||||||
local function read_file(path)
|
|
||||||
local f = io.open(path, "r")
|
|
||||||
if not f then error("Cannot open " .. path) end
|
|
||||||
local content = f:read("*a")
|
|
||||||
f:close()
|
|
||||||
return content
|
|
||||||
end
|
|
||||||
|
|
||||||
local function write_file(path, content)
|
|
||||||
local f = io.open(path, "w")
|
|
||||||
if not f then error("Cannot write " .. path) end
|
|
||||||
f:write(content)
|
|
||||||
f:close()
|
|
||||||
end
|
|
||||||
|
|
||||||
-- PowerShell aliases `mkdir` to New-Item, which treats `-p` as a path, so guard the call.
|
|
||||||
local function ensure_dir(path)
|
|
||||||
local is_win = package.config:sub(1, 1) == "\\"
|
|
||||||
os.execute(is_win and ('if not exist "' .. path .. '" mkdir "' .. path .. '"') or ('mkdir -p "' .. path .. '" 2>/dev/null'))
|
|
||||||
end
|
|
||||||
|
|
||||||
-- ============================================================
|
|
||||||
-- String primitives
|
|
||||||
-- ============================================================
|
|
||||||
|
|
||||||
local function trim(s)
|
|
||||||
local a = 1; while a <= #s and is_space(s:sub(a, a)) do a = a + 1 end
|
|
||||||
local b = #s; while b >= a and is_space(s:sub(b, b)) do b = b - 1 end
|
|
||||||
return s:sub(a, b)
|
|
||||||
end
|
|
||||||
|
|
||||||
local function starts_with(s, prefix)
|
|
||||||
if #s < #prefix then return false end
|
|
||||||
for i = 1, #prefix do
|
|
||||||
if s:sub(i, i) ~= prefix:sub(i, i) then return false end
|
|
||||||
end
|
|
||||||
return true
|
|
||||||
end
|
|
||||||
|
|
||||||
local function ends_with(s, suffix)
|
|
||||||
if #s < #suffix then return false end
|
|
||||||
local off = #s - #suffix
|
|
||||||
for i = 1, #suffix do
|
|
||||||
if s:sub(off + i, off + i) ~= suffix:sub(i, i) then return false end
|
|
||||||
end
|
|
||||||
return true
|
|
||||||
end
|
|
||||||
|
|
||||||
local function find_byte(haystack, target, start)
|
|
||||||
for i = start or 1, #haystack do
|
|
||||||
if haystack:sub(i, i) == target then return i end
|
|
||||||
end
|
|
||||||
return nil
|
|
||||||
end
|
|
||||||
|
|
||||||
local function dirname(path)
|
|
||||||
local last_sep = 0
|
|
||||||
for i = 1, #path do
|
|
||||||
local c = path:sub(i, i)
|
|
||||||
if c == "/" or c == "\\" then last_sep = i end
|
|
||||||
end
|
|
||||||
if last_sep == 0 then return "." end
|
|
||||||
return path:sub(1, last_sep - 1)
|
|
||||||
end
|
|
||||||
|
|
||||||
local function basename_no_ext(path)
|
|
||||||
local last_sep = 0
|
|
||||||
for i = 1, #path do
|
|
||||||
local c = path:sub(i, i)
|
|
||||||
if c == "/" or c == "\\" then last_sep = i end
|
|
||||||
end
|
|
||||||
local a = last_sep + 1
|
|
||||||
local last_dot = #path + 1
|
|
||||||
for i = #path, a, -1 do
|
|
||||||
if path:sub(i, i) == "." then last_dot = i; break end
|
|
||||||
end
|
|
||||||
return path:sub(a, last_dot - 1)
|
|
||||||
end
|
|
||||||
|
|
||||||
local function to_upper(s) return s:upper() end
|
|
||||||
local function to_alnum_underscore(s)
|
|
||||||
local out = ""
|
|
||||||
for i = 1, #s do
|
|
||||||
local c = s:sub(i, i)
|
|
||||||
if is_alnum(c) then out = out .. c
|
|
||||||
else out = out .. "_" end
|
|
||||||
end
|
|
||||||
return out
|
|
||||||
end
|
|
||||||
local function pad_right(s, w) return s .. string.rep(" ", w - #s) end
|
|
||||||
|
|
||||||
-- ============================================================
|
|
||||||
-- Lexer helpers
|
|
||||||
-- ============================================================
|
|
||||||
|
|
||||||
-- If position i starts a C string literal ("..."), char literal ('.'),
|
|
||||||
-- // line comment, or /* block comment, advance past it and return the
|
|
||||||
-- position just after the construct (or #s+1 if unterminated).
|
|
||||||
-- Otherwise return i unchanged.
|
|
||||||
local function skip_str_or_cmt(s, i)
|
|
||||||
local c = s:sub(i, i)
|
|
||||||
if c == '"' or c == "'" then
|
|
||||||
i = i + 1
|
|
||||||
while i <= #s do
|
|
||||||
if s:sub(i, i) == "\\" then i = i + 2
|
|
||||||
elseif s:sub(i, i) == c then return i + 1
|
|
||||||
else i = i + 1 end
|
|
||||||
end
|
|
||||||
return #s + 1
|
|
||||||
elseif c == "/" then
|
|
||||||
local nx = s:sub(i+1, i+1)
|
|
||||||
if nx == "/" then
|
|
||||||
while i <= #s and s:sub(i, i) ~= "\n" do i = i + 1 end
|
|
||||||
return i
|
|
||||||
elseif nx == "*" then
|
|
||||||
i = i + 2
|
|
||||||
while i <= #s - 1 do
|
|
||||||
if s:sub(i, i) == "*" and s:sub(i+1, i+1) == "/" then
|
|
||||||
return i + 2
|
|
||||||
end
|
|
||||||
i = i + 1
|
|
||||||
end
|
|
||||||
return #s + 1
|
|
||||||
end
|
|
||||||
end
|
|
||||||
return i
|
|
||||||
end
|
|
||||||
|
|
||||||
local function skip_ws_and_cmt(s, i)
|
|
||||||
while i <= #s do
|
|
||||||
if is_space(s:sub(i, i)) then i = i + 1
|
|
||||||
else
|
|
||||||
local nx = skip_str_or_cmt(s, i)
|
|
||||||
if nx > i then i = nx else break end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
return i
|
|
||||||
end
|
|
||||||
|
|
||||||
local function read_ident(source, i)
|
|
||||||
if not is_alpha(source:sub(i, i)) then return nil, i end
|
|
||||||
local a = i
|
|
||||||
i = i + 1
|
|
||||||
while i <= #source and is_alnum(source:sub(i, i)) do i = i + 1 end
|
|
||||||
return source:sub(a, i - 1), i
|
|
||||||
end
|
|
||||||
|
|
||||||
local function read_balanced(s, open_char, close_char, i)
|
|
||||||
if s:sub(i, i) ~= open_char then return nil, i end
|
|
||||||
i = i + 1
|
|
||||||
local len = #s
|
|
||||||
local depth = 1
|
|
||||||
local a = i
|
|
||||||
while i <= len and depth > 0 do
|
|
||||||
local c = s:sub(i, i)
|
|
||||||
if c == open_char then
|
|
||||||
depth = depth + 1
|
|
||||||
i = i + 1
|
|
||||||
elseif c == close_char then
|
|
||||||
depth = depth - 1
|
|
||||||
if depth == 0 then break end
|
|
||||||
i = i + 1
|
|
||||||
else
|
|
||||||
local nx = skip_str_or_cmt(s, i)
|
|
||||||
if nx > i then i = nx else i = i + 1 end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
return s:sub(a, i - 1), i + 1
|
|
||||||
end
|
|
||||||
|
|
||||||
local read_parens = function(s, i) return read_balanced(s, "(", ")", i) end
|
|
||||||
local read_braces = function(s, i) return read_balanced(s, "{", "}", i) end
|
|
||||||
local read_brackets = function(s, i) return read_balanced(s, "[", "]", i) end
|
|
||||||
|
|
||||||
local function scan_to_char(s, target, start)
|
|
||||||
local i = start
|
|
||||||
while i <= #s do
|
|
||||||
local c = s:sub(i, i)
|
|
||||||
if c == target then return i end
|
|
||||||
if c == "(" then local _, a = read_balanced(s, "(", ")", i); i = a
|
|
||||||
elseif c == "{" then local _, a = read_balanced(s, "{", "}", i); i = a
|
|
||||||
elseif c == "[" then local _, a = read_balanced(s, "[", "]", i); i = a
|
|
||||||
else
|
|
||||||
local nx = skip_str_or_cmt(s, i)
|
|
||||||
if nx > i then i = nx else i = i + 1 end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
-- ============================================================
|
|
||||||
-- Extract comma-separated identifier args from a parenthesized group
|
|
||||||
-- after a function-like macro call.
|
|
||||||
-- ============================================================
|
|
||||||
|
|
||||||
local function extract_ident_args(token, after_ident)
|
|
||||||
local arg_start = skip_ws_and_cmt(token, after_ident)
|
|
||||||
if token:sub(arg_start, arg_start) ~= "(" then return {}, nil end
|
|
||||||
local inner, after_paren = read_parens(token, arg_start)
|
|
||||||
|
|
||||||
local args = {}
|
|
||||||
local n = 1
|
|
||||||
local len = #inner
|
|
||||||
while n <= len do
|
|
||||||
n = skip_ws_and_cmt(inner, n)
|
|
||||||
if n > len then break end
|
|
||||||
local ident, after = read_ident(inner, n)
|
|
||||||
if ident and ident ~= "" then
|
|
||||||
table.insert(args, ident)
|
|
||||||
n = after
|
|
||||||
else
|
|
||||||
n = n + 1
|
|
||||||
end
|
|
||||||
n = skip_ws_and_cmt(inner, n)
|
|
||||||
if n <= len and inner:sub(n, n) == "," then n = n + 1 end
|
|
||||||
end
|
|
||||||
|
|
||||||
return args, after_paren
|
|
||||||
end
|
|
||||||
|
|
||||||
-- ============================================================
|
|
||||||
-- Load WORD_COUNT manifest
|
|
||||||
-- ============================================================
|
|
||||||
|
|
||||||
local function load_word_counts(metadata_path)
|
|
||||||
local counts = {}
|
|
||||||
local content = read_file(metadata_path)
|
|
||||||
local len = #content
|
|
||||||
local i = 1
|
|
||||||
local prefix = "WORD_COUNT("
|
|
||||||
while i <= len do
|
|
||||||
local nl = find_byte(content, "\n", i)
|
|
||||||
local line_end = nl or (len + 1)
|
|
||||||
local line = content:sub(i, line_end - 1)
|
|
||||||
local trimmed = trim(line)
|
|
||||||
if starts_with(trimmed, prefix) and ends_with(trimmed, ")") then
|
|
||||||
local inner = trimmed:sub(#prefix + 1, #trimmed - 1)
|
|
||||||
local comma = find_byte(inner, ",", 1)
|
|
||||||
if comma then
|
|
||||||
counts[trim(inner:sub(1, comma - 1))] = tonumber(trim(inner:sub(comma + 1)))
|
|
||||||
end
|
|
||||||
end
|
|
||||||
i = line_end + 1
|
|
||||||
end
|
|
||||||
return counts
|
|
||||||
end
|
|
||||||
|
|
||||||
-- ============================================================
|
|
||||||
-- Count words for a single comma-separated token
|
|
||||||
-- ============================================================
|
|
||||||
|
|
||||||
local function word_count_of_token(token, wc)
|
|
||||||
local s = trim(token)
|
|
||||||
if s == "" then return 0 end
|
|
||||||
local name, after = read_ident(s, 1)
|
|
||||||
if not name then return 1 end
|
|
||||||
if wc[name] then return wc[name] end
|
|
||||||
local j = skip_ws_and_cmt(s, after)
|
|
||||||
if s:sub(j, j) == "(" then
|
|
||||||
io.stderr:write(" warning: unknown macro '" .. name .. "', assuming 1 word\n")
|
|
||||||
end
|
|
||||||
return 1
|
|
||||||
end
|
|
||||||
|
|
||||||
-- ============================================================
|
|
||||||
-- Split brace-body into top-level comma-separated tokens
|
|
||||||
-- ============================================================
|
|
||||||
|
|
||||||
local function split_top_level_commas(body)
|
|
||||||
local tokens = {}
|
|
||||||
local i = 1
|
|
||||||
local token_start = 1
|
|
||||||
while i <= #body do
|
|
||||||
local c = body:sub(i, i)
|
|
||||||
if c == "(" then local _, a = read_parens(body, i); i = a
|
|
||||||
elseif c == "{" then local _, a = read_braces(body, i); i = a
|
|
||||||
elseif c == "[" then local _, a = read_brackets(body, i); i = a
|
|
||||||
elseif c == "," then
|
|
||||||
table.insert(tokens, body:sub(token_start, i - 1))
|
|
||||||
i = i + 1
|
|
||||||
token_start = i
|
|
||||||
else
|
|
||||||
local nx = skip_str_or_cmt(body, i)
|
|
||||||
if nx > i then i = nx else i = i + 1 end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
local last = body:sub(token_start)
|
|
||||||
if trim(last) ~= "" then table.insert(tokens, last) end
|
|
||||||
return tokens
|
|
||||||
end
|
|
||||||
|
|
||||||
-- ============================================================
|
|
||||||
-- Scan token for atom_label/atom_offset markers, walking through
|
|
||||||
-- balanced groups transparently (so nested calls are found)
|
|
||||||
-- ============================================================
|
|
||||||
|
|
||||||
local function scan_for_atom_markers(token, at_pos, labels, branches)
|
|
||||||
local i = 1
|
|
||||||
local len = #token
|
|
||||||
while i <= len do
|
|
||||||
i = skip_ws_and_cmt(token, i)
|
|
||||||
if i > len then break end
|
|
||||||
local c = token:sub(i, i)
|
|
||||||
if is_alpha(c) then
|
|
||||||
local ident, after = read_ident(token, i)
|
|
||||||
if ident == "atom_label" then
|
|
||||||
local args, after_paren = extract_ident_args(token, after)
|
|
||||||
if #args >= 1 then labels[args[1]] = at_pos end
|
|
||||||
if after_paren then i = after_paren else i = after end
|
|
||||||
elseif ident == "atom_offset" then
|
|
||||||
local args, after_paren = extract_ident_args(token, after)
|
|
||||||
if #args >= 2 then table.insert(branches, {pos = at_pos, target = args[2], tag = args[1]}) end
|
|
||||||
if after_paren then i = after_paren else i = after end
|
|
||||||
else
|
|
||||||
i = after
|
|
||||||
end
|
|
||||||
else
|
|
||||||
local nx = skip_str_or_cmt(token, i)
|
|
||||||
if nx > i then i = nx else i = i + 1 end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
-- ============================================================
|
|
||||||
-- Scan atom body, count words, find markers
|
|
||||||
-- ============================================================
|
|
||||||
|
|
||||||
local function scan_atom_body(body, word_counts)
|
|
||||||
local pos = 0
|
|
||||||
local labels = {}
|
|
||||||
local branches = {}
|
|
||||||
for _, tok in ipairs(split_top_level_commas(body)) do
|
|
||||||
local k = 1
|
|
||||||
local tlen = #tok
|
|
||||||
while k <= tlen and is_space(tok:sub(k, k)) do k = k + 1 end
|
|
||||||
local leading_ident = read_ident(tok, k)
|
|
||||||
if leading_ident == "atom_label" or leading_ident == "atom_offset" then
|
|
||||||
scan_for_atom_markers(tok, pos, labels, branches)
|
|
||||||
else
|
|
||||||
local words = word_count_of_token(tok, word_counts)
|
|
||||||
scan_for_atom_markers(tok, pos, labels, branches)
|
|
||||||
pos = pos + words
|
|
||||||
end
|
|
||||||
end
|
|
||||||
return labels, branches, pos
|
|
||||||
end
|
|
||||||
|
|
||||||
-- ============================================================
|
|
||||||
-- Find every MipsAtom_(name) { ... } in a source
|
|
||||||
-- ============================================================
|
|
||||||
|
|
||||||
local function skip_qualifiers(source, i)
|
|
||||||
local keywords = {
|
|
||||||
["static"] = true, ["const"] = true, ["volatile"] = true,
|
|
||||||
["extern"] = true, ["register"] = true, ["auto"] = true,
|
|
||||||
["inline"] = true, ["typedef"] = true,
|
|
||||||
["internal"]= true, ["LP_"] = true, ["global"] = true, ["gkknown"] = true
|
|
||||||
}
|
|
||||||
while true do
|
|
||||||
i = skip_ws_and_cmt(source, i)
|
|
||||||
local ident, after = read_ident(source, i)
|
|
||||||
if not ident then return i end
|
|
||||||
if keywords[ident] then i = after else return i end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
local function find_atoms(source_text)
|
|
||||||
local atoms = {}
|
|
||||||
local len = #source_text
|
|
||||||
local i = 1
|
|
||||||
|
|
||||||
local function try_wrapped(after_pos)
|
|
||||||
local paren_pos = skip_ws_and_cmt(source_text, after_pos)
|
|
||||||
if source_text:sub(paren_pos, paren_pos) ~= "(" then return nil end
|
|
||||||
local inner, after_paren = read_parens(source_text, paren_pos)
|
|
||||||
local n = 1
|
|
||||||
while n <= #inner and is_space(inner:sub(n, n)) do n = n + 1 end
|
|
||||||
local ns = n
|
|
||||||
while n <= #inner and is_alnum(inner:sub(n, n)) do n = n + 1 end
|
|
||||||
local name = inner:sub(ns, n - 1)
|
|
||||||
if name == "" then return nil end
|
|
||||||
local brace_pos = scan_to_char(source_text, "{", after_paren)
|
|
||||||
if not brace_pos then return nil end
|
|
||||||
local body, after_brace = read_braces(source_text, brace_pos)
|
|
||||||
return {name = name, body = body, after_brace = after_brace}
|
|
||||||
end
|
|
||||||
|
|
||||||
local function try_raw(after_pos)
|
|
||||||
local next_pos = skip_ws_and_cmt(source_text, after_pos)
|
|
||||||
local next_ident, next_after = read_ident(source_text, next_pos)
|
|
||||||
if not next_ident then return nil end
|
|
||||||
if not starts_with(next_ident, "code_") then return nil end
|
|
||||||
if #next_ident <= 5 then return nil end
|
|
||||||
local atom_name = next_ident:sub(6)
|
|
||||||
local brace_pos = scan_to_char(source_text, "{", next_after)
|
|
||||||
if not brace_pos then return nil end
|
|
||||||
local body, after_brace = read_braces(source_text, brace_pos)
|
|
||||||
return {name = atom_name, body = body, after_brace = after_brace}
|
|
||||||
end
|
|
||||||
|
|
||||||
while i <= len do
|
|
||||||
i = skip_ws_and_cmt(source_text, i); if i > len then break end
|
|
||||||
i = skip_qualifiers(source_text, i); if i > len then break end
|
|
||||||
local ident, after = read_ident(source_text, i)
|
|
||||||
if not ident then
|
|
||||||
i = i + 1
|
|
||||||
elseif ident == "MipsAtom_" then
|
|
||||||
local atom = try_wrapped(after)
|
|
||||||
if atom then
|
|
||||||
table.insert(atoms, {name = atom.name, body = atom.body})
|
|
||||||
i = atom.after_brace
|
|
||||||
else
|
|
||||||
i = i + 1
|
|
||||||
end
|
|
||||||
elseif ident == "MipsCode" then
|
|
||||||
local atom = try_raw(after)
|
|
||||||
if atom then
|
|
||||||
table.insert(atoms, {name = atom.name, body = atom.body})
|
|
||||||
i = atom.after_brace
|
|
||||||
else
|
|
||||||
i = after
|
|
||||||
end
|
|
||||||
else
|
|
||||||
i = after
|
|
||||||
end
|
|
||||||
end
|
|
||||||
return atoms
|
|
||||||
end
|
|
||||||
|
|
||||||
-- ============================================================
|
|
||||||
-- Compute branch offsets (target - branch - 1)
|
|
||||||
-- ============================================================
|
|
||||||
|
|
||||||
local function compute_offsets(labels, branches)
|
|
||||||
local results = {}
|
|
||||||
for _, br in ipairs(branches) do
|
|
||||||
local target = labels[br.target]
|
|
||||||
if not target then
|
|
||||||
error("Branch target '" .. br.target .. "' has no atom_label (at word " .. br.pos .. ")")
|
|
||||||
end
|
|
||||||
table.insert(results, {target = br.target, tag = br.tag, offset = target - br.pos - 1 })
|
|
||||||
end
|
|
||||||
return results
|
|
||||||
end
|
|
||||||
|
|
||||||
-- ============================================================
|
|
||||||
-- Generate header for one source
|
|
||||||
-- ============================================================
|
|
||||||
|
|
||||||
local function generate_header(source_path, atoms_data)
|
|
||||||
local basename = basename_no_ext(source_path)
|
|
||||||
local guard = to_alnum_underscore(to_upper(basename)) .. "_OFFSETS_H"
|
|
||||||
|
|
||||||
local lines = {}
|
|
||||||
local function add(s) table.insert(lines, s) end
|
|
||||||
|
|
||||||
add("// Auto-generated by tape_atom_offset_gen.meta.lua — DO NOT EDIT")
|
|
||||||
add("// Source: " .. source_path)
|
|
||||||
add("#pragma once")
|
|
||||||
add("")
|
|
||||||
add("#pragma region " .. basename)
|
|
||||||
add("")
|
|
||||||
-- add("// Dispatch macro: token-pastes <tag>_<target> to the enum name")
|
|
||||||
-- add("#undef atom_offset")
|
|
||||||
-- add("#define atom_offset(tag, name) atom_offset_##tag##_##name")
|
|
||||||
add("")
|
|
||||||
for _, atom in ipairs(atoms_data) do
|
|
||||||
if #atom.offsets > 0 then
|
|
||||||
add("// --- atom: " .. atom.name .. " (" .. atom.total_words .. " words) ---")
|
|
||||||
add("")
|
|
||||||
local consts = {}
|
|
||||||
for _, r in ipairs(atom.offsets) do
|
|
||||||
table.insert(consts, {
|
|
||||||
macro_name = "_atom_offset_" .. r.tag .. "_" .. r.target,
|
|
||||||
enum_name = "atom_offset_" .. r.tag .. "_" .. r.target,
|
|
||||||
value = r.offset
|
|
||||||
})
|
|
||||||
end
|
|
||||||
for _, c in ipairs(consts) do add("#define " .. pad_right(c.macro_name, 44) .. " " .. c.value .. "") end
|
|
||||||
add("")
|
|
||||||
add("enum {")
|
|
||||||
for _, c in ipairs(consts) do add(" " .. c.enum_name .. " = " .. c.macro_name .. ",") end
|
|
||||||
add("};")
|
|
||||||
add("")
|
|
||||||
end
|
|
||||||
end
|
|
||||||
add("#pragma endregion " .. basename)
|
|
||||||
add("")
|
|
||||||
return table.concat(lines, "\n") .. "\n"
|
|
||||||
end
|
|
||||||
|
|
||||||
-- ============================================================
|
|
||||||
-- Process one source
|
|
||||||
-- ============================================================
|
|
||||||
|
|
||||||
local function process_source(source_path, word_counts)
|
|
||||||
local source = read_file(source_path)
|
|
||||||
local atoms_raw = find_atoms(source)
|
|
||||||
|
|
||||||
if #atoms_raw == 0 then
|
|
||||||
-- io.stderr:write(" note: no MipsAtom_ declarations in " .. source_path .. "\n")
|
|
||||||
return
|
|
||||||
end
|
|
||||||
|
|
||||||
local atoms_data = {}
|
|
||||||
for _, atom in ipairs(atoms_raw) do
|
|
||||||
local labels, branches, total = scan_atom_body(atom.body, word_counts)
|
|
||||||
local offsets = compute_offsets(labels, branches)
|
|
||||||
table.insert(atoms_data, {
|
|
||||||
name = atom.name,
|
|
||||||
total_words = total,
|
|
||||||
offsets = offsets
|
|
||||||
})
|
|
||||||
end
|
|
||||||
|
|
||||||
local basename = basename_no_ext(source_path)
|
|
||||||
local out_dir = dirname(source_path) .. "/gen"
|
|
||||||
ensure_dir(out_dir)
|
|
||||||
local out_path = out_dir .. "/" .. basename .. ".offsets.h"
|
|
||||||
write_file(out_path, generate_header(source_path, atoms_data))
|
|
||||||
|
|
||||||
local total_branches = 0
|
|
||||||
for _, a in ipairs(atoms_data) do total_branches = total_branches + #a.offsets end
|
|
||||||
print(" " .. basename .. ": " .. #atoms_data .. " atom(s), " .. total_branches .. " branch(es)")
|
|
||||||
for _, a in ipairs(atoms_data) do
|
|
||||||
for _, r in ipairs(a.offsets) do
|
|
||||||
print(" " .. a.name .. " -> " .. r.tag .. ":" .. r.target .. " : " .. r.offset)
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
-- ============================================================
|
|
||||||
-- Main
|
|
||||||
-- ============================================================
|
|
||||||
|
|
||||||
local function main(args)
|
|
||||||
if #args < 2 then
|
|
||||||
print("Usage: gen_atom_offsets.lua <metadata.h> <source1> [source2 ...]")
|
|
||||||
os.exit(1)
|
|
||||||
end
|
|
||||||
local word_counts = load_word_counts(args[1])
|
|
||||||
for i = 2, #args do process_source(args[i], word_counts) end
|
|
||||||
end
|
|
||||||
|
|
||||||
main({...})
|
|
||||||
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user