Author SHA1 Message Date
ed 4fbf550d3c hot-reload attempt (unreviewed, not working) 2026-08-06 10:44:34 -04:00
65 changed files with 2540 additions and 16553 deletions
-26
View File
@@ -1,26 +0,0 @@
# Cozy and Windy
Editor theme ported from the Rider scheme of the same name.
It colors the editor surface, C/C++ syntax, and tape-atom DSL keywords
emitted by `local.tape-atom-syntax`. It does not change workbench chrome.
## Install
```powershell
cd C:\projects\Pikuma\ps1\.vscode\cozy-and-windy
npm run package
code --install-extension .\cozy-and-windy-0.1.0.vsix --force
```
Reload the window. Select **Cozy and Windy** as the color theme, or set
`workbench.colorTheme` to `Cozy and Windy` in the PS1 workspace settings.
Keep `local.tape-atom-syntax` installed. This theme colors those token
types; it does not classify them.
## Inspect
Open `hello_camera.atom.c` and run **Developer: Inspect Editor Tokens and Scopes**
on `MipsAtom_`, an atom name, `atom_info`, `R_PrimCursor`, a `gte_*` call,
and a `mac_*` call.
Binary file not shown.
-25
View File
@@ -1,25 +0,0 @@
{
"name": "cozy-and-windy",
"displayName": "Cozy and Windy",
"description": "Editor theme ported from the Rider Cozy and Windy scheme. Colors C/C++ and tape-atom DSL keywords.",
"publisher": "local",
"version": "0.1.0",
"engines": {
"vscode": "^1.80.0"
},
"categories": [
"Themes"
],
"scripts": {
"package": "npx --yes @vscode/vsce@3.6.1 package --allow-missing-repository --skip-license --out cozy-and-windy-0.1.0.vsix"
},
"contributes": {
"themes": [
{
"label": "Cozy and Windy",
"uiTheme": "vs-dark",
"path": "./themes/cozy-and-windy-color-theme.json"
}
]
}
}
@@ -1,122 +0,0 @@
{
"name": "Cozy and Windy",
"type": "dark",
"semanticHighlighting": true,
"colors": {
"editor.background": "#121212",
"editor.foreground": "#D25A6EFF",
"editor.lineHighlightBackground": "#1c1c1c",
"editor.selectionBackground": "#164371",
"editor.selectionForeground": "#c8c8c8",
"editorLineNumber.foreground": "#43c3c3",
"editorLineNumber.activeForeground": "#00fff4",
"editorIndentGuide.background1": "#181818",
"editorIndentGuide.activeBackground1": "#202020",
"editorRuler.foreground": "#505050",
"editorGutter.background": "#121212",
"editorBracketMatch.background": "#3b514d",
"editor.foldBackground": "#0c0c0c",
"editor.wordHighlightBackground": "#121212",
"editor.wordHighlightStrongBackground": "#303030",
"editorCursor.foreground": "#00fff4",
"editorWhitespace.foreground": "#181818",
"editorLineHighlightBorder": "#1c1c1c",
"editorWidget.background": "#121212",
"editorSuggestWidget.background": "#2c334b",
"editorHoverWidget.background": "#2c334b"
},
"semanticTokenColors": {
"comment": { "foreground": "#868686", "fontStyle": "italic" },
"keyword": { "foreground": "#d8bd5b" },
"string": { "foreground": "#d46a54" },
"number": { "foreground": "#b5cea8" },
"operator": { "foreground": "#00ffff" },
"class": { "foreground": "#54a4d6" },
"struct": { "foreground": "#54a4d6" },
"enum": { "foreground": "#54a4d6" },
"type": { "foreground": "#54a4d6" },
"interface": { "foreground": "#7984ab" },
"function": { "foreground": "#cccab5" },
"method": { "foreground": "#6090a9" },
"variable": { "foreground": "#bc966c" },
"parameter": { "foreground": "#867660", "fontStyle": "underline" },
"property": { "foreground": "#acb8c8" },
"*.static": { "foreground": "#9e95c6" },
"macro": { "foreground": "#5ea852" },
"namespace": { "foreground": "#8e8e8e" },
"typeParameter": { "foreground": "#b8d7a3" },
"enumMember": { "foreground": "#a373b0" },
"label": { "foreground": "#c8c8c8", "fontStyle": "bold" },
"tapeAtomKeyword": { "foreground": "#d8bd5b", "fontStyle": "bold" },
"tapeAtomName": { "foreground": "#cccab5", "fontStyle": "bold" },
"tapeComponentKeyword": { "foreground": "#d68a36", "fontStyle": "bold" },
"tapeComponentName": { "foreground": "#6090a9", "fontStyle": "bold" },
"tapeAnnotation": { "foreground": "#d8bd5b" },
"tapeBindType": { "foreground": "#54a4d6" },
"tapePhase": { "foreground": "#b8d7a3", "fontStyle": "italic" },
"tapeLabel": { "foreground": "#c8c8c8", "fontStyle": "bold" },
"tapeCpuInstruction": { "foreground": "#6090a9" },
"tapeGteInstruction": { "foreground": "#9e95c6" },
"tapeGpuInstruction": { "foreground": "#bf7dac" },
"tapeComponentInstruction": { "foreground": "#8baa5d" },
"tapeGprRegister": { "foreground": "#bc966c" },
"tapeCop2Register": { "foreground": "#9e95c6" },
"tapeDuffleType": { "foreground": "#54a4d6" },
"tapeAttribute": { "foreground": "#73a07c" },
"tapeGprRegister.tapeRead": { "foreground": "#8baa5d", "fontStyle": "italic" },
"tapeGprRegister.tapeWrite": { "foreground": "#d68a36", "fontStyle": "bold" },
"tapeCop2Register.tapeRead": { "foreground": "#a373b0", "fontStyle": "italic" },
"tapeCop2Register.tapeWrite": { "foreground": "#d46a54", "fontStyle": "bold" },
"*.tapeAuto": { "fontStyle": "underline" },
"tapeControlFlow": { "foreground": "#76ff7d", "fontStyle": "bold" },
"tapeDelaySlot": { "foreground": "#ff5647" }
},
"tokenColors": [
{ "scope": ["comment", "comment.block", "comment.line", "comment.block.documentation"], "settings": { "foreground": "#868686", "fontStyle": "italic" } },
{ "scope": ["keyword", "keyword.control", "keyword.other"], "settings": { "foreground": "#d8bd5b" } },
{ "scope": ["string", "string.quoted"], "settings": { "foreground": "#d46a54" } },
{ "scope": ["string.quoted.other"], "settings": { "foreground": "#d69d85" } },
{ "scope": ["constant.numeric"], "settings": { "foreground": "#b5cea8" } },
{ "scope": ["punctuation", "keyword.operator"], "settings": { "foreground": "#00ffff" } },
{ "scope": ["keyword.operator.overload"], "settings": { "foreground": "#00ff64" } },
{ "scope": ["entity.name.type", "entity.name.type.class", "entity.name.type.struct", "entity.name.type.enum"], "settings": { "foreground": "#54a4d6" } },
{ "scope": ["entity.name.type.interface"], "settings": { "foreground": "#7984ab" } },
{ "scope": ["entity.name.function"], "settings": { "foreground": "#cccab5" } },
{ "scope": ["entity.name.function.member"], "settings": { "foreground": "#6090a9" } },
{ "scope": ["variable.other.local"], "settings": { "foreground": "#bc966c" } },
{ "scope": ["variable.parameter"], "settings": { "foreground": "#867660", "fontStyle": "underline" } },
{ "scope": ["variable.other.property"], "settings": { "foreground": "#acb8c8" } },
{ "scope": ["variable.other.constant"], "settings": { "foreground": "#9e95c6" } },
{ "scope": ["variable.other.global", "variable.other.defaultLibrary"], "settings": { "foreground": "#bf7dac" } },
{ "scope": ["support.type", "support.function"], "settings": { "foreground": "#8baa5d" } },
{ "scope": ["meta.preprocessor"], "settings": { "foreground": "#5ea852" } },
{ "scope": ["variable.parameter.preprocessor"], "settings": { "foreground": "#636363" } },
{ "scope": ["keyword.control.directive"], "settings": { "foreground": "#d68a36" } },
{ "scope": ["entity.name.namespace"], "settings": { "foreground": "#8e8e8e" } },
{ "scope": ["entity.name.type.parameter"], "settings": { "foreground": "#b8d7a3" } },
{ "scope": ["variable.other.enummember"], "settings": { "foreground": "#a373b0" } },
{ "scope": ["entity.name.type.concept"], "settings": { "foreground": "#76ff7d" } },
{ "scope": ["entity.name.type.dependent"], "settings": { "foreground": "#448b5a", "fontStyle": "bold" } },
{ "scope": ["entity.name.label"], "settings": { "foreground": "#c8c8c8", "fontStyle": "bold" } },
{ "scope": ["invalid"], "settings": { "foreground": "#ff5647" } },
{ "scope": ["keyword.codetag.todo"], "settings": { "foreground": "#c10000", "fontStyle": "bold italic" } },
{ "scope": ["keyword.control.duffle.atom"], "settings": { "foreground": "#d8bd5b", "fontStyle": "bold" } },
{ "scope": ["entity.name.function.duffle.atom"], "settings": { "foreground": "#cccab5", "fontStyle": "bold" } },
{ "scope": ["keyword.control.duffle.component"], "settings": { "foreground": "#d68a36", "fontStyle": "bold" } },
{ "scope": ["entity.name.function.duffle.component"], "settings": { "foreground": "#6090a9", "fontStyle": "bold" } },
{ "scope": ["support.function.duffle.annotation"], "settings": { "foreground": "#d8bd5b" } },
{ "scope": ["entity.name.type.duffle.bind"], "settings": { "foreground": "#54a4d6" } },
{ "scope": ["entity.name.tag.duffle.phase"], "settings": { "foreground": "#b8d7a3", "fontStyle": "italic" } },
{ "scope": ["entity.name.label.duffle.atom"], "settings": { "foreground": "#c8c8c8", "fontStyle": "bold" } },
{ "scope": ["support.function.duffle.cpu"], "settings": { "foreground": "#6090a9" } },
{ "scope": ["support.function.duffle.gte"], "settings": { "foreground": "#9e95c6" } },
{ "scope": ["support.function.duffle.gpu"], "settings": { "foreground": "#bf7dac" } },
{ "scope": ["support.function.duffle.component"], "settings": { "foreground": "#8baa5d" } },
{ "scope": ["keyword.control.duffle.branch"], "settings": { "foreground": "#76ff7d", "fontStyle": "bold" } },
{ "scope": ["keyword.operator.duffle.delayslot"], "settings": { "foreground": "#ff5647" } },
{ "scope": ["variable.other.constant.duffle.gpr"], "settings": { "foreground": "#bc966c" } },
{ "scope": ["variable.other.constant.duffle.cop2"], "settings": { "foreground": "#9e95c6" } },
{ "scope": ["storage.type.duffle.type"], "settings": { "foreground": "#54a4d6" } },
{ "scope": ["storage.modifier.duffle.attr"], "settings": { "foreground": "#73a07c" } }
]
}
+33
View File
@@ -177,6 +177,39 @@
"tbreak main",
"continue"
]
},
{
"name": "Debug: Hello Camera! (attach only)",
"type": "gdb",
"request": "attach",
"target": "localhost:3333",
"remote": true,
"cwd": "${workspaceRoot}",
"valuesFormatting": "parseText",
"registerLimit": "1-32",
"frameFilters": false,
"showDevDebugOutput": false,
"printCalls": false,
"stopAtConnect": true,
"gdbpath": "gdb-multiarch",
"windows": {
"gdbpath": "gdb-multiarch.exe"
},
"osx": {
"gdbpath": "gdb"
},
"executable": "${workspaceRoot}/build/hello_camera.dwarf-injected.elf",
"setupCommands": [
{ "text": "set mi-async off" },
{ "text": "set remotetimeout 0" },
{ "text": "set logging file build/gen/hello_camera.gdb.log" },
{ "text": "set logging redirect on" }
],
"autorun": [
"source scripts/gdb/gdb_tape_atoms.gdb",
"tbreak hot_reload_entry",
"continue"
]
}
]
}
-199
View File
@@ -1,199 +0,0 @@
"use strict";
const { nearestCall } = require("./lexer");
const { mergeIndexes, scanSource } = require("./source-index");
const TOKEN_TYPES = [
"tapeAtomKeyword",
"tapeAtomName",
"tapeComponentKeyword",
"tapeComponentName",
"tapeAnnotation",
"tapeBindType",
"tapePhase",
"tapeLabel",
"tapeCpuInstruction",
"tapeControlFlow",
"tapeGteInstruction",
"tapeGpuInstruction",
"tapeComponentInstruction",
"tapeDelaySlot",
"tapeGprRegister",
"tapeCop2Register",
"tapeDuffleType",
"tapeAttribute",
];
const TOKEN_MODIFIERS = ["declaration", "tapeRead", "tapeWrite", "tapeAuto"];
const TOKEN_TYPE_INDEX = new Map(TOKEN_TYPES.map((name, index) => [name, index]));
const TOKEN_MODIFIER_INDEX = new Map(TOKEN_MODIFIERS.map((name, index) => [name, index]));
const ATOM_KEYWORDS = new Set(["MipsAtom_", "MipsAtom_Proc_"]);
const COMPONENT_KEYWORDS = new Set(["MipsAtomComp_", "MipsAtomComp_Proc_"]);
const ANNOTATIONS = new Set([
"atom_info", "atom_bind", "atom_reads", "atom_writes", "atom_label",
"atom_offset", "atom_reg", "atom_type", "atom_ctx", "atom_phase",
"atom_auto_reg", "phase_auto_reg", "atom_dbg_skip",
]);
const DELAY_SLOT_KEYWORDS = new Set(["LdSlot_", "BdSlot_"]);
const CONTROL_FLOW_PREFIXES = /^(?:branch_|jump_|call_)/;
const ROLE_TO_TYPE = {
atomName: "tapeAtomName",
componentName: "tapeComponentName",
bindType: "tapeBindType",
duffleType: "tapeDuffleType",
gprRegister: "tapeGprRegister",
cop2Register: "tapeCop2Register",
};
function registerType(name, index) {
const kind = index.registers.get(name);
if (kind === "gpr" || /^R_[A-Za-z0-9_]+$/.test(name)) return "tapeGprRegister";
if (kind === "cop2" || /^(?:C2_|gte_cr_)[A-Za-z0-9_]+$/.test(name)) return "tapeCop2Register";
return null;
}
function instructionType(name, index) {
const domain = index.macros.get(name);
if (domain === "cpu") return "tapeCpuInstruction";
if (domain === "gte") return "tapeGteInstruction";
if (domain === "gpu") return "tapeGpuInstruction";
if (domain === "component") return "tapeComponentInstruction";
if (/^gte_(?!cr_)/.test(name)) return "tapeGteInstruction";
if (/^gp[01]_/.test(name)) return "tapeGpuInstruction";
if (/^mac_gte_/.test(name)) return "tapeGteInstruction";
if (/^mac_gp/.test(name)) return "tapeGpuInstruction";
if (/^mac_/.test(name)) return "tapeCpuInstruction";
return null;
}
function modifierMask(modifiers) {
let mask = 0;
for (const modifier of modifiers) {
const index = TOKEN_MODIFIER_INDEX.get(modifier);
if (index !== undefined) mask |= (1 << index);
}
return mask;
}
function isRegUseAccess(tokens, tokenIndex) {
const prev = tokens[tokenIndex - 1];
if (!prev || prev.text !== ".") return false;
const prevPrev = tokens[tokenIndex - 2];
if (!prevPrev || prevPrev.kind !== "identifier") return false;
const next = tokens[tokenIndex + 1];
if (next && next.text === ".") return false;
if (prevPrev.text === "r") return true;
const prev3 = tokens[tokenIndex - 3];
const prev4 = tokens[tokenIndex - 4];
if (prev3 && prev3.text === "." && prev4 && prev4.kind === "identifier" && prev4.text === "r") return true;
return false;
}
function classifyDocument(source, filePath, workspaceIndex, shouldCancel = () => false) {
const scanned = scanSource(source, filePath);
const index = mergeIndexes(workspaceIndex, scanned.index);
const spans = [];
for (let tokenIndex = 0; tokenIndex < scanned.tokens.length; tokenIndex += 1) {
if (shouldCancel()) break;
const token = scanned.tokens[tokenIndex];
if (token.kind !== "identifier") continue;
let type = null;
let modifiers = [];
const declaration = scanned.declarations.get(token.start);
const context = nearestCall(scanned.contexts, tokenIndex);
if (declaration) {
type = ROLE_TO_TYPE[declaration.role] || null;
modifiers = declaration.modifiers.slice();
} else if (ATOM_KEYWORDS.has(token.text)) {
type = "tapeAtomKeyword";
} else if (COMPONENT_KEYWORDS.has(token.text)) {
type = "tapeComponentKeyword";
} else if (ANNOTATIONS.has(token.text)) {
type = "tapeAnnotation";
} else if (context && context.callee === "atom_bind" && context.argIndex === 0) {
type = "tapeBindType";
} else if (context && context.callee === "atom_phase" && context.argIndex === 0) {
type = "tapePhase";
modifiers = ["declaration"];
} else if (context && context.callee === "atom_ctx" && context.argIndex === 0) {
type = "tapeAtomName";
} else if (context && context.callee === "atom_label" && context.argIndex === 0) {
type = "tapeLabel";
modifiers = ["declaration"];
} else if (context && context.callee === "atom_offset" && context.argIndex <= 1) {
type = "tapeLabel";
} else if (context && context.callee === "atom_reads") {
type = registerType(token.text, index);
if (type) modifiers = ["tapeRead"];
} else if (context && context.callee === "atom_writes") {
type = registerType(token.text, index);
if (type) modifiers = ["tapeWrite"];
} else if (context && context.callee === "atom_auto_reg") {
if (context.argIndex === 0) type = "tapeAtomName";
if (context.argIndex === 1) {
type = "tapeGprRegister";
modifiers = ["declaration", "tapeAuto"];
}
} else if (context && context.callee === "phase_auto_reg") {
if (context.argIndex === 0) type = "tapePhase";
if (context.argIndex === 1) {
type = "tapeGprRegister";
modifiers = ["declaration", "tapeAuto"];
}
}
if (!type && index.bindTypes.has(token.text)) type = "tapeBindType";
if (!type && index.types.has(token.text)) type = "tapeDuffleType";
if (!type && index.attributes.has(token.text)) type = "tapeAttribute";
if (!type) type = registerType(token.text, index);
if (!type && DELAY_SLOT_KEYWORDS.has(token.text)) type = "tapeDelaySlot";
if (!type) {
const domain = index.macros.get(token.text);
if (domain && CONTROL_FLOW_PREFIXES.test(token.text)) {
type = "tapeControlFlow";
}
}
if (!type && isRegUseAccess(scanned.tokens, tokenIndex)) type = "tapeGprRegister";
if (!type) type = instructionType(token.text, index);
if (!type && index.atoms.has(token.text)) type = "tapeAtomName";
if (!type && index.components.has(token.text)) type = "tapeComponentName";
if (!type && index.phases.has(token.text)) type = "tapePhase";
if (!type && index.labels.has(token.text)) type = "tapeLabel";
if (!type) continue;
spans.push({
text: token.text,
type,
typeIndex: TOKEN_TYPE_INDEX.get(type),
modifiers,
modifierMask: modifierMask(modifiers),
start: token.start,
length: token.end - token.start,
line: token.line,
character: token.character,
});
}
spans.sort((left, right) => left.start - right.start || left.length - right.length);
const nonOverlapping = [];
for (const span of spans) {
const previous = nonOverlapping[nonOverlapping.length - 1];
if (!previous || previous.start + previous.length <= span.start) nonOverlapping.push(span);
}
return { spans: nonOverlapping, errors: scanned.errors };
}
module.exports = {
TOKEN_MODIFIERS,
TOKEN_TYPES,
classifyDocument,
modifierMask,
};
-111
View File
@@ -1,111 +0,0 @@
"use strict";
const vscode = require("vscode");
const { TOKEN_MODIFIERS, TOKEN_TYPES, classifyDocument } = require("./classifier");
const { createIndex, mergeIndexes, scanSource } = require("./source-index");
const SOURCE_GLOB = "**/*.{c,h,cc,cpp,cxx,hh,hpp,hxx}";
const EXCLUDE_GLOB = "**/{gen,build,.slop_cache,toolchain,node_modules}/**";
const EXCLUDED_SEGMENTS = new Set(["gen", "build", ".slop_cache", "toolchain", "node_modules"]);
function isExcluded(uri) {
const segments = uri.fsPath.replaceAll("\\", "/").split("/");
return segments.some((segment) => EXCLUDED_SEGMENTS.has(segment));
}
function formatError(filePath, error) {
return `${filePath}:${error.offset}: ${error.kind}`;
}
async function activate(context) {
const output = vscode.window.createOutputChannel("Tape Atom DSL");
const emitter = new vscode.EventEmitter();
const legend = new vscode.SemanticTokensLegend(TOKEN_TYPES, TOKEN_MODIFIERS);
let workspaceIndex = createIndex();
let rebuildGeneration = 0;
let debounceHandle = null;
async function rebuildIndex() {
const generation = ++rebuildGeneration;
const files = await vscode.workspace.findFiles(SOURCE_GLOB, EXCLUDE_GLOB);
let nextIndex = createIndex();
for (const uri of files) {
if (generation !== rebuildGeneration) return;
if (isExcluded(uri)) continue;
try {
const bytes = await vscode.workspace.fs.readFile(uri);
const source = Buffer.from(bytes).toString("utf8");
const result = scanSource(source, uri.fsPath);
nextIndex = mergeIndexes(nextIndex, result.index);
for (const error of result.errors) output.appendLine(formatError(uri.fsPath, error));
} catch (error) {
output.appendLine(`${uri.fsPath}: ${error.stack || error.message || error}`);
}
}
if (generation !== rebuildGeneration) return;
workspaceIndex = nextIndex;
emitter.fire();
}
function scheduleRebuild(uri) {
if (uri && isExcluded(uri)) return;
if (debounceHandle !== null) clearTimeout(debounceHandle);
debounceHandle = setTimeout(() => {
debounceHandle = null;
rebuildIndex().catch((error) => output.appendLine(error.stack || String(error)));
}, 100);
}
const provider = {
onDidChangeSemanticTokens: emitter.event,
provideDocumentSemanticTokens(document, cancellationToken) {
try {
const result = classifyDocument(
document.getText(),
document.uri.fsPath,
workspaceIndex,
() => cancellationToken.isCancellationRequested
);
const builder = new vscode.SemanticTokensBuilder(legend);
for (const span of result.spans) {
if (cancellationToken.isCancellationRequested) break;
builder.push(span.line, span.character, span.length, span.typeIndex, span.modifierMask);
}
for (const error of result.errors) {
output.appendLine(formatError(document.uri.fsPath || document.uri.toString(), error));
}
return builder.build();
} catch (error) {
output.appendLine(`${document.uri}: ${error.stack || error.message || error}`);
return new vscode.SemanticTokensBuilder(legend).build();
}
},
};
const selector = [
{ language: "c", scheme: "file" },
{ language: "c", scheme: "untitled" },
{ language: "cpp", scheme: "file" },
{ language: "cpp", scheme: "untitled" },
];
const watcher = vscode.workspace.createFileSystemWatcher(SOURCE_GLOB);
context.subscriptions.push(
output,
emitter,
watcher,
watcher.onDidCreate(scheduleRebuild),
watcher.onDidChange(scheduleRebuild),
watcher.onDidDelete(scheduleRebuild),
vscode.languages.registerDocumentSemanticTokensProvider(selector, provider, legend),
{ dispose() { if (debounceHandle !== null) clearTimeout(debounceHandle); } }
);
await rebuildIndex();
}
function deactivate() {}
module.exports = { activate, deactivate };
-186
View File
@@ -1,186 +0,0 @@
"use strict";
function isIdentifierStart(code) {
return code === 95 ||
(code >= 65 && code <= 90) ||
(code >= 97 && code <= 122);
}
function isIdentifierContinue(code) {
return isIdentifierStart(code) || (code >= 48 && code <= 57);
}
function lex(source) {
if (typeof source !== "string") throw new TypeError("source must be a string");
const tokens = [];
const errors = [];
let offset = 0;
let line = 0;
let character = 0;
function advance() {
if (source[offset] === "\r" && source[offset + 1] === "\n") {
offset += 2;
line += 1;
character = 0;
return;
}
if (source[offset] === "\n") {
offset += 1;
line += 1;
character = 0;
return;
}
offset += 1;
character += 1;
}
function pushToken(kind, start, startLine, startCharacter) {
tokens.push({
kind,
text: source.slice(start, offset),
start,
end: offset,
line: startLine,
character: startCharacter,
});
}
while (offset < source.length) {
const ch = source[offset];
if (/\s/.test(ch)) {
advance();
continue;
}
if (ch === "/" && source[offset + 1] === "/") {
while (offset < source.length && source[offset] !== "\r" && source[offset] !== "\n") advance();
continue;
}
if (ch === "/" && source[offset + 1] === "*") {
const start = offset;
advance();
advance();
let closed = false;
while (offset < source.length) {
if (source[offset] === "*" && source[offset + 1] === "/") {
advance();
advance();
closed = true;
break;
}
advance();
}
if (!closed) errors.push({ kind: "unterminated-block-comment", offset: start });
continue;
}
if (ch === "\"" || ch === "'") {
const quote = ch;
const start = offset;
advance();
let closed = false;
while (offset < source.length) {
if (source[offset] === "\\") {
advance();
if (offset < source.length) advance();
continue;
}
if (source[offset] === quote) {
advance();
closed = true;
break;
}
if (source[offset] === "\n" || source[offset] === "\r") break;
advance();
}
if (!closed) errors.push({ kind: "unterminated-literal", offset: start });
continue;
}
const code = source.charCodeAt(offset);
if (isIdentifierStart(code)) {
const start = offset;
const startLine = line;
const startCharacter = character;
advance();
while (offset < source.length && isIdentifierContinue(source.charCodeAt(offset))) advance();
pushToken("identifier", start, startLine, startCharacter);
continue;
}
const start = offset;
const startLine = line;
const startCharacter = character;
advance();
pushToken("punctuation", start, startLine, startCharacter);
}
return { tokens, errors };
}
function buildCallContexts(tokens) {
const contexts = Array.from({ length: tokens.length }, () => []);
const calls = [];
const errors = [];
const stack = [];
for (let tokenIndex = 0; tokenIndex < tokens.length; tokenIndex += 1) {
const token = tokens[tokenIndex];
if (token.text === ")") {
if (stack.length === 0) {
errors.push({ kind: "unmatched-close-paren", offset: token.start });
} else {
const frame = stack.pop();
if (frame.callee !== null) calls.push({ ...frame, closeTokenIndex: tokenIndex });
}
}
contexts[tokenIndex] = stack
.filter((frame) => frame.callee !== null)
.map((frame) => ({
callee: frame.callee,
calleeTokenIndex: frame.calleeTokenIndex,
openTokenIndex: frame.openTokenIndex,
argIndex: frame.argIndex,
}));
if (token.text === "(") {
const previous = tokens[tokenIndex - 1];
const hasCallee = previous && previous.kind === "identifier";
stack.push({
callee: hasCallee ? previous.text : null,
calleeTokenIndex: hasCallee ? tokenIndex - 1 : -1,
openTokenIndex: tokenIndex,
argIndex: 0,
});
continue;
}
if (token.text === "," && stack.length > 0) {
const frame = stack[stack.length - 1];
if (frame.callee !== null) frame.argIndex += 1;
}
}
for (const frame of stack) {
errors.push({ kind: "unmatched-open-paren", offset: tokens[frame.openTokenIndex].start });
}
return { contexts, calls, errors };
}
function nearestCall(contexts, tokenIndex, callee) {
const entries = contexts[tokenIndex] || [];
for (let contextIndex = entries.length - 1; contextIndex >= 0; contextIndex -= 1) {
const entry = entries[contextIndex];
if (callee === undefined || entry.callee === callee) return entry;
}
return null;
}
module.exports = { buildCallContexts, lex, nearestCall };
-81
View File
@@ -1,81 +0,0 @@
{
"name": "tape-atom-syntax",
"displayName": "Tape Atom DSL",
"description": "Source-derived semantic highlighting for the Tape/Atom MIPS macro DSL",
"publisher": "local",
"version": "0.2.0",
"engines": { "vscode": "^1.80.0" },
"categories": ["Programming Languages"],
"activationEvents": ["onLanguage:c", "onLanguage:cpp"],
"main": "./extension.js",
"files": [
"classifier.js",
"extension.js",
"lexer.js",
"source-index.js",
"syntaxes/tape_atom.tmLanguage.json"
],
"scripts": {
"test": "node --test test/*.test.js",
"package": "npx --yes @vscode/vsce@3.6.1 package --allow-missing-repository --skip-license --out tape-atom-syntax-0.2.0.vsix"
},
"contributes": {
"semanticTokenTypes": [
{ "id": "tapeAtomKeyword", "superType": "keyword", "description": "Tape atom declaration keyword" },
{ "id": "tapeAtomName", "superType": "function", "description": "Tape atom name" },
{ "id": "tapeComponentKeyword", "superType": "keyword", "description": "Tape atom component declaration keyword" },
{ "id": "tapeComponentName", "superType": "function", "description": "Tape atom component name" },
{ "id": "tapeAnnotation", "superType": "macro", "description": "Tape atom annotation" },
{ "id": "tapeBindType", "superType": "type", "description": "Tape bind structure type" },
{ "id": "tapePhase", "superType": "label", "description": "Tape atom phase" },
{ "id": "tapeLabel", "superType": "label", "description": "Tape atom branch label" },
{ "id": "tapeCpuInstruction", "superType": "macro", "description": "MIPS CPU instruction emitter" },
{ "id": "tapeControlFlow", "superType": "keyword", "description": "MIPS branch or jump instruction" },
{ "id": "tapeGteInstruction", "superType": "macro", "description": "GTE instruction emitter" },
{ "id": "tapeGpuInstruction", "superType": "macro", "description": "GPU command emitter" },
{ "id": "tapeComponentInstruction", "superType": "macro", "description": "Tape atom component invocation" },
{ "id": "tapeDelaySlot", "superType": "keyword", "description": "Load or branch delay slot annotation" },
{ "id": "tapeGprRegister", "superType": "variable", "description": "MIPS GPR alias" },
{ "id": "tapeCop2Register", "superType": "variable", "description": "COP2 data or control register alias" },
{ "id": "tapeDuffleType", "superType": "type", "description": "Duffle type or type constructor" },
{ "id": "tapeAttribute", "superType": "keyword", "description": "Duffle linkage or storage attribute" }
],
"semanticTokenModifiers": [
{ "id": "tapeRead", "description": "Register declared in atom_reads" },
{ "id": "tapeWrite", "description": "Register declared in atom_writes" },
{ "id": "tapeAuto", "description": "Auto-allocated register" }
],
"semanticTokenScopes": [
{
"language": "c",
"scopes": {
"tapeAtomKeyword": ["keyword.control.duffle.atom"],
"tapeAtomName": ["entity.name.function.duffle.atom"],
"tapeComponentKeyword": ["keyword.control.duffle.component"],
"tapeComponentName": ["entity.name.function.duffle.component"],
"tapeAnnotation": ["support.function.duffle.annotation"],
"tapeBindType": ["entity.name.type.duffle.bind"],
"tapePhase": ["entity.name.tag.duffle.phase"],
"tapeLabel": ["entity.name.label.duffle.atom"],
"tapeCpuInstruction": ["support.function.duffle.cpu"],
"tapeControlFlow": ["keyword.control.duffle.branch"],
"tapeGteInstruction": ["support.function.duffle.gte"],
"tapeGpuInstruction": ["support.function.duffle.gpu"],
"tapeComponentInstruction": ["support.function.duffle.component"],
"tapeDelaySlot": ["keyword.operator.duffle.delayslot"],
"tapeGprRegister": ["variable.other.constant.duffle.gpr"],
"tapeCop2Register": ["variable.other.constant.duffle.cop2"],
"tapeDuffleType": ["storage.type.duffle.type"],
"tapeAttribute": ["storage.modifier.duffle.attr"]
}
}
],
"grammars": [
{
"scopeName": "tape_atom.injection",
"path": "./syntaxes/tape_atom.tmLanguage.json",
"injectTo": ["source.c", "source.cpp"]
}
]
}
}
-234
View File
@@ -1,234 +0,0 @@
"use strict";
const path = require("node:path");
const { buildCallContexts, lex, nearestCall } = require("./lexer");
const BASE_TYPES = [
"B1", "B2", "B4", "B8", "F4", "F8", "S1", "S2", "S4", "S8",
"U1", "U2", "U4", "U8", "MipsAtom", "MipsCode", "Reg",
];
const BASE_ATTRIBUTES = [
"FI_", "I_", "NI_", "Relative_", "Struct_", "Enum_", "Union_",
"TypeR_", "TypeV_", "align_", "internal", "local_persist", "global",
];
function createIndex() {
return {
atoms: new Set(),
components: new Set(),
componentAliases: new Set(),
macros: new Map(),
registers: new Map(),
bindTypes: new Set(),
types: new Set(BASE_TYPES),
phases: new Set(),
labels: new Set(),
attributes: new Set(BASE_ATTRIBUTES),
};
}
function cloneIndex(source) {
const result = createIndex();
for (const key of ["atoms", "components", "componentAliases", "bindTypes", "types", "phases", "labels", "attributes"]) {
for (const value of source[key]) result[key].add(value);
}
for (const [name, domain] of source.macros) result.macros.set(name, domain);
for (const [name, domain] of source.registers) result.registers.set(name, domain);
return result;
}
function mergeIndexes(...sources) {
const result = createIndex();
for (const source of sources) {
if (!source) continue;
for (const key of ["atoms", "components", "componentAliases", "bindTypes", "types", "phases", "labels", "attributes"]) {
for (const value of source[key]) result[key].add(value);
}
for (const [name, domain] of source.macros) result.macros.set(name, domain);
for (const [name, domain] of source.registers) result.registers.set(name, domain);
}
return result;
}
function domainFromPath(filePath) {
const base = path.basename(filePath.replaceAll("\\", "/")).toLowerCase();
if (base === "mips.h" || base === "mips.atom.c") return "cpu";
if (base === "gte.h" || base === "gte.atom.c") return "gte";
if (base === "gp.h" || base === "gp.atom.c") return "gpu";
return "component";
}
function registerKind(name) {
if (/^R_[A-Za-z0-9_]+$/.test(name)) return "gpr";
if (/^(?:C2_|gte_cr_)[A-Za-z0-9_]+$/.test(name)) return "cop2";
return null;
}
function componentAlias(name) {
return name.startsWith("ac_") ? `mac_${name.slice(3)}` : null;
}
function findFunctionNameBefore(tokens, calleeTokenIndex) {
let closeIndex = calleeTokenIndex - 1;
while (closeIndex >= 0 && tokens[closeIndex].kind === "identifier" && tokens[closeIndex].text === "atom_dbg_skip") {
closeIndex -= 1;
}
if (!tokens[closeIndex] || tokens[closeIndex].text !== ")") return null;
let depth = 1;
for (let tokenIndex = closeIndex - 1; tokenIndex >= 0; tokenIndex -= 1) {
if (tokens[tokenIndex].text === ")") depth += 1;
if (tokens[tokenIndex].text === "(") depth -= 1;
if (depth !== 0) continue;
const name = tokens[tokenIndex - 1];
return name && name.kind === "identifier" ? name : null;
}
return null;
}
function scanSource(source, filePath) {
const lexical = lex(source);
const balanced = buildCallContexts(lexical.tokens);
const tokens = lexical.tokens;
const contexts = balanced.contexts;
const index = createIndex();
const declarations = new Map();
const domain = domainFromPath(filePath);
function mark(token, role, modifiers = ["declaration"]) {
declarations.set(token.start, { role, modifiers });
}
function componentDomain(name) {
if (name.startsWith("ac_gte_") || name.startsWith("mac_gte_")) return "gte";
if (name.startsWith("ac_gp_") || name.startsWith("mac_gp_")) return "gpu";
return "cpu";
}
function addComponent(token) {
index.components.add(token.text);
mark(token, "componentName");
const alias = componentAlias(token.text);
if (alias) {
index.componentAliases.add(alias);
index.macros.set(alias, componentDomain(token.text));
}
}
for (let tokenIndex = 0; tokenIndex < tokens.length; tokenIndex += 1) {
const token = tokens[tokenIndex];
if (token.kind !== "identifier") continue;
const kind = registerKind(token.text);
if (kind) {
index.registers.set(token.text, kind);
if (tokens[tokenIndex + 1] && tokens[tokenIndex + 1].text === "=") {
mark(token, kind === "gpr" ? "gprRegister" : "cop2Register");
}
}
const context = nearestCall(contexts, tokenIndex);
if (context && context.argIndex === 0) {
if (context.callee === "MipsAtom_") {
index.atoms.add(token.text);
mark(token, "atomName");
}
if (context.callee === "MipsAtomComp_") addComponent(token);
if (context.callee === "atom_bind") index.bindTypes.add(token.text);
if (context.callee === "atom_phase" || context.callee === "phase_auto_reg") index.phases.add(token.text);
if (context.callee === "atom_label" || context.callee === "atom_offset") index.labels.add(token.text);
}
const isWrappedType = context && (
((context.callee === "Struct_" || context.callee === "Union_") && context.argIndex === 0) ||
(context.callee === "Enum_" && context.argIndex === 1)
);
if (isWrappedType) {
index.types.add(token.text);
mark(token, token.text.startsWith("Binds_") ? "bindType" : "duffleType");
if (token.text.startsWith("Binds_")) index.bindTypes.add(token.text);
}
if (context && context.callee === "atom_offset" && context.argIndex === 1) index.labels.add(token.text);
if (context && context.callee === "atom_auto_reg") {
if (context.argIndex === 0) index.atoms.add(token.text);
if (context.argIndex === 1) {
index.registers.set(token.text, "gpr");
mark(token, "gprRegister", ["declaration", "tapeAuto"]);
}
}
if (context && context.callee === "phase_auto_reg" && context.argIndex === 1) {
index.registers.set(token.text, "gpr");
mark(token, "gprRegister", ["declaration", "tapeAuto"]);
}
if (token.text === "define" && tokens[tokenIndex - 1] && tokens[tokenIndex - 1].text === "#") {
const name = tokens[tokenIndex + 1];
if (name && name.kind === "identifier" && name.line === token.line) {
if (/^(?:RegUse_|Struct_|Enum_|Union_|TypeR_|TypeV_|Relative_|Binds_)/.test(name.text)) {
index.types.add(name.text);
} else {
index.macros.set(name.text, domain);
}
}
}
if (token.text === "typedef") {
let endIndex = tokenIndex + 1;
let hasBrace = false;
let lastIdentifier = null;
while (endIndex < tokens.length && tokens[endIndex].text !== ";") {
if (tokens[endIndex].text === "{") hasBrace = true;
if (tokens[endIndex].kind === "identifier") lastIdentifier = tokens[endIndex];
endIndex += 1;
}
if (!hasBrace && lastIdentifier) {
index.types.add(lastIdentifier.text);
mark(lastIdentifier, "duffleType");
}
}
if (token.text === "MipsAtom_Proc_") {
const functionName = findFunctionNameBefore(tokens, tokenIndex);
if (functionName) {
const atomName = functionName.text.endsWith("_proc")
? functionName.text.slice(0, -5)
: functionName.text;
index.atoms.add(atomName);
index.atoms.add(functionName.text);
mark(functionName, "atomName");
}
}
if (token.text === "MipsAtomComp_Proc_") {
const functionName = findFunctionNameBefore(tokens, tokenIndex);
if (functionName) addComponent(functionName);
}
}
for (const call of balanced.calls) {
if (domain === "component") continue;
const name = tokens[call.calleeTokenIndex];
const after = tokens[call.closeTokenIndex + 1];
if (!name || !after || after.text !== "{") continue;
if (/^(?:gp0_|gp1_|gte_|mac_)/.test(name.text)) index.macros.set(name.text, domain);
}
return {
index: cloneIndex(index),
declarations,
tokens,
contexts,
errors: [...lexical.errors, ...balanced.errors],
};
}
module.exports = {
createIndex,
domainFromPath,
mergeIndexes,
scanSource,
};
@@ -1,71 +0,0 @@
{
"scopeName": "tape_atom.injection",
"injectionSelector": "L:source.c -comment -string, L:source.cpp -comment -string",
"patterns": [
{ "include": "#atom-declarations" },
{ "include": "#component-declarations" },
{ "include": "#annotation-arguments" },
{ "include": "#annotations" },
{ "include": "#delay-slots" },
{ "include": "#types" },
{ "include": "#attributes" }
],
"repository": {
"atom-declarations": {
"patterns": [
{
"match": "\\b(MipsAtom_)\\s*\\(\\s*([A-Za-z_][A-Za-z0-9_]*)",
"captures": {
"1": { "name": "keyword.control.duffle.atom" },
"2": { "name": "entity.name.function.duffle.atom" }
}
},
{ "match": "\\bMipsAtom_Proc_\\b", "name": "keyword.control.duffle.atom" },
{ "match": "\\b[A-Za-z_][A-Za-z0-9_]*_proc\\b", "name": "entity.name.function.duffle.atom" }
]
},
"component-declarations": {
"patterns": [
{
"match": "\\b(MipsAtomComp_)\\s*\\(\\s*(ac_[A-Za-z0-9_]*)",
"captures": {
"1": { "name": "keyword.control.duffle.component" },
"2": { "name": "entity.name.function.duffle.component" }
}
},
{ "match": "\\bMipsAtomComp_Proc_\\b", "name": "keyword.control.duffle.component" }
]
},
"annotation-arguments": {
"patterns": [
{
"match": "\\b(atom_offset)\\s*\\(\\s*([A-Za-z_][A-Za-z0-9_]*)\\s*,\\s*([A-Za-z_][A-Za-z0-9_]*)",
"captures": {
"1": { "name": "support.function.duffle.annotation" },
"2": { "name": "entity.name.label.duffle.atom" },
"3": { "name": "entity.name.label.duffle.atom" }
}
},
{ "match": "(?<=\\batom_bind\\()\\s*Binds_[A-Za-z0-9_]+", "name": "entity.name.type.duffle.bind" },
{ "match": "(?<=\\batom_phase\\()\\s*[A-Za-z_][A-Za-z0-9_]*", "name": "entity.name.tag.duffle.phase" },
{ "match": "(?<=\\batom_label\\()\\s*[A-Za-z_][A-Za-z0-9_]*", "name": "entity.name.label.duffle.atom" }
]
},
"annotations": {
"match": "\\b(atom_info|atom_bind|atom_reads|atom_writes|atom_label|atom_offset|atom_reg|atom_type|atom_ctx|atom_phase|atom_auto_reg|phase_auto_reg|atom_dbg_skip)\\b",
"name": "support.function.duffle.annotation"
},
"delay-slots": {
"match": "\\b(LdSlot_|BdSlot_)\\b",
"name": "keyword.operator.duffle.delayslot"
},
"types": {
"match": "\\b(?:Binds_[A-Za-z0-9_]+|RegUse_[A-Za-z0-9_]+)\\b",
"name": "storage.type.duffle.type"
},
"attributes": {
"match": "\\b(?:FI_|I_|NI_|Relative_|align_|internal|local_persist|global)\\b",
"name": "storage.modifier.duffle.attr"
}
}
}
Binary file not shown.
-83
View File
@@ -1,83 +0,0 @@
"use strict";
const assert = require("node:assert/strict");
const test = require("node:test");
const { classifyDocument } = require("../classifier");
const { createIndex } = require("../source-index");
function byText(result, text) {
return result.spans.filter((span) => span.text === text);
}
test("classifyDocument distinguishes declaration, annotation, phase, bind, and label roles", () => {
const source = [
"typedef Struct_(Binds_CubeTri) { U4 PrimCursor; };",
"MipsAtom_(cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4),",
"\tatom_reads(R_PrimCursor), atom_writes(R_FaceCursor)) {",
"\tbranch_le_zero(R_T0, atom_offset(cull, exit)),",
"\tatom_label(exit)",
"};",
].join("\n");
const result = classifyDocument(source, "C:/x/code/hello_camera/hello_camera.atom.c", createIndex());
assert.equal(byText(result, "MipsAtom_")[0].type, "tapeAtomKeyword");
assert.deepEqual(byText(result, "cube_g4_face")[0].modifiers, ["declaration"]);
assert.equal(byText(result, "atom_bind")[0].type, "tapeAnnotation");
assert.equal(byText(result, "Binds_CubeTri").at(-1).type, "tapeBindType");
assert.equal(byText(result, "cube_g4")[0].type, "tapePhase");
assert.deepEqual(byText(result, "cube_g4")[0].modifiers, ["declaration"]);
assert.equal(byText(result, "cull")[0].type, "tapeLabel");
assert.equal(byText(result, "exit").every((span) => span.type === "tapeLabel"), true);
});
test("classifyDocument applies read and write modifiers to GPRs", () => {
const source = "atom_info(atom_reads(R_PrimCursor), atom_writes(R_FaceCursor))";
const result = classifyDocument(source, "C:/x/code/test.atom.c", createIndex());
assert.deepEqual(byText(result, "R_PrimCursor")[0].modifiers, ["tapeRead"]);
assert.deepEqual(byText(result, "R_FaceCursor")[0].modifiers, ["tapeWrite"]);
});
test("classifyDocument separates CPU, GTE, GPU, and component domains", () => {
const workspace = createIndex();
workspace.macros.set("load_word", "cpu");
workspace.macros.set("gte_cmdw_rtpt", "gte");
workspace.macros.set("gp1_word_DisplayOn", "gpu");
workspace.macros.set("mac_yield", "component");
workspace.componentAliases.add("mac_yield");
const source = "load_word(R_T0, R_T1, 0), gte_cmdw_rtpt, gp1_word_DisplayOn(), mac_yield(), C2_MAC0, gte_cr_OFX_Code";
const result = classifyDocument(source, "C:/x/code/test.c", workspace);
assert.equal(byText(result, "load_word")[0].type, "tapeCpuInstruction");
assert.equal(byText(result, "gte_cmdw_rtpt")[0].type, "tapeGteInstruction");
assert.equal(byText(result, "gp1_word_DisplayOn")[0].type, "tapeGpuInstruction");
assert.equal(byText(result, "mac_yield")[0].type, "tapeComponentInstruction");
assert.equal(byText(result, "C2_MAC0")[0].type, "tapeCop2Register");
assert.equal(byText(result, "gte_cr_OFX_Code")[0].type, "tapeCop2Register");
});
test("document-local declarations override an empty workspace index", () => {
const source = [
"MipsAtomComp_(ac_new_component) { nop };",
"mac_new_component(),",
].join("\n");
const result = classifyDocument(source, "C:/x/code/duffle/math.atom.c", createIndex());
assert.equal(byText(result, "ac_new_component")[0].type, "tapeComponentName");
assert.equal(byText(result, "mac_new_component")[0].type, "tapeCpuInstruction");
});
test("classifier returns ordered non-overlapping spans and partial malformed output", () => {
const source = "atom_reads(R_A /* broken";
const result = classifyDocument(source, "C:/x/code/test.atom.c", createIndex());
assert.equal(result.errors.some((error) => error.kind === "unterminated-block-comment"), true);
for (let spanIndex = 1; spanIndex < result.spans.length; spanIndex += 1) {
const previous = result.spans[spanIndex - 1];
const current = result.spans[spanIndex];
assert.equal(previous.start + previous.length <= current.start, true);
}
});
-91
View File
@@ -1,91 +0,0 @@
"use strict";
const assert = require("node:assert/strict");
const fs = require("node:fs");
const path = require("node:path");
const test = require("node:test");
const { TOKEN_MODIFIERS, TOKEN_TYPES } = require("../classifier");
const ROOT = path.resolve(__dirname, "..");
function readJson(filePath) {
const raw = fs.readFileSync(filePath, "utf8");
const stripped = raw.replace(/\/\/.*$/gm, "").replace(/,\s*([}\]])/g, "$1");
return JSON.parse(stripped);
}
function collectScopeNames(value, output = new Set()) {
if (Array.isArray(value)) {
for (const entry of value) collectScopeNames(entry, output);
return output;
}
if (!value || typeof value !== "object") return output;
if (typeof value.name === "string") output.add(value.name);
for (const child of Object.values(value)) collectScopeNames(child, output);
return output;
}
test("package semantic legend matches classifier exports", () => {
const packageJson = readJson(path.join(ROOT, "package.json"));
const contributedTypes = packageJson.contributes.semanticTokenTypes.map((entry) => entry.id);
const contributedModifiers = packageJson.contributes.semanticTokenModifiers.map((entry) => entry.id);
assert.equal(packageJson.version, "0.2.0");
assert.deepEqual(contributedTypes, TOKEN_TYPES);
assert.deepEqual(contributedModifiers, TOKEN_MODIFIERS.filter((name) => name !== "declaration"));
});
test("package includes runtime files only and acknowledges local-only metadata", () => {
const packageJson = readJson(path.join(ROOT, "package.json"));
assert.deepEqual(packageJson.files, [
"classifier.js",
"extension.js",
"lexer.js",
"source-index.js",
"syntaxes/tape_atom.tmLanguage.json",
]);
assert.equal(packageJson.scripts.package.includes("--allow-missing-repository"), true);
assert.equal(packageJson.scripts.package.includes("--skip-license"), true);
});
test("every semantic token has a scope mapping; DSL-specific tokens also have grammar scopes", () => {
const packageJson = readJson(path.join(ROOT, "package.json"));
const grammar = readJson(path.join(ROOT, "syntaxes", "tape_atom.tmLanguage.json"));
const mappings = packageJson.contributes.semanticTokenScopes[0].scopes;
const grammarScopes = collectScopeNames(grammar);
const grammarRequired = new Set([
"tapeAtomKeyword", "tapeAtomName", "tapeComponentKeyword", "tapeComponentName",
"tapeAnnotation", "tapeBindType", "tapePhase", "tapeLabel",
"tapeDelaySlot", "tapeDuffleType", "tapeAttribute",
]);
for (const tokenType of TOKEN_TYPES) {
assert.equal(Array.isArray(mappings[tokenType]), true, `missing scope mapping: ${tokenType}`);
if (grammarRequired.has(tokenType)) {
assert.equal(mappings[tokenType].some((scope) => grammarScopes.has(scope)), true, `grammar does not emit: ${tokenType}`);
}
}
});
test("TextMate offset labels stay scoped to atom_offset calls", () => {
const grammar = readJson(path.join(ROOT, "syntaxes", "tape_atom.tmLanguage.json"));
const serialized = JSON.stringify(grammar);
const offsetRule = grammar.repository["annotation-arguments"].patterns
.find((rule) => rule.match.includes("atom_offset"));
assert.equal(serialized.includes("(?<=,)"), false);
assert.equal(offsetRule.captures[1].name, "support.function.duffle.annotation");
assert.equal(offsetRule.captures[2].name, "entity.name.label.duffle.atom");
assert.equal(offsetRule.captures[3].name, "entity.name.label.duffle.atom");
});
test("workspace enables semantic highlighting", () => {
const settings = readJson(path.resolve(ROOT, "..", "settings.json"));
assert.equal(settings["editor.semanticHighlighting.enabled"], true);
const semantic = settings["editor.semanticTokenColorCustomizations"];
assert.equal(semantic.enabled, true);
});
-77
View File
@@ -1,77 +0,0 @@
"use strict";
const assert = require("node:assert/strict");
const test = require("node:test");
const { buildCallContexts, lex, nearestCall } = require("../lexer");
test("lex skips comments, strings, and character literals", () => {
const source = [
"MipsAtom_(visible)",
"// MipsAtom_(line_comment)",
"const char *s = \"atom_reads(R_Hidden)\";",
"char c = '\\''; /* gte_cmdw_hidden */",
"atom_reads(R_Visible)",
].join("\n");
const result = lex(source);
const identifiers = result.tokens
.filter((token) => token.kind === "identifier")
.map((token) => token.text);
assert.deepEqual(result.errors, []);
assert.equal(identifiers.includes("visible"), true);
assert.equal(identifiers.includes("R_Visible"), true);
assert.equal(identifiers.includes("line_comment"), false);
assert.equal(identifiers.includes("R_Hidden"), false);
assert.equal(identifiers.includes("gte_cmdw_hidden"), false);
});
test("lex reports unterminated block comments without returning comment tokens", () => {
const result = lex("R_Visible /* atom_reads(R_Hidden)");
assert.equal(result.tokens.some((token) => token.text === "R_Visible"), true);
assert.equal(result.tokens.some((token) => token.text === "R_Hidden"), false);
assert.deepEqual(result.errors.map((error) => error.kind), ["unterminated-block-comment"]);
});
test("line comments stop at CRLF boundaries", () => {
const result = lex("// atom_reads(R_Hidden)\r\natom_reads(R_Visible)\r\n");
const identifiers = result.tokens
.filter((token) => token.kind === "identifier")
.map((token) => token.text);
assert.equal(identifiers.includes("R_Hidden"), false);
assert.equal(identifiers.includes("R_Visible"), true);
});
test("balanced contexts retain multiline nesting and argument indexes", () => {
const source = [
"atom_info(",
"\tatom_phase(cube_g4),",
"\tatom_reads(R_A, nested(R_B, R_C)),",
"\tatom_writes(R_D)",
")",
].join("\n");
const lexical = lex(source);
const balanced = buildCallContexts(lexical.tokens);
const byText = new Map();
lexical.tokens.forEach((token, index) => {
if (token.kind === "identifier") byText.set(token.text, index);
});
assert.equal(nearestCall(balanced.contexts, byText.get("cube_g4")).callee, "atom_phase");
assert.equal(nearestCall(balanced.contexts, byText.get("R_A")).callee, "atom_reads");
assert.equal(nearestCall(balanced.contexts, byText.get("R_A")).argIndex, 0);
assert.equal(nearestCall(balanced.contexts, byText.get("R_C")).callee, "nested");
assert.equal(nearestCall(balanced.contexts, byText.get("R_D")).callee, "atom_writes");
assert.deepEqual(balanced.errors, []);
});
test("balanced contexts report unmatched parentheses", () => {
const lexical = lex("atom_reads(R_A");
const balanced = buildCallContexts(lexical.tokens);
assert.deepEqual(balanced.errors.map((error) => error.kind), ["unmatched-open-paren"]);
});
-77
View File
@@ -1,77 +0,0 @@
"use strict";
const assert = require("node:assert/strict");
const test = require("node:test");
const {
createIndex,
domainFromPath,
mergeIndexes,
scanSource,
} = require("../source-index");
test("scanSource discovers current atom and component forms", () => {
const source = [
"MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4), atom_reads(R_PrimCursor)) { mac_yield() };",
"MipsAtomComp_(ac_load_pair) { load_word(R_T0, R_T1, 0) };",
"internal MipsAtom* normalize_proc(AtomArena_R aa) MipsAtom_Proc_(aa, { mac_yield() })",
"FI_ void ac_store_pair(MipsAtomBuilder_R ab) atom_dbg_skip MipsAtomComp_Proc_(ab, { store_word(R_T0, R_T1, 0) })",
].join("\n");
const result = scanSource(source, "C:/projects/Pikuma/ps1/code/duffle/mips.atom.c");
assert.equal(result.index.atoms.has("cube_g4_face"), true);
assert.equal(result.index.atoms.has("normalize"), true);
assert.equal(result.index.components.has("ac_load_pair"), true);
assert.equal(result.index.components.has("ac_store_pair"), true);
assert.equal(result.index.componentAliases.has("mac_load_pair"), true);
assert.equal(result.index.componentAliases.has("mac_store_pair"), true);
assert.equal(result.index.macros.get("mac_store_pair"), "cpu");
assert.equal(result.index.phases.has("cube_g4"), true);
assert.equal(result.index.registers.get("R_PrimCursor"), "gpr");
assert.deepEqual(result.errors, []);
});
test("scanSource discovers binds, labels, registers, typedefs, and macro domains", () => {
const source = [
"typedef Struct_(Binds_CubeTri) { U4 PrimCursor; };",
"typedef Enum_(U4, PadStatus) { PadStatus_Ok };",
"typedef U4 const MipsCode;",
"enum { R_PrimCursor = R_T7 atom_reg, C2_Custom = 12, gte_cr_Custom = 13 };",
"#define load_word(rt, base, off) enc_i(rt, base, off)",
"atom_bind(Binds_CubeTri)",
"atom_label(exit)",
"atom_offset(entry, exit)",
].join("\n");
const result = scanSource(source, "C:/projects/Pikuma/ps1/code/duffle/mips.h");
assert.equal(result.index.bindTypes.has("Binds_CubeTri"), true);
assert.equal(result.index.types.has("PadStatus"), true);
assert.equal(result.index.types.has("MipsCode"), true);
assert.equal(result.index.registers.get("R_PrimCursor"), "gpr");
assert.equal(result.index.registers.get("C2_Custom"), "cop2");
assert.equal(result.index.registers.get("gte_cr_Custom"), "cop2");
assert.equal(result.index.macros.get("load_word"), "cpu");
assert.equal(result.index.labels.has("entry"), true);
assert.equal(result.index.labels.has("exit"), true);
});
test("domainFromPath uses the declaration file rather than parent directory names", () => {
assert.equal(domainFromPath("C:/x/code/hello_gte/hello_gte.atom.c"), "component");
assert.equal(domainFromPath("C:/x/code/duffle/mips.h"), "cpu");
assert.equal(domainFromPath("C:/x/code/duffle/gte.atom.c"), "gte");
assert.equal(domainFromPath("C:/x/code/duffle/gp.h"), "gpu");
});
test("mergeIndexes preserves domain-specific aliases", () => {
const left = createIndex();
left.macros.set("load_word", "cpu");
const right = createIndex();
right.componentAliases.add("mac_gte_store");
right.macros.set("mac_gte_store", "gte");
const merged = mergeIndexes(left, right);
assert.equal(merged.macros.get("load_word"), "cpu");
assert.equal(merged.macros.get("mac_gte_store"), "gte");
});
+21 -14
View File
@@ -1,17 +1,24 @@
Copyright (C) 2026 Edward R. Gonzalez
This is free and unencumbered software released into the public domain.
This software is provided 'as-is', without any express or implied
warranty. In no event will the authors be held liable for any damages
arising from the use of this software.
Anyone is free to copy, modify, publish, use, compile, sell, or
distribute this software, either in source code form or as a compiled
binary, for any purpose, commercial or non-commercial, and by any
means.
Permission is granted to anyone to use this software for any purpose,
including commercial applications, and to alter it and redistribute it
freely, subject to the following restrictions:
In jurisdictions that recognize copyright laws, the author or authors
of this software dedicate any and all copyright interest in the
software to the public domain. We make this dedication for the benefit
of the public at large and to the detriment of our heirs and
successors. We intend this dedication to be an overt act of
relinquishment in perpetuity of all present and future rights to this
software under copyright law.
1. The origin of this software must not be misrepresented; you must not
claim that you wrote the original software. If you use this software
in a product, an acknowledgment in the product documentation would be
appreciated but is not required.
2. Altered source versions must be plainly marked as such, and must not be
misrepresented as being the original software.
3. This notice may not be removed or altered from any source distribution.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
IN NO EVENT SHALL THE AUTHORS BE LIABLE FOR ANY CLAIM, DAMAGES OR
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
OTHER DEALINGS IN THE SOFTWARE.
For more information, please refer to <https://unlicense.org>
-14
View File
@@ -1,14 +0,0 @@
#ifdef INTELLISENSE_DIRECTIVES
# pragma once
#endif
enum {
bios_init_pad_2 = 0x12,
bios_start_pad_2 = 0x13,
bios_flushcache = 0x44,
bios_table_addr = 0xA0,
bios_btable_addr = 0xB0,
};
enum {
bios_pad_buffer_size = 0x22,
};
+2 -22
View File
@@ -70,31 +70,11 @@
/* ----------------------------------------------------------------------------
* atom_reg (per-enum opt-in marker for the DWARF register-alias registry)
*
* Bare `atom_reg` token adjacent to an enum entry that alias as debug-visible for scan_source's register_alias_registry.
* Lua scanner reads the bare token.
* The bare `atom_reg` token adjacent to an enum entry in mips.h / lottes_tape.h flags that alias as debug-visible for scan_source's register_alias_registry.
* The C preprocessor strips it to a comment so no runtime symbol is created; the Lua scanner reads the bare token.
* ----------------------------------------------------------------------------*/
#define atom_reg /* atom_reg: opt the preceding enum entry into the DWARF registry */
// ----------------------------------------------------------------------------
// atom_auto_reg(atom, sym) — per-atom auto-allocated GPR binding.
// enum {
// atom_auto_reg(cube_g4_face, R_Fwdx), // expands to: R_Fwdx = R_Fwdx_Code /* atom_auto_reg: cube_g4_face */,
// atom_auto_reg(cube_g4_face, R_Eye_z) atom_type(S4), // atom_type chains after
// };
// (The macro IS the entire enum entry — no separate LHS=RHS. The `atom` scope is
// preserved in a trailing C-comment on the RHS so the Lua scanner can recover
// it after preprocessing strips the macro form. R_<Sym>_Code is resolved from gen/auto_reg.h which the .c file #include's before the enum declaration.)
#define atom_auto_reg(atom, sym) sym = sym ## _Code /* atom_auto_reg: atom */
// ----------------------------------------------------------------------------
// phase_auto_reg(phase, sym) — per-phase auto-allocated GPR binding.
// enum {
// phase_auto_reg(cube_g4, R_Temp0), // expands to: R_Temp0 = R_Temp0_Code /* phase_auto_reg: cube_g4 */,
// phase_auto_reg(cube_g4, R_Temp1),
// };
// (Same macro-as-enum-entry form as atom_auto_reg above; the `phase` scope is preserved in a trailing C-comment on the RHS for the Lua scanner to recover.)
#define phase_auto_reg(phase, sym) sym = sym ## _Code /* phase_auto_reg: phase */
/* ============================================================================
* atom_info :
* MipsAtom_(cube_tri) atom_info(
+18 -24
View File
@@ -3,7 +3,7 @@
# include "assert.h"
#endif
#define offset_of(type, member) cast(U8,__builtin_offsetof(type,member)) // Compiler builtin version of O_
#define offset_of(type, member) cast(U8,__builtin_offsetof(type,member))
#define static_assert _Static_assert
#define typeof __typeof__
#define typeof_ptr(ptr) typeof((ptr)[0])
@@ -28,9 +28,8 @@
#define internal static // internal
#define asm __asm__
#define A_(data) (& data)
#define align_(value) __attribute__((aligned (value))) // for easy alignment
#define align_(value) __attribute__((aligned (value))) // for easy alignment
#define C_(type,data) ((type)(data)) // for enforced precedence
#define expect_(x, y) __builtin_expect(x, y) // so compiler knows the common path
@@ -91,13 +90,12 @@
#define PtrSet_(type) TypeR_(type); typedef TypeV_(type)
#define TSet_(type) type; typedef PtrSet_(type)
#define Array_len(a) (U4)(sizeof(a) / sizeof(typeof((a)[0])))
#define Array_decl(type, ...) (type[]){__VA_ARGS__}
#define array_len(a) (U4)(sizeof(a) / sizeof(typeof((a)[0])))
#define array_decl(type, ...) (type[]){__VA_ARGS__}
#define Array_sym(type,len) A ## len ## _ ## type
#define Array_expand(type,len) type Array_sym(type, len)[len]; typedef PtrSet_(Array_sym(type, len))
#define Array_(type,len) Array_expand(type,len)
#define Bit_(id,b) id = (1 << b), tmpl(id,pos) = b
#define Bitmask_(b) (1u << b)
#define Enum_(underlying_type, symbol) underlying_type TSet_(symbol); enum symbol
#define Proc_(symbol) symbol
#define Relative_(symbol) // Does nothing but annotate that a symbol is associated with another.
@@ -135,20 +133,21 @@ typedef __UINT32_TYPE__ TSet_(B4);
#define u4_v(value) C_(U4 V_*, value)
enum { false = 0, true = 1, true_overflow, };
#define u4_lo(value) (u4_(value) & 0xFFFFU)
#define u4_hi(value) (u4_(value) >> (S_(U2) * 8))
#define u4_lo(value) ((value) & 0xFFFFU)
#define u4_hi(value) ((value) >> 12)
typedef void Proc_(VoidFn) (void);
#define Kilo_(n) (C_(U4, n) << 10)
#define Mega_(n) (C_(U4, n) << 20)
#define Giga_(n) (C_(U4, n) << 30)
#define Tera_(n) (C_(U4, n) << 40)
#define kilo(n) (C_(U4, n) << 10)
#define mega(n) (C_(U4, n) << 20)
#define giga(n) (C_(U4, n) << 30)
#define tera(n) (C_(U4, n) << 40)
#define null C_(U4, 0)
#define nullptr C_(void*, 0)
#define O_(type, field) C_(U4, & C_(type*,0)->field)
#define OT_(field) O_(typeof_ptr(& field), field))
#define OA_(type, member, idx) C_(U4, & C_(type*,0)->member[idx])
#define OT_(field) O_(typeof_ptr(& field), filed))
#define S_(data) C_(U4, sizeof(data))
#define sop_1(op,a,b) C_(U1, s1_(a) op s1_(b))
@@ -169,8 +168,6 @@ def_signed_ops(le, <=)
#undef def_signed_ops
#undef def_signed_op
// Unused, we arent' doing any C-like asm since we have the asm dsl. We'll keep the non-generics if we somehow do.
#if 0
#define def_generic_sop(op, a, ...) _Generic((a), U1: op ## _s1, U2: op ## _s2, U4: op ## _s4) (a, __VA_ARGS__)
#define add_s(a,b) def_generic_sop(add,a,b)
#define sub_s(a,b) def_generic_sop(sub,a,b)
@@ -180,12 +177,11 @@ def_signed_ops(le, <=)
#define ge_s(a,b) def_generic_sop(ge, a,b)
#define le_s(a,b) def_generic_sop(le, a,b)
#undef def_generic_sop
#endif
#define alignas _Alignas
#define alignof _Alignof
#define byte_pad(amount, ...) B1 glue(_PAD_, __VA_ARGS__) [amount]
#define C_ptr(type, data) (C_(type*, & (data)) [0])
#define pcast(type, data) (C_(type*, & (data)) [0])
#define dbg_args(...) __VA_ARGS__
@@ -200,8 +196,6 @@ def_signed_ops(le, <=)
#define defer_info(type,expr, ...) for(type info= {__VA_ARGS__}; info.once!=1;++info.once,(expr)) // Defer with tracked state
#define do_while(cond) for (U8 once=0; once!=1 || (cond); ++once)
#define Jmp_nZero_(cond,label) if (cond) goto label;
#pragma endregion Control Flow & Iteration
#define span_iter(type, iter, m_begin, op, m_end) ( \
@@ -218,16 +212,16 @@ def_signed_ops(le, <=)
typedef Span_(S4);
typedef Span_(U4);
#if 0
#pragma region Debug
#define debug_trap() __builtin_trap()
#define debug_trap() __builtin_debugtrap()
#if BUILD_DEBUG
#define assert(cond) if(cond == false){debug_trap();}
IA_ void assert(U8 cond) { if(cond){return;} else{debug_trap(); ms_exit_process(1);} }
#else
# ifndef assert
# include <assert.h>
# endif
#define assert(cond)
#endif
#pragma endregion Debug
#endif
#define GCC_OPTIMIZATION_DISABLE _Pragma("GCC push_options") _Pragma("GCC optimize(\"O0\")")
#define GCC_OPTIMIZATION_ENABLE _Pragma("GCC pop_options")
+19 -143
View File
@@ -14,9 +14,7 @@
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
@@ -60,8 +58,8 @@ WORD_COUNT(mac_yield_tail, 3)
/* atom_dbg_skip */
#define mac_load_v2s2(rs_x, rs_y, r_base, offset) \
load_half( rs_x, r_base, offset + O_(V3_S2,x)) \
, load_half( rs_y, r_base, offset + O_(V3_S2,y))
load_half( rs_x, r_base, O_(V3_S2,x)) \
, load_half( rs_y, r_base, O_(V3_S2,y))
WORD_COUNT(mac_load_v2s2, 2)
/* atom_dbg_skip */
@@ -70,27 +68,6 @@ WORD_COUNT(mac_load_v2s2, 2)
, store_half(rt_y, base, offset + O_(V2_S2,y))
WORD_COUNT(mac_store_v2s2, 2)
/* atom_dbg_skip */
#define mac_load_v3s4(rs_x, rs_y, rs_z, r_base, offset) \
load_word( rs_x, r_base, offset + O_(V3_S4,x)) \
, load_word( rs_y, r_base, offset + O_(V3_S4,y)) \
, load_word( rs_z, r_base, offset + O_(V3_S4,z))
WORD_COUNT(mac_load_v3s4, 3)
/* atom_dbg_skip */
#define mac_store_v3s4(rt_x, rt_y, rt_z, base, offset) \
store_word(rt_x, base, offset + O_(V3_S4,x)) \
, store_word(rt_y, base, offset + O_(V3_S4,y)) \
, store_word(rt_z, base, offset + O_(V3_S4,z))
WORD_COUNT(mac_store_v3s4, 3)
/* atom_dbg_skip */
#define mac_sub_v3s4(rds_x, rds_y, rds_z, rt_x, rt_y, rt_z) \
sub_s(rds_x, rds_x, rt_x) \
, sub_s(rds_y, rds_y, rt_y) \
, sub_s(rds_z, rds_z, rt_z)
WORD_COUNT(mac_sub_v3s4, 3)
/* atom_dbg_skip */
#define mac_store_rects2(rt_x, rt_y, rt_width, rt_height, base, offset) \
store_half(rt_x, base, offset + O_(Rect_S2,x)) \
@@ -99,12 +76,6 @@ WORD_COUNT(mac_sub_v3s4, 3)
, store_half(rt_height, base, offset + O_(Rect_S2,height))
WORD_COUNT(mac_store_rects2, 4)
/* atom_dbg_skip */
#define mac_load_word_imm(dst, imm) \
load_upper_i(dst, u4_hi(imm)) \
, or_i_self( dst, u4_lo(imm))
WORD_COUNT(mac_load_word_imm, 2)
/* atom_dbg_skip */
#define mac_load_tri_indices(r_face_cusor, r_i0, r_i1, r_i2) \
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)) \
@@ -153,91 +124,9 @@ WORD_COUNT(mac_gte_store_g4_p012, 3)
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3))
WORD_COUNT(mac_gte_store_g4_p3, 1)
/* atom_dbg_skip */
#define mac_gte_sqr_v3(r_sx, r_sy, r_sz, r_sq_x, r_sq_y, r_sq_z) \
mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop) \
, gte_mv_from_data_r(r_sq_x, C2_MAC1) \
, gte_mv_from_data_r(r_sq_y, C2_MAC2) \
, gte_mv_from_data_r(r_sq_z, C2_MAC3)
WORD_COUNT(mac_gte_sqr_v3, 8)
/* atom_dbg_skip */
#define mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop_slot) \
gte_mv_to_data_r(r_sx, C2_IR1) \
, gte_mv_to_data_r(r_sy, C2_IR2) \
, gte_mv_to_data_r(r_sz, C2_IR3) \
, nop_slot \
, gte_cmdw_sqr
WORD_COUNT(mac_gte_sqr_v3s4, 5)
/* atom_dbg_skip */
#define mac_gte_gpf_scale(r_sx, r_sy, r_sz, r_recip_est, r_shift, r_dx, r_dy, r_dz) \
gte_mv_to_data_r(r_recip_est, C2_IR0) \
, gte_mv_to_data_r(r_sx, C2_IR1) \
, gte_mv_to_data_r(r_sy, C2_IR2) \
, gte_mv_to_data_r(r_sz, C2_IR3) \
, nop2 /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */ \
, gte_cmdw_gpf \
, gte_mv_from_data_r(r_dx, C2_MAC1) \
, gte_mv_from_data_r(r_dy, C2_MAC2) \
, gte_mv_from_data_r(r_dz, C2_MAC3) \
, shift_aright_var(r_dx, r_dx, r_shift) \
, shift_aright_var(r_dy, r_dy, r_shift) \
, shift_aright_var(r_dz, r_dz, r_shift)
WORD_COUNT(mac_gte_gpf_scale, 13)
#define mac_trans_mt3s3s4(r_mtx, r_off, r_t0, r_t1, r_t2) \
load_word(r_t0, r_off, O_(V3_S4,x)) \
, load_word(r_t1, r_off, O_(V3_S4,y)) \
, load_word(r_t2, r_off, O_(V3_S4,z)) \
, store_word(r_t0, r_mtx, O_(MT3_S2S4,t[0])) \
, store_word(r_t1, r_mtx, O_(MT3_S2S4,t[1])) \
, store_word(r_t2, r_mtx, O_(MT3_S2S4,t[2]))
WORD_COUNT(mac_trans_mt3s3s4, 6)
/* atom_dbg_skip */
#define mac_lzcr_round_even_half_shift(r_shift, r_mag_sq, r_mag_sq_copy) \
and_i(r_shift, r_shift, gte_lzcr_even_mask) \
, or_u(r_mag_sq_copy, r_mag_sq, 0) \
, li_s(r_mag_sq, 31) \
, sub_s(r_mag_sq, r_mag_sq, r_shift) \
, shift_aright(r_mag_sq, r_mag_sq, 1)
WORD_COUNT(mac_lzcr_round_even_half_shift, 5)
#define mac_shift_aright_var_v3(rd_v0, rd_v1, rd_v2, rs_v0, rs_v1, rs_v2, r_shift) \
shift_aright_var(rd_v0, rs_v0, r_shift) \
, shift_aright_var(rd_v1, rs_v1, r_shift) \
, shift_aright_var(rd_v2, rs_v2, r_shift)
WORD_COUNT(mac_shift_aright_var_v3, 3)
#define mac_shift_aright_var_v3_self(rds_v0, rds_v1, rds_v2, r_shift) \
shift_aright_var(rds_v0, rds_v0, r_shift) \
, shift_aright_var(rds_v1, rds_v1, r_shift) \
, shift_aright_var(rds_v2, rds_v2, r_shift)
WORD_COUNT(mac_shift_aright_var_v3_self, 3)
#define mac_gte_general_purpose_interopolation(to_ir0, to_ir1, to_ir2, to_ir3, fr_mac1, fr_mac2, fr_mac3, nop_slot1, nop_slot2) \
gte_mv_to_data_r(to_ir0, C2_IR0) \
, gte_mv_to_data_r(to_ir1, C2_IR1) /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */ \
, gte_mv_to_data_r(to_ir2, C2_IR2) \
, gte_mv_to_data_r(to_ir3, C2_IR3) /* IR3 = src.z (reloaded) */ \
, LdSlot_ nop_slot1 \
, LdSlot_ nop_slot2 \
, gte_cmdw_gpf \
, gte_mv_from_data_r(fr_mac1, C2_MAC1) \
, gte_mv_from_data_r(fr_mac2, C2_MAC2) \
, gte_mv_from_data_r(fr_mac3, C2_MAC3)
WORD_COUNT(mac_gte_general_purpose_interopolation, 10)
#define mac_gte_mv_from_data_r_mac123(fr_mac1, fr_mac2, fr_mac3) \
gte_mv_from_data_r(fr_mac1, C2_MAC1) \
, gte_mv_from_data_r(fr_mac2, C2_MAC2) \
, gte_mv_from_data_r(fr_mac3, C2_MAC3)
WORD_COUNT(mac_gte_mv_from_data_r_mac123, 3)
/* atom_dbg_skip */
#define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \
mac_load_word_imm(reg_transfer, cmd) \
load_upper_i(reg_transfer, cmd >> 16) \
, or_i_self( reg_transfer, cmd & 0xFFFF) \
, store_word( reg_transfer, reg_base, port)
WORD_COUNT(mac_gcmd_push, 3)
@@ -260,7 +149,6 @@ WORD_COUNT(mac_pack_color_word, 3)
mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b)
WORD_COUNT(mac_format_f3_color, 3)
/* atom_dbg_skip */
#define mac_format_g4_color(r_prim_cursor, r0, g0, b0, r1, g1, b1, r2, g2, b2, r3, g3, b3) \
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0) \
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1) \
@@ -268,41 +156,29 @@ WORD_COUNT(mac_format_f3_color, 3)
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3)
WORD_COUNT(mac_format_g4_color, 12)
#define mac_insert_ot_tag(r_ot_base, r_prim_cursor, poly_size) \
#define mac_insert_ot_tag_f3(r_ot_base, r_prim_cursor) \
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
, add_u_self( R_T1, r_ot_base) /* T1 = & OrderingTable[OTZ] */ \
, load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \
, load_upper_i(R_V0, (poly_size/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) \
, load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) /* V0 = (5 - 1) << 24 = 4 << 24 */ \
, mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
, store_word( R_AT, r_prim_cursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \
, shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
, shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
WORD_COUNT(mac_insert_ot_tag, 11)
WORD_COUNT(mac_insert_ot_tag_f3, 11)
/* atom_dbg_skip */
#define mac_pad_set_centered_axes(state, scratch) \
load_upper_i(scratch, (PadAxis_Centered >> 16) & 0xFFFF) \
, or_i_self( scratch, PadAxis_Centered & 0xFFFF) /* mac_load_word_imm(scratch, PadAxis_Centered), */ \
, store_word( scratch, state, O_(PadState,axes))
WORD_COUNT(mac_pad_set_centered_axes, 3)
/* atom_dbg_skip */
#define mac_pad_set_id_byte(state, r_id, id_value) \
add_ui( r_id, R_0, id_value) \
, store_byte(r_id, state, O_(PadState,id))
WORD_COUNT(mac_pad_set_id_byte, 2)
/* atom_dbg_skip */
#define mac_pad_set_status(r_tmp, r_state, pad_status) \
add_ui( r_tmp, R_0, pad_status) \
, store_word(r_tmp, r_state, O_(PadState,status))
WORD_COUNT(mac_pad_set_status, 2)
/* atom_dbg_skip */
#define mac_pad_store_inverted_buttons(r_buttons, r_pad_state) \
nor_u( r_buttons, r_buttons, R_0) \
, store_half( r_buttons, r_pad_state, O_(PadState,buttons))
WORD_COUNT(mac_pad_store_inverted_buttons, 2)
#define mac_insert_ot_tag_g4(r_ot_base, r_prim_cursor) \
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
, add_u_self( R_T1, r_ot_base) /* T1 = & OrderingTable[OTZ] */ \
, load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \
, load_upper_i(R_V0, (S_(Poly_G4)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) /* V0 = (9 - 1) << 24 = 8 << 24 */ \
, mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
, store_word( R_AT, r_prim_cursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \
, shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
, shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
WORD_COUNT(mac_insert_ot_tag_g4, 11)
+10 -22
View File
@@ -11,9 +11,7 @@
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
@@ -25,27 +23,17 @@
#pragma region duffle
// --- atom: normalize_v3s4 (47 words) ---
// --- atom: pad_bios_snapshot (78 words) ---
#define _atom_offset_aligned_done_srav_path 3
#define _atom_offset_srav_path_aligned_done 4
enum {
atom_offset_aligned_done_srav_path = _atom_offset_aligned_done_srav_path,
atom_offset_srav_path_aligned_done = _atom_offset_srav_path_aligned_done,
};
// --- atom: pad_bios_snapshot (84 words) ---
#define _atom_offset_snap_root_skip_disconnected 10
#define _atom_offset_disconnected_snap_end 65
#define _atom_offset_case_2_id_dispatch 9
#define _atom_offset_pending_snap_end 54
#define _atom_offset_id_dispatch_try_analog_stick 12
#define _atom_offset_id_dispatch_snap_end 40
#define _atom_offset_try_analog_stick_try_analog_pad 13
#define _atom_offset_analog_stick_snap_end 25
#define _atom_offset_try_analog_pad_try_unsupported 12
#define _atom_offset_snap_root_skip_disconnected 8
#define _atom_offset_disconnected_snap_end 61
#define _atom_offset_case_2_id_dispatch 8
#define _atom_offset_pending_snap_end 51
#define _atom_offset_id_dispatch_try_analog_stick 11
#define _atom_offset_id_dispatch_snap_end 38
#define _atom_offset_try_analog_stick_try_analog_pad 12
#define _atom_offset_analog_stick_snap_end 24
#define _atom_offset_try_analog_pad_try_unsupported 11
#define _atom_offset_analog_pad_snap_end 10
enum {
+37 -15
View File
@@ -8,47 +8,54 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(gp_atom_c);
#pragma region MACs (Mips Atom Components)
FI_ Slice_MipsCode ac_gcmd_push(AtomBuilder_R ab, U4 cmd, U4 reg_transfer, U4 reg_base, U2 port)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
mac_load_word_imm(reg_transfer, cmd),
FI_ Slice_MipsCode ac_gcmd_push(U4 cmd, U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_gcmd_push, {
load_upper_i(reg_transfer, cmd >> 16),
or_i_self( reg_transfer, cmd & 0xFFFF),
store_word( reg_transfer, reg_base, port),
})
FI_ Slice_MipsCode ac_store_rgb8(AtomBuilder_R ab, U1 rr, U1 rg, U1 rb, U4 base, U4 offset)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
FI_ Slice_MipsCode ac_store_rgb8(U1 rr, U1 rg, U1 rb, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rgb8, {
store_byte(rr, base, offset + O_(RGB8,r)),
store_byte(rg, base, offset + O_(RGB8,g)),
store_byte(rb, base, offset + O_(RGB8,b)),
store_byte(rb, base, offset + O_(RGB8,b)),
})
FI_ Slice_MipsCode ac_pack_color_word(AtomBuilder_R ab, U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
/* Words: 3; Emits one (cmd|color) word to R_PrimCursor at the given
* byte offset. Internal helper used by the *_format_*_color macros. */
FI_ Slice_MipsCode ac_pack_color_word(U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b)
atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, {
load_upper_i(R_AT, (cmd) << 8 | (b)),
or_i_self( R_AT, ((g) << 8) | (r)),
store_word( R_AT, r_base, (off)),
})
FI_ Slice_MipsCode ac_format_f3_color(AtomBuilder_R ab, U4 r_base, U1 r, U1 g, U1 b)
atom_dbg_skip MipsAtomComp_Proc_(ab, { mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
/* Words: 3; Emits the F3 command+color word (cmd byte | BLUE | GREEN | RED)
* Args: _r, _g, _b are 8-bit RGB byte values (not raw 16-bit fields). */
FI_ Slice_MipsCode ac_format_f3_color(U4 r_base, U1 r, U1 g, U1 b)
atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, { mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
FI_ Slice_MipsCode ac_format_g4_color(AtomBuilder_R ab, U4 r_prim_cursor,
/* Words: 12; Emits the four (code|color) words of a Poly_G4.
* Args: rN,gN,bN are 8-bit RGB byte values for each of the 4 vertices. */
FI_ Slice_MipsCode ac_format_g4_color(U4 r_prim_cursor,
U1 r0, U1 g0, U1 b0,
U1 r1, U1 g1, U1 b1,
U1 r2, U1 g2, U1 b2,
U1 r3, U1 g3, U1 b3)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
MipsAtomComp_Proc_(ac_format_g4_color, {
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0),
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1),
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c2), 0, r2,g2,b2),
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3),
})
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list. */
I_ Slice_MipsCode ac_insert_ot_tag(AtomBuilder_R ab, U4 r_ot_base, U4 r_prim_cursor, U4 poly_size) MipsAtomComp_Proc_(ab, {
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
* Hardcoded for Poly_F3 (5 words). For Poly_G4, use ac_insert_ot_tag_g4. */
I_ Slice_MipsCode ac_insert_ot_tag_f3(U4 r_ot_base, U4 r_prim_cursor) MipsAtomComp_Proc_(ac_insert_ot_tag_f3, {
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
load_upper_i(R_V0, (poly_size/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits),
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits), // V0 = (5 - 1) << 24 = 4 << 24
mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
or_u( R_AT, R_AT, R_V0), // Merge length
store_word( R_AT, r_prim_cursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr)
@@ -57,4 +64,19 @@ I_ Slice_MipsCode ac_insert_ot_tag(AtomBuilder_R ab, U4 r_ot_base, U4 r_prim_cur
store_word( R_AT, R_T1, O_(PolyTag,code)), // OrderingTable[OTZ] = PrimCursor
})
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
* Hardcoded for Poly_G4 (9 words). For Poly_F3, use ac_insert_ot_tag_f3. */
I_ Slice_MipsCode ac_insert_ot_tag_g4(U4 r_ot_base, U4 r_prim_cursor) MipsAtomComp_Proc_(ac_insert_ot_tag_g4, {
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
load_upper_i(R_V0, (S_(Poly_G4)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits), // V0 = (9 - 1) << 24 = 8 << 24
mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
or_u( R_AT, R_AT, R_V0), // Merge length
store_word( R_AT, r_prim_cursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr)
shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)), // AT = (prim_length << 24) | old_addr
shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
store_word( R_AT, R_T1, O_(PolyTag,code)), // OrderingTable[OTZ] = PrimCursor
})
#pragma endregion MACs (Mips Atom Components)
+55 -54
View File
@@ -21,7 +21,7 @@
* 4. Semantic encoders gp0_word_poly_f3(r,g,b)
* 3. Composite encoders enc_color_word(cmd, r, g, b)
* 2. Per-field encoders enc_gp0_color_r(r), enc_gp0_color_g(g), ...
* 1. Bitfield layout consts gp0_color_red_shift = 0, gp0_color_red_width = 8
* 1. Bitfield layout consts gp0_color_red_shift = 0, gp0_color_red_mask = 0xFF
* 0. Opcode IDs gp0_cmd_poly_f3 = 0x20
*
* Vendor mnemonics (gte_mtc2, gte_mfc2, etc.) are NOT in this header.
@@ -74,7 +74,7 @@ enum {
* ============================================================================
* 8-bit GP0 opcodes (the upper byte of a primitive's first word). These are the BYTE only.
* NO macro body past this point uses a raw shift or raw mask.
* Mirrors the OPCODE_SHIFT / RS_SHIFT convention from mips.h.
* Mirrors the OPCODE_SHIFT / RS_SHIFT / REG_MASK convention from mips.h.
* ============================================================================ */
enum {
gp0_cmd_Nop = 0x00,
@@ -116,20 +116,21 @@ enum {
gp0_cmd_SetDrawOffset = 0xE5,
gp0_cmd_SetMaskBit = 0xE6,
/* bitfield shifts / widths ----
/* bitfield shifts / widths / masks ----
* Generic GP0/GP1 command byte (upper 8 bits of every word sent to either port). */
gp0_cmd_shift = 24,
gp0_cmd_width = 8,
gp0_cmd_mask = 0xFF,
/* Color word layout (lives in Poly_F3.color, Poly_G4.c0..c3, etc.):
* bits 31..24 = command byte
* bits 23..16 = BLUE
* bits 15..08 = GREEN
* bits 07..00 = RED (PSX GPU is BGR, NOT RGB) */
gp0_color_cmd_shift = 24, gp0_color_cmd_width = 8,
gp0_color_blue_shift = 16, gp0_color_blue_width = 8,
gp0_color_green_shift = 8, gp0_color_green_width = 8,
gp0_color_red_shift = 0, gp0_color_red_width = 8,
gp0_color_cmd_shift = 24, gp0_color_cmd_width = 8, gp0_color_cmd_mask = 0xFF,
gp0_color_blue_shift = 16, gp0_color_blue_width = 8, gp0_color_blue_mask = 0xFF,
gp0_color_green_shift = 8, gp0_color_green_width = 8, gp0_color_green_mask = 0xFF,
gp0_color_red_shift = 0, gp0_color_red_width = 8, gp0_color_red_mask = 0xFF,
};
/* ============================================================================
@@ -142,12 +143,12 @@ enum {
* ============================================================================ */
/* ---- Layer 1.5: per-field encoders ---- */
#define enc_gp0_cmd(cmd) ((cmd) << gp0_cmd_shift)
#define enc_gp0_cmd(cmd) (((cmd) & gp0_cmd_mask) << gp0_cmd_shift)
#define enc_gp0_color_cmd(cmd) ((cmd) << gp0_color_cmd_shift)
#define enc_gp0_color_r(r) ((r) << gp0_color_red_shift)
#define enc_gp0_color_g(g) ((g) << gp0_color_green_shift)
#define enc_gp0_color_b(b) ((b) << gp0_color_blue_shift)
#define enc_gp0_color_cmd(cmd) (((cmd) & gp0_color_cmd_mask) << gp0_color_cmd_shift)
#define enc_gp0_color_r(r) (((r) & gp0_color_red_mask) << gp0_color_red_shift)
#define enc_gp0_color_g(g) (((g) & gp0_color_green_mask) << gp0_color_green_shift)
#define enc_gp0_color_b(b) (((b) & gp0_color_blue_mask) << gp0_color_blue_shift)
/* ---- Layer 2: composite encoders ---- */
#define enc_color_word(cmd, r, g, b) (enc_gp0_color_cmd(cmd) | enc_gp0_color_r(r) | enc_gp0_color_g(g) | enc_gp0_color_b(b))
@@ -210,38 +211,38 @@ enum {
gp1_disp_Color24 = 0x1,
gp1_disp_VInterlace = 0x1,
/* ---- Layer 1: GP1 display-mode + range + draw-area shifts/widths ---- */
gp1_disp_hres_shift = 0, gp1_disp_hres_width = 2,
gp1_disp_vres_shift = 2, gp1_disp_vres_width = 1,
gp1_disp_color_shift = 4, gp1_disp_color_width = 1,
gp1_disp_interlace_shift = 5, gp1_disp_interlace_width = 1,
/* ---- Layer 1: GP1 display-mode + range + draw-area shifts/masks ---- */
gp1_disp_hres_shift = 0, gp1_disp_hres_width = 2, gp1_disp_hres_mask = 0x3,
gp1_disp_vres_shift = 2, gp1_disp_vres_width = 1, gp1_disp_vres_mask = 0x1,
gp1_disp_color_shift = 4, gp1_disp_color_width = 1, gp1_disp_color_mask = 0x1,
gp1_disp_interlace_shift = 5, gp1_disp_interlace_width = 1, gp1_disp_interlace_mask = 0x1,
/* GP1 horizontal display range: bits 0..11 = X2, bits 12..23 = X1 */
gp1_hrange_x1_shift = 12, gp1_hrange_x1_width = 12,
gp1_hrange_x2_shift = 0, gp1_hrange_x2_width = 12,
gp1_hrange_x1_shift = 12, gp1_hrange_x1_width = 12, gp1_hrange_x1_mask = 0xFFF,
gp1_hrange_x2_shift = 0, gp1_hrange_x2_width = 12, gp1_hrange_x2_mask = 0xFFF,
/* GP1 vertical display range: bits 0..9 = Y2, bits 10..19 = Y1 */
gp1_vrange_y1_shift = 10, gp1_vrange_y1_width = 10,
gp1_vrange_y2_shift = 0, gp1_vrange_y2_width = 10,
gp1_vrange_y1_shift = 10, gp1_vrange_y1_width = 10, gp1_vrange_y1_mask = 0x3FF,
gp1_vrange_y2_shift = 0, gp1_vrange_y2_width = 10, gp1_vrange_y2_mask = 0x3FF,
/* GP1 draw area (top-left or bottom-right): bits 0..9 = X, bits 10..19 = Y
* (10-bit signed — caller pre-signs) */
gp1_draw_x_shift = 0, gp1_draw_x_width = 10,
gp1_draw_y_shift = 10, gp1_draw_y_width = 10,
* (10-bit signed — caller pre-signs and masks with the named mask) */
gp1_draw_x_shift = 0, gp1_draw_x_width = 10, gp1_draw_x_mask = 0x3FF,
gp1_draw_y_shift = 10, gp1_draw_y_width = 10, gp1_draw_y_mask = 0x3FF,
};
/* ---- Layer 1.5: GP1 per-field encoders ---- */
#define enc_gp1_disp_hres(h) ((h) << gp1_disp_hres_shift)
#define enc_gp1_disp_vres(v) ((v) << gp1_disp_vres_shift)
#define enc_gp1_disp_color(c) ((c) << gp1_disp_color_shift)
#define enc_gp1_disp_interlace(i) ((i) << gp1_disp_interlace_shift)
#define enc_gp1_disp_hres(h) (((h) & gp1_disp_hres_mask) << gp1_disp_hres_shift)
#define enc_gp1_disp_vres(v) (((v) & gp1_disp_vres_mask) << gp1_disp_vres_shift)
#define enc_gp1_disp_color(c) (((c) & gp1_disp_color_mask) << gp1_disp_color_shift)
#define enc_gp1_disp_interlace(i) (((i) & gp1_disp_interlace_mask) << gp1_disp_interlace_shift)
#define enc_gp1_hrange_x1(x1) ((x1) << gp1_hrange_x1_shift)
#define enc_gp1_hrange_x2(x2) ((x2) << gp1_hrange_x2_shift)
#define enc_gp1_vrange_y1(y1) ((y1) << gp1_vrange_y1_shift)
#define enc_gp1_vrange_y2(y2) ((y2) << gp1_vrange_y2_shift)
#define enc_gp1_draw_x(x) ((x) << gp1_draw_x_shift)
#define enc_gp1_draw_y(y) ((y) << gp1_draw_y_shift)
#define enc_gp1_hrange_x1(x1) (((x1) & gp1_hrange_x1_mask) << gp1_hrange_x1_shift)
#define enc_gp1_hrange_x2(x2) (((x2) & gp1_hrange_x2_mask) << gp1_hrange_x2_shift)
#define enc_gp1_vrange_y1(y1) (((y1) & gp1_vrange_y1_mask) << gp1_vrange_y1_shift)
#define enc_gp1_vrange_y2(y2) (((y2) & gp1_vrange_y2_mask) << gp1_vrange_y2_shift)
#define enc_gp1_draw_x(x) (((x) & gp1_draw_x_mask) << gp1_draw_x_shift)
#define enc_gp1_draw_y(y) (((y) & gp1_draw_y_mask) << gp1_draw_y_shift)
/* ---- Layer 2: GP1 composite encoders ---- */
#define enc_gp1_disp_mode_word(h, v, c, i) (enc_gp0_cmd(gp1_cmd_DisplayMode) | enc_gp1_disp_hres(h) | enc_gp1_disp_vres(v) | enc_gp1_disp_color(c) | enc_gp1_disp_interlace(i))
@@ -554,14 +555,14 @@ typedef Struct_(Poly_GT4) {
* bits 12..31 = reserved (zero)
* ============================================================================ */
enum {
/* ---- Layer 1: TPage bitfield shifts / widths ---- */
gp0_tpage_x_shift = 0, gp0_tpage_x_width = 4,
gp0_tpage_y_shift = 4, gp0_tpage_y_width = 1,
gp0_tpage_semi_trans_shift = 5, gp0_tpage_semi_trans_width = 2,
gp0_tpage_color_depth_shift = 7, gp0_tpage_color_depth_width = 2,
gp0_tpage_dither_shift = 9, gp0_tpage_dither_width = 1,
gp0_tpage_draw_to_disp_shift = 10, gp0_tpage_draw_to_disp_width = 1,
gp0_tpage_tex_disable_shift = 11, gp0_tpage_tex_disable_width = 1,
/* ---- Layer 1: TPage bitfield shifts / widths / masks ---- */
gp0_tpage_x_shift = 0, gp0_tpage_x_width = 4, gp0_tpage_x_mask = 0xF,
gp0_tpage_y_shift = 4, gp0_tpage_y_width = 1, gp0_tpage_y_mask = 0x1,
gp0_tpage_semi_trans_shift = 5, gp0_tpage_semi_trans_width = 2, gp0_tpage_semi_trans_mask = 0x3,
gp0_tpage_color_depth_shift = 7, gp0_tpage_color_depth_width = 2, gp0_tpage_color_depth_mask = 0x3,
gp0_tpage_dither_shift = 9, gp0_tpage_dither_width = 1, gp0_tpage_dither_mask = 0x1,
gp0_tpage_draw_to_disp_shift = 10, gp0_tpage_draw_to_disp_width = 1, gp0_tpage_draw_to_disp_mask = 0x1,
gp0_tpage_tex_disable_shift = 11, gp0_tpage_tex_disable_width = 1, gp0_tpage_tex_disable_mask = 0x1,
/* TPage color-depth payload values (NOT bit positions — these go in
* the 2-bit field at gp0_tpage_color_depth_shift). */
@@ -580,13 +581,13 @@ enum {
};
/* ---- Layer 1.5: TPage per-field encoders. Mirrors enc_gte_sf/mx/v in gte.h. ---- */
#define enc_gp0_tpage_x(x) ((x) << gp0_tpage_x_shift)
#define enc_gp0_tpage_y(y) ((y) << gp0_tpage_y_shift)
#define enc_gp0_tpage_semi_trans(s) ((s) << gp0_tpage_semi_trans_shift)
#define enc_gp0_tpage_color_depth(c) ((c) << gp0_tpage_color_depth_shift)
#define enc_gp0_tpage_dither(d) ((d) << gp0_tpage_dither_shift)
#define enc_gp0_tpage_draw_to_disp(d) ((d) << gp0_tpage_draw_to_disp_shift)
#define enc_gp0_tpage_tex_disable(t) ((t) << gp0_tpage_tex_disable_shift)
#define enc_gp0_tpage_x(x) (((x) & gp0_tpage_x_mask) << gp0_tpage_x_shift)
#define enc_gp0_tpage_y(y) (((y) & gp0_tpage_y_mask) << gp0_tpage_y_shift)
#define enc_gp0_tpage_semi_trans(s) (((s) & gp0_tpage_semi_trans_mask) << gp0_tpage_semi_trans_shift)
#define enc_gp0_tpage_color_depth(c) (((c) & gp0_tpage_color_depth_mask) << gp0_tpage_color_depth_shift)
#define enc_gp0_tpage_dither(d) (((d) & gp0_tpage_dither_mask) << gp0_tpage_dither_shift)
#define enc_gp0_tpage_draw_to_disp(d) (((d) & gp0_tpage_draw_to_disp_mask) << gp0_tpage_draw_to_disp_shift)
#define enc_gp0_tpage_tex_disable(t) (((t) & gp0_tpage_tex_disable_mask) << gp0_tpage_tex_disable_shift)
/* ---- Layer 2: TPage composite encoder. Mirrors enc_gte_cmdw in gte.h ---- */
#define enc_gp0_tpage_word(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
@@ -616,17 +617,17 @@ typedef Struct_(TexturePage) { U4 raw; };
* bits 24..31 = command byte — 0x20 (4bpp load) or 0x25 (8bpp load)
* ============================================================================ */
enum {
/* ---- Layer 1: CLUT bitfield shifts / widths ---- */
gp0_clut_y_shift = 0, gp0_clut_y_width = 6,
gp0_clut_x_shift = 6, gp0_clut_x_width = 9,
/* ---- Layer 1: CLUT bitfield shifts / widths / masks ---- */
gp0_clut_y_shift = 0, gp0_clut_y_width = 6, gp0_clut_y_mask = 0x3F,
gp0_clut_x_shift = 6, gp0_clut_x_width = 9, gp0_clut_x_mask = 0x1FF,
/* CLUT-load cmd-byte variants — the upper byte of the GP0 word. */
gp0_clut_cmd_Load4bpp = 0x20,
gp0_clut_cmd_Load8bpp = 0x25,
};
/* ---- Layer 1.5: CLUT per-field encoders ---- */
#define enc_gp0_clut_x(x) ((x) << gp0_clut_x_shift)
#define enc_gp0_clut_y(y) ((y) << gp0_clut_y_shift)
#define enc_gp0_clut_x(x) (((x) & gp0_clut_x_mask) << gp0_clut_x_shift)
#define enc_gp0_clut_y(y) (((y) & gp0_clut_y_mask) << gp0_clut_y_shift)
/* ---- Layer 2: CLUT composite encoder ---- */
#define enc_gp0_clut_word(cmd, x, y) (enc_gp0_cmd(cmd) | enc_gp0_clut_x(x) | enc_gp0_clut_y(y))
+14 -326
View File
@@ -11,8 +11,7 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(gte_atom_c);
#pragma region MACs (Mips Atom Components)
/* Words: 3; Loads 3 S2 indices from the face array */
FI_ Slice_MipsCode ac_load_tri_indices(AtomBuilder_R ab, U4 r_face_cusor, U4 r_i0, U4 r_i1, U4 r_i2)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
FI_ Slice_MipsCode ac_load_tri_indices(U4 r_face_cusor, U4 r_i0, U4 r_i1, U4 r_i2) atom_dbg_skip MipsAtomComp_Proc_(ac_load_tri_indices, {
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)),
load_half_u(r_i1, r_face_cusor, 1 * S_(S2)),
load_half_u(r_i2, r_face_cusor, 2 * S_(S2)),
@@ -20,14 +19,14 @@ atom_dbg_skip MipsAtomComp_Proc_(ab, {
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
FI_ Slice_MipsCode ac_gte_store_f3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
FI_ Slice_MipsCode ac_gte_store_f3(U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_f3, {
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)),
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_F3,p1)),
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_F3,p2)),
})
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ab, {
I_ Slice_MipsCode ac_gte_load_tri_verts(U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_load_tri_verts, {
shift_lleft(R_AT, r_v0, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
shift_lleft(R_AT, r_v1, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
shift_lleft(R_AT, r_v2, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
@@ -35,10 +34,10 @@ I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, U4 r_vert_base, U4 r_v
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
* G4 triangle portion to p0/p1/p2.
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
* (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
FI_ Slice_MipsCode ac_gte_store_g4_p012(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
FI_ Slice_MipsCode ac_gte_store_g4_p012(U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p012, {
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)),
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)),
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2)),
@@ -48,335 +47,24 @@ FI_ Slice_MipsCode ac_gte_store_g4_p012(AtomBuilder_R ab, U4 r_primitive_cursor)
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2;
* SXY0 still holds v0.screen from the earlier RTPT.
*/
FI_ Slice_MipsCode ac_gte_store_g4_p3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) })
/* ─── STAGE 1 of normalize: SQR + mfc2 MAC1/2/3 ───
* Emits squared magnitude per component (in MAC1/2/3) into caller-provided scratch regs.
* Stage 2 of normalize consumes these directly.
* Words: 8. Clobbers: IR1/2/3, MAC1/2/3. Uses gte_cmdw_sqr (sf=0, lm=1). */
FI_ Slice_MipsCode ac_gte_sqr_v3(AtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_sq_x, U4 r_sq_y, U4 r_sq_z) atom_dbg_skip MipsAtomComp_Proc_(ab, {
mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop),
gte_mv_from_data_r(r_sq_x, C2_MAC1),
gte_mv_from_data_r(r_sq_y, C2_MAC2),
gte_mv_from_data_r(r_sq_z, C2_MAC3),
})
/* ─── SQR FIRE — mtc2 3 GPRs into IR1/IR2/IR3, then fire SQR. ───
* The SQR command always squares IR1/IR2/IR3 — those C2 registers are fixed.
* The GPRs holding the source vector are caller-determined.
* Words: 5 (3 mtc2 + 1 nop hazard + 1 cmd). */
FI_ Slice_MipsCode ac_gte_sqr_v3s4(AtomBuilder_R ab, Reg r_sx, Reg r_sy, Reg r_sz, MipsCode nop_slot)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
gte_mv_to_data_r(r_sx, C2_IR1),
gte_mv_to_data_r(r_sy, C2_IR2),
gte_mv_to_data_r(r_sz, C2_IR3),
nop_slot, gte_cmdw_sqr,
})
/* ─── STAGE 4 of normalize: mtc2 IR0..3 + GPF + mfc2 MAC + srav finalize ───
* Reusable standalone — given an IR0 = 1/|v| estimate (typically from a sqrtbl lookup) and a shift count
* (typically (31 - LZCR)/2), multiplies IR0*IR[i] via GPF and shifts right to produce the normalized output.
* Used standalone for "scale vector by scalar".
* Words: 11. Clobbers: IR0..3, MAC1..3. Uses gte_cmdw_gpf (sf=0, lm=0). */
FI_ Slice_MipsCode ac_gte_gpf_scale(AtomBuilder_R ab,
U4 r_sx, U4 r_sy, U4 r_sz,
U4 r_recip_est, U4 r_shift,
U4 r_dx, U4 r_dy, U4 r_dz)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
gte_mv_to_data_r(r_recip_est, C2_IR0),
gte_mv_to_data_r(r_sx, C2_IR1),
gte_mv_to_data_r(r_sy, C2_IR2),
gte_mv_to_data_r(r_sz, C2_IR3),
nop2, /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */
gte_cmdw_gpf,
gte_mv_from_data_r(r_dx, C2_MAC1),
gte_mv_from_data_r(r_dy, C2_MAC2),
gte_mv_from_data_r(r_dz, C2_MAC3),
shift_aright_var(r_dx, r_dx, r_shift),
shift_aright_var(r_dy, r_dy, r_shift),
shift_aright_var(r_dz, r_dz, r_shift),
})
/* ─── TRANS MATRIX (libgte TransMatrix port) ───
* Atom component — auto-generates mac_trans_matrix Mac composer macro.
* m->t = v (struct copy; libgte's TransMatrix at 0x8001a540 is just 3 store_words, no GTE, no add).
* Uses 1 GPR (r_t1 = off value) per axis; per-axis load-delay-slot pattern.
* Words: 9. Clobbers: r_t1. */
FI_ Slice_MipsCode ac_trans_mt3s3s4(AtomBuilder_R ab
, U4 r_mtx, U4 r_off
, U4 r_t0, U4 r_t1, U4 r_t2
) MipsAtomComp_Proc_(ab, {
load_word(r_t0, r_off, O_(V3_S4,x)),
load_word(r_t1, r_off, O_(V3_S4,y)),
load_word(r_t2, r_off, O_(V3_S4,z)),
store_word(r_t0, r_mtx, O_(MT3_S2S4,t[0])),
store_word(r_t1, r_mtx, O_(MT3_S2S4,t[1])),
store_word(r_t2, r_mtx, O_(MT3_S2S4,t[2])),
})
/* ─── LZCR ROUND EVEN + HALF-SHIFT ───
* Takes the raw LZCR leading-zero/ones count (from mfc2 C2_LZCR, range 1..32
* per PSX-SPX cop2r31) and the |v|² sum (in r_mag_sq from the MAC1+MAC2+MAC3
* add). Produces:
* r_shift ← LZCR rounded down to even (clear bit 0)
* r_mag_sq_copy ← |v|² sum (moved out of r_mag_sq before it's overwritten)
* r_mag_sq ← (31 - even_LZCR) / 2 = the final srav/GPF shift amount
*
* Rounding to even ensures (31 - LZCR) is always odd, so the >> 1 division
* is consistent — no 0.5 loss. The caller branches on LZCR < 24 to decide
* left-shift vs right-shift of r_mag_sq_copy, then saves the shift count.
*
* Note: C2_LZCR (cop2r31) is a fixed read-only C2 data register — the caller
* must read it via mfc2 from C2_LZCR; there is no register choice at the
* hardware level. Only the GPR that holds the result is caller-determined. */
FI_ Slice_MipsCode ac_lzcr_round_even_half_shift(AtomBuilder_R ab,
U4 r_shift,
U4 r_mag_sq,
U4 r_mag_sq_copy
)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
and_i(r_shift, r_shift, gte_lzcr_even_mask),
or_u(r_mag_sq_copy, r_mag_sq, 0),
li_s(r_mag_sq, 31),
sub_s(r_mag_sq, r_mag_sq, r_shift),
shift_aright(r_mag_sq, r_mag_sq, 1),
})
FI_ Slice_MipsCode ac_shift_aright_var_v3(AtomBuilder_R ab
, Reg rd_v0, Reg rd_v1, Reg rd_v2
, Reg rs_v0, Reg rs_v1, Reg rs_v2
, Reg r_shift)
MipsAtomComp_Proc_(ab, {
shift_aright_var(rd_v0, rs_v0, r_shift),
shift_aright_var(rd_v1, rs_v1, r_shift),
shift_aright_var(rd_v2, rs_v2, r_shift),
})
FI_ Slice_MipsCode ac_shift_aright_var_v3_self(AtomBuilder_R ab
, Reg rds_v0, Reg rds_v1, Reg rds_v2
, Reg r_shift)
MipsAtomComp_Proc_(ab, {
shift_aright_var(rds_v0, rds_v0, r_shift),
shift_aright_var(rds_v1, rds_v1, r_shift),
shift_aright_var(rds_v2, rds_v2, r_shift),
})
FI_ Slice_MipsCode ac_gte_general_purpose_interopolation(AtomBuilder_R ab
, Reg to_ir0, Reg to_ir1, Reg to_ir2, Reg to_ir3
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3
, MipsCode nop_slot1, MipsCode nop_slot2)
MipsAtomComp_Proc_(ab, {
gte_mv_to_data_r(to_ir0, C2_IR0),
gte_mv_to_data_r(to_ir1, C2_IR1), /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */
gte_mv_to_data_r(to_ir2, C2_IR2),
gte_mv_to_data_r(to_ir3, C2_IR3), /* IR3 = src.z (reloaded) */
LdSlot_ nop_slot1,
LdSlot_ nop_slot2,
gte_cmdw_gpf,
gte_mv_from_data_r(fr_mac1, C2_MAC1),
gte_mv_from_data_r(fr_mac2, C2_MAC2),
gte_mv_from_data_r(fr_mac3, C2_MAC3),
})
FI_ Slice_MipsCode gte_mv_from_data_r_mac123(AtomBuilder_R ab
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3
)
MipsAtomComp_Proc_(ab, {
gte_mv_from_data_r(fr_mac1, C2_MAC1),
gte_mv_from_data_r(fr_mac2, C2_MAC2),
gte_mv_from_data_r(fr_mac3, C2_MAC3),
})
FI_ Slice_MipsCode ac_gte_store_g4_p3(U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p3, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) })
#pragma endregion MACs (Mips Atom Components)
#pragma region Atom Procs
#pragma region Bsked Atoms
/* ─── Local copy of PSYQ's sqrtbl (1/sqrt lookup table for VectorNormal). ───
* Source: PSYQ 4.7 libgte sqrtbl at 0x800185B4 in hello_camera.elf.
* objdump -s --start-address=0x800185B4 --stop-address=0x800185F4 hello_camera.elf
* → 192 entries × 16-bit signed, in 1.12 fixed-point (max value 0x1000 = 1.0).
*
* Data is identical to the libgte original (byte-for-byte verified).
*
* ─── Per-entry semantics (decoded from libgte msc02 VectorNormal) ───
* Each entry is `1/sqrt(x)` in 1.12 fixed point (value / 4096).
* The 192 entries span 4 octaves of the input magnitude, with 48 entries per octave:
* Octave 0 (entries 0- 47): mantissa in [0x8000, 0x10000) output ~[1.000, 0.707]
* Octave 1 (entries 48- 95): mantissa in [0x10000, 0x20000) output ~[0.707, 0.500]
* Octave 2 (entries 96-143): mantissa in [0x20000, 0x40000) output ~[0.500, 0.354]
* Octave 3 (entries144-191): mantissa in [0x40000, 0x80000) output ~[0.354, 0.251]
* Within each octave, 8 sub-entries interpolate over the 8 fractional bits of the mantissa
* (the byte `(0x80 | (i mod 8))` for the lower-byte of the aligned value).
* Sampling the first value of each octave:
* [0] 0x1000 = 1.0000 ; 1 / sqrt(1.0000)
* [48] 0x0e4f = 0.8940 ; 1 / sqrt(1.2500)
* [96] 0x0d10 = 0.8164 ; 1 / sqrt(1.5000)
* [144] 0x0c0a = 0.7520 ; 1 / sqrt(1.7500)
* And representative sub-entries within octave 0 (mantissa in [0x8000, 0x8100)):
* [0] 0x1000 = 1.0000 ; 1 / sqrt(0x8000)
* [1] 0x0fe0 = 0.9922 ; 1 / sqrt(0x8100)
* [2] 0x0fc1 = 0.9846 ; 1 / sqrt(0x8200)
* [3] 0x0fa3 = 0.9773 ; 1 / sqrt(0x8300)
* [4] 0x0f85 = 0.9700 ; 1 / sqrt(0x8400)
* [5] 0x0f68 = 0.9629 ; 1 / sqrt(0x8500)
* [6] 0x0f4c = 0.9561 ; 1 / sqrt(0x8600)
* [7] 0x0f30 = 0.9492 ; 1 / sqrt(0x8700)
*
* The algorithm's `addi -64 / sll 1 / lh` selects the entry at `(aligned - 64) * 2` for the case where `aligned` has its top bit at bit 24.
* After the sllv/srav pair, `aligned` always lands in `[0x80, 0x100)`
* (with top bit at bit 24 → after `sub $aligned - 64`, the index sits in `[0x40, 0x80) * 2 = [0x80, 0x100)` bytes = entries [64, 128) within the sqrtbl).
* The earlier 64 entries (octave 0) are reached when the magnitude after shifting puts the top bit below bit 24 (the `sllv` branch),
* and the load upper_halves of the table bracket the input range.
* The later 64 entries (octaves 2-3) are the `srav` branch when the magnitude's top bit is well above bit 24.
*
* 192-entry table is reproduced verbatim from libgte (verified against libpsn00b/psxgte/vector.s:100-123 — 24 rows × 8 halfwords, last entry 0x0804). */
internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
0x1000, 0x0fe0, 0x0fc1, 0x0fa3, 0x0f85, 0x0f68, 0x0f4c, 0x0f30,
0x0f15, 0x0efb, 0x0ee1, 0x0ec7, 0x0eae, 0x0e96, 0x0e7e, 0x0e66,
0x0e4f, 0x0e38, 0x0e22, 0x0e0c, 0x0df7, 0x0de2, 0x0dcd, 0x0db9,
0x0da5, 0x0d91, 0x0d7e, 0x0d6b, 0x0d58, 0x0d45, 0x0d33, 0x0d21,
0x0d10, 0x0cff, 0x0cee, 0x0cdd, 0x0ccc, 0x0cbc, 0x0cac, 0x0c9c,
0x0c8d, 0x0c7d, 0x0c6e, 0x0c5f, 0x0c51, 0x0c42, 0x0c34, 0x0c26,
0x0c18, 0x0c0a, 0x0bfd, 0x0bef, 0x0be2, 0x0bd5, 0x0bc8, 0x0bbb,
0x0baf, 0x0ba2, 0x0b96, 0x0b8a, 0x0b7e, 0x0b72, 0x0b67, 0x0b5b,
0x0b50, 0x0b45, 0x0b39, 0x0b2e, 0x0b24, 0x0b19, 0x0b0e, 0x0b04,
0x0af9, 0x0aef, 0x0ae5, 0x0adb, 0x0ad1, 0x0ac7, 0x0abd, 0x0ab4,
0x0aaa, 0x0aa1, 0x0a97, 0x0a8e, 0x0a85, 0x0a7c, 0x0a73, 0x0a6a,
0x0a61, 0x0a59, 0x0a50, 0x0a47, 0x0a3f, 0x0a37, 0x0a2e, 0x0a26,
0x0a1e, 0x0a16, 0x0a0e, 0x0a06, 0x09fe, 0x09f6, 0x09ef, 0x09e7,
0x09e0, 0x09d8, 0x09d1, 0x09c9, 0x09c2, 0x09bb, 0x09b4, 0x09ad,
0x09a5, 0x099e, 0x0998, 0x0991, 0x098a, 0x0983, 0x097c, 0x0976,
0x096f, 0x0969, 0x0962, 0x095c, 0x0955, 0x094f, 0x0949, 0x0943,
0x093c, 0x0936, 0x0930, 0x092a, 0x0924, 0x091e, 0x0918, 0x0912,
0x090d, 0x0907, 0x0901, 0x08fb, 0x08f6, 0x08f0, 0x08eb, 0x08e5,
0x08e0, 0x08da, 0x08d5, 0x08cf, 0x08ca, 0x08c5, 0x08bf, 0x08ba,
0x08b5, 0x08b0, 0x08ab, 0x08a6, 0x08a1, 0x089c, 0x0897, 0x0892,
0x088d, 0x0888, 0x0883, 0x087e, 0x087a, 0x0875, 0x0870, 0x086b,
0x0867, 0x0862, 0x085e, 0x0859, 0x0855, 0x0850, 0x084c, 0x0847,
0x0843, 0x083e, 0x083a, 0x0836, 0x0831, 0x082d, 0x0829, 0x0824,
0x0820, 0x081c, 0x0818, 0x0814, 0x0810, 0x080c, 0x0808, 0x0804,
typedef Struct_(Binds_SetGteWorld) {
M3_S2* transform;
};
#define RegUse_(proc_name) (tmpl(RegUse,proc_name))
typedef Struct_(RegUse_normalize_v3s4_proc) {
Reg scratch; // Scratch base carrier.
Reg src_ptr;
Reg dst_ptr;
Reg recip_est; // |v|² sum + shift-input + sqrtbl[index]
Reg norm; Reg shift;
Reg src_x;
union { Reg mac1_scratch; } t3;
union { Reg mac2_scratch; } t4;
union { Reg shift_count, btarget, lookup_addr, src_z; } t5;
};
/* ─── Full normalize (all 4 stages inline) ───
* Generic 4-stage GTE normalize (SQR → sum+LZCR → align+sqrtbl → GPF+srav).
*
* Parameterized by caller-provided scratch base + src/dst offsets.
* The caller passes r_src_offset and r_dst_offset as compile-time constants
* (typically derived from O_ macros in the caller's struct schema, e.g., `O_(CallerBundleScratch, fwd)`).
*
* This design lets any caller (with a scratch base + struct schema) use `normalize_v3s4_proc`
* without putting magic offsets in the C-side bundle helper — the offsets come from O_ macros at the call site.
*
* Body uses 9 GPRs (r_src_ptr..r_branch_tmp):
* r_src_ptr, r_dst_ptr : src/dst pointers (computed from r_scratch + caller offsets)
* r_tmp : src.x PRESERVED across stages 1-2 (NOT clobbered by mfc2 MAC2) → fed to IR1 in stage 4
* r_mac1_scratch : MAC1 result scratch (also holds aligned |v|² in stage 3)
* r_mac2_scratch : MAC2 result scratch → result.x after stage 4 sra
* r_recip_est : src.y PRESERVED across stages 1-2 → fed to IR2 in stage 4 → result.y
* r_norm : |v|² sum (stage 2) → half-shift (stage 3) → 1/|v| (stage 4 IR0)
* r_shift : shift count SAVED in stage 3 → consumed by stage 4 srav
* r_branch_tmp : src.z PRESERVED across stages 1-2 → fed to IR3 in stage 4 → result.z (also sqrtbl base addr)
*
* Atom_labels are srav_path / aligned_done
* (NOT namespaced — they're internal to this proc;
* the metaprogram's per-atom-name enum emission handles any collision across different atoms/files that share the same labels).
*
* Pool cost: 11 GPRs (well within the 9-10 caller-trash GPR budget when r_scratch is a wave-context carrier).
*
* Direct port of PSYQ libgte msc02.rel.text VectorNormal disassembly (0x800160a0..0x8001615c).
* Words: ~59 (matches libgte 0x800160a0..0x8001615c at +/- 0-2 words for BD-slot reshuffling).
* Sqrtbl: hardcoded to 0x800185B4 (libgte msc02.rel.data). Note: swapped to local.
* Pipeline: clobbers IR0..3, MAC1..3, LZCS, LZCR.
*/
internal MipsAtom* normalize_v3s4_proc(AtomArena_R aa, U2 src_offset, U2 dst_offset, RegUse_normalize_v3s4_proc r)
MipsAtom_Proc_(aa, {
add_si(r.src_ptr, r.scratch, src_offset), /* r_src_ptr = &src */
/* Load src.x/y/z from r_src_ptr (caller-determined address) into r_tmp/r_recip_est/r_branch_tmp.
* r.rt1_src_x holds src.x throughout stages 1-2 — r_mac2_scratch is clobbered to MAC2 in stage 1.5 (line below). */
mac_load_v3s4(r.src_x, r.recip_est, r.t5.lookup_addr, r.src_ptr, 0),
/* Stage 1: mtc2 src → IR1/2/3, SQR fires. */
LdSlot_ mac_gte_sqr_v3s4(r.src_x, r.recip_est, r.t5.src_z, LdSlot_ nop),
/* Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS. */
mac_gte_mv_from_data_r_mac123(r.t3.mac1_scratch, r.t4.mac2_scratch, r.norm), LdSlot_ nop,
add_u_self( r.norm, r.t3.mac1_scratch),
add_u_self( r.norm, r.t4.mac2_scratch),
gte_mv_to_data_r( r.norm, C2_LZCS), LdSlot_ nop2,
gte_mv_from_data_r(r.shift, C2_LZCR), LdSlot_ nop,
/* Stage 3: round LZCR to even, compute half-shift, align |v|² to bit 24.
* r_norm holds |v|² sum; r_shift holds the LZCR count from mfc2.
* After the component: r_shift = even(LZCR), r_norm = half-shift, r_mac1_scratch = |v|². */
mac_lzcr_round_even_half_shift(r.shift, r.norm, r.t3.mac1_scratch),
/* r_branch_tmp = LZCR - 24 (overwrites r_branch_tmp; src.z no longer needed after SQR) */
add_si( r.t5.btarget, r.shift, -24),
branch_lt_zero(r.t5.btarget, atom_offset(aligned_done, srav_path)), BdSlot_ nop, /* bltz → srav_path (LZCR < 24 path) */
jump_rel(atom_offset(srav_path, aligned_done)), /* b → aligned_done (LZCR >= 24 path) */
BdSlot_ shift_lleft_var(r.t3.mac1_scratch, r.t3.mac1_scratch, r.t5.btarget), /* src=sum (r_mac1_scratch), dst=same */
atom_label(srav_path)
li_s( r.t5.shift_count, 24),
sub_s(r.t5.shift_count, r.t5.shift_count, r.shift),
shift_aright_var(r.t3.mac1_scratch, r.t3.mac1_scratch, r.t5.shift_count), /* src=sum (r_mac1_scratch), dst=same */
atom_label(aligned_done)
// Save the shift count to r_shift before the next 5 instructions overwrite r_norm (the sqrtbl lookup loads 1/|v| into r_norm, which becomes IR0 in stage 4).
or_u(r.shift, r.norm, 0), /* r_shift ← shift count (preserved through stage 4) */
/* r_mac1_scratch holds |v|² aligned (top bit at bit 7). */
add_si( r.t3.mac1_scratch, r.t3.mac1_scratch, -64),
shift_lleft(r.t3.mac1_scratch, r.t3.mac1_scratch, 1),
mac_load_word_imm(r.t5.lookup_addr, & gte_normalize_sqr_tbl), add_u_self(r.t5.lookup_addr, r.t3.mac1_scratch),
load_half(r.norm, r.t5.lookup_addr, 0), /* r_norm = sqrtbl[aligned-64] = 1/|v| (IR0 in stage 4) */
/* r_branch_tmp held the sqrtbl base+index, NOT src.z. Reload src.z from scratch now that r_branch_tmp is free. */
LdSlot_ load_word(r.t5.src_z, r.src_ptr, O_(V3_S4,z)), /* r_branch_tmp = src.z (for IR3 in stage 4) */
/* Stage 4: GPF + srav finalize (r_shift = shift count, r_norm = 1/|v|). */
LdSlot_ mac_gte_general_purpose_interopolation(
r.norm,
r.src_x, /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */
r.recip_est,
r.t5.src_z, /* IR3 = src.z (reloaded) */
r.t4.mac2_scratch, r.recip_est, r.t5.src_z,
LdSlot_ add_si(r.dst_ptr, r.scratch, dst_offset), // pre-laoding destination to register here.
LdSlot_ nop
),
/* sra by r_shift = (31-LZCR)/2 (saved before sqrtbl lookup) */
mac_shift_aright_var_v3_self(r.t4.mac2_scratch, r.recip_est, r.t5.src_z, r.shift),
/* Store result.x/y/z to r_dst_ptr (caller-determined dst address). */
mac_store_v3s4(r.t4.mac2_scratch, r.recip_est, r.t5.src_z, r.dst_ptr, 0),
mac_yield()
})
#pragma endregion Atom Procs
#pragma region Baked Atoms
typedef Struct_(Binds_SetGteMT3S2S4) {
MT3_S2S4* transform;
};
internal MipsAtom_(set_gte_mt3s2s4) atom_info(
atom_bind(Binds_SetGteMT3S2S4)
internal MipsAtom_(set_gte_world) atom_info(
atom_bind(Binds_SetGteWorld)
, atom_reads(R_TapePtr)
){
/* Pop matrix address from tape into R_T3 ($11) */
load_word(R_T3, R_TapePtr, O_(Binds_SetGteMT3S2S4,transform)),
add_ui_self( R_TapePtr, S_(Binds_SetGteMT3S2S4)),
load_word(R_T3, R_TapePtr, O_(Binds_SetGteWorld,transform)),
add_ui_self( R_TapePtr, S_(Binds_SetGteWorld)),
/* Load 3x3 Rotation + 3x1 Translation from R_T3 into GTE CONTROL Regs (ctc2) */
load_word(R_T0, R_T3, 0), load_word(R_T1, R_T3, 4),
load_word(R_T0, R_T3, 0), load_word(R_T1, R_T3, 4),
gte_mv_to_ctrl_r(R_T0, gte_cr_RT11), gte_mv_to_ctrl_r(R_T1, gte_cr_RT12),
load_word(R_T0, R_T3, 8), load_word(R_T1, R_T3, 12), load_word(R_T2, R_T3, 16),
gte_mv_to_ctrl_r(R_T0, gte_cr_RT13), gte_mv_to_ctrl_r(R_T1, gte_cr_RT21), gte_mv_to_ctrl_r(R_T2, gte_cr_RT22),
+24 -135
View File
@@ -161,8 +161,6 @@ enum {
gte_cmd_nclip = 0x06, /* Normal Clipping (Backface culling) */
gte_cmd_op = 0x0C, /* Outer Product */
gte_cmd_mvmva = 0x12, /* Matrix Vector Multiply & Add (Custom math) */
gte_cmd_sqr = 0x28, /* Square vector — MAC[i] = IR[i]²; IR[i] ← MAC[i] saturated */
gte_cmd_gpf = 0x3D, /* General-purpose Interpolation — MAC[i] = IR0 * IR[i] */
/* --- GTE Command Bit-Field Layout ---
* A GTE command word (sent to COP2 with RS=1) is laid out as:
@@ -173,47 +171,19 @@ enum {
* +------------+--+-----+------+------+------+------+---+--------+----------+
* \_____ GTE_PAYLOAD _____/ \__ GTE_CMD __/
*
* Shifts/masks below are the *bit positions* and *bit widths* of each configurable field, used by the ENC_GTE_CMD encoder.
* Shifts/masks below are the *bit positions* and *bit widths* of each
* configurable field, used by the ENC_GTE_CMD encoder.
* Mirrors the OPCODE_SHIFT / RS_SHIFT convention used in mips.h.
*/
gte_shift_sf = 19, gte_width_sf = 1,
gte_shift_mx = 17, gte_width_mx = 2,
gte_shift_v = 15, gte_width_v = 2,
gte_shift_cv = 13, gte_width_cv = 2,
gte_shift_lm = 10, gte_width_lm = 1,
gte_shift_cmd = 0, gte_width_cmd = 6,
/* Fake command number (bits 24-20) — IGNORED by the GTE hardware per PSX-SPX `geometrytransformationenginegte.md` line 48.
* libgte's compiler emits non-zero values in this field as a disassembly signature. */
gte_shift_fake_cmd = 20,
gte_width_fake_cmd = 5,
gte_shift_sf = 19, gte_width_sf = 1, gte_mask_sf = 0x1,
gte_shift_mx = 17, gte_width_mx = 2, gte_mask_mx = 0x3,
gte_shift_v = 15, gte_width_v = 2, gte_mask_v = 0x3,
gte_shift_cv = 13, gte_width_cv = 2, gte_mask_cv = 0x3,
gte_shift_lm = 10, gte_width_lm = 1, gte_mask_lm = 0x1,
gte_shift_cmd = 0, gte_width_cmd = 6, gte_mask_cmd = 0x3F,
};
/* --- GTE Control Register Aliases (Pitfall 1) ---
* Three pairs of aliases map to the SAME C2 control-register slot on real silicon:
* C2[24] = gte_cr_RBK (background R) | gte_cr_OFX (screen offset X)
* C2[25] = gte_cr_GBK (background G) | gte_cr_OFY (screen offset Y)
* C2[26] = gte_cr_BBK (background B) | gte_cr_H (projection plane distance H)
* Cross-alias writes inside one atom body, or across the wave-context boundary,
* silently clobber each other. The metaprogram's check_gte_cr_alias_writes
* (CHECK_RULES row) warns about each pair per source. See
* docs/gte_reference.md §"Control-register alias table" for the silicon
* rationale and the libgte outer-product convention.
*/
/* --- RT-matrix packed-slot convention (Pitfall 4) ---
* The silicon packs two 16-bit RT elements per 32-bit C2 slot:
* C2[2] = (RT22 << 16) | RT13 (gte_cr_RT13 writes the low half, gte_cr_RT22 writes the high half)
* C2[4] = (RT33 << 16) | RT22 (gte_cr_RT22 writes the low half — clobbers prior RT22 value if RT13 was also written)
* OP and MVMVA read D1/D2/D3 from these packed slots. The libgte outer-product
* convention (see ac_apply_matrix_lv at gte.atom.c:108-122) writes C2[2] then
* C2[4] in sequence; the SECOND write's low half is RT22, not RT13. An agent
* who writes gte_cr_RT13 then gte_cr_RT22 to the SAME source GPR clobbers the
* RT13 value. See docs/gte_reference.md §"RT-matrix packed-slot convention"
* for the canonical write pattern.
*/
/* --- GTE Control Register Indices (for ctc2/cfc2) ---
* Preprocessor-visible integer ids for the COP2 control register file.
* Each enum value is bound to a parallel `_Code` `#define` so the preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
@@ -273,10 +243,10 @@ enum { _C2_OPS_ = 0
* bit 1 (0x02): register class — 0 = data, 1 = control
* bit 2 (0x04): direction — 0 = read, 1 = write
*
* The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit numbers as general MIPS `cop_mf` / `cop_mt` defined in mips.h
* The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit numbers as the general MIPS `cop_mf` / `cop_mt` defined in mips.h
* (which target the data register file on any coprocessor).
* They are re-aliased here so the four-way table reads like the spec mnemonics (MFC2 / CFC2 / MTC2 / CTC2)
* and so the encoding is next to its only consumer (this header).
* and so the encoding lives next to its only consumer (this header).
*
* Vendor mnemonic aliases (gte_mfc2 / gte_mtc2 / gte_cfc2 / gte_ctc2) live in gte_vendor_sym.h. */
enum { _C2_TX_SUBS_ = 0
@@ -339,24 +309,23 @@ enum { _C2_TX_SUBS_ = 0
/* GTE Command Format
* Opcode is always MIPS_OP_COP2, RS is always 1 (CO).
* Lower 25 bits are GTE-specific command payload.
* The lower 25 bits are the GTE-specific command payload.
*
* The `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs` pattern in mips.h:
* The granular `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs` pattern in mips.h:
* Each one self-masks and shifts its own field, so a caller can build up a GTE command piece by piece
* (handy for state-driven MVMVA emitters that vary one field at a time).
*
* `ENC_GTE_CMD` is an all-in-one convenience for emitting a full command word.
* `ENC_GTE_CMD` is the all-in-one convenience for emitting a full command word in one go.
* It just ORs the per-field encoders together. */
#define gte_cmd_base (enc_op(op_cop2) | (1 << 25))
/* Per-field encoders. Each one does (value & mask) << shift on its own. */
#define enc_gte_sf(sf) ((sf) << gte_shift_sf )
#define enc_gte_mx(mx) ((mx) << gte_shift_mx )
#define enc_gte_v(v) ((v) << gte_shift_v )
#define enc_gte_cv(cv) ((cv) << gte_shift_cv )
#define enc_gte_lm(lm) ((lm) << gte_shift_lm )
#define enc_gte_cmd(cmd) ((cmd) << gte_shift_cmd )
#define enc_gte_fake_cmd(x) ((x) << gte_shift_fake_cmd)
#define enc_gte_sf(sf) (((sf) & gte_mask_sf ) << gte_shift_sf )
#define enc_gte_mx(mx) (((mx) & gte_mask_mx ) << gte_shift_mx )
#define enc_gte_v(v) (((v) & gte_mask_v ) << gte_shift_v )
#define enc_gte_cv(cv) (((cv) & gte_mask_cv ) << gte_shift_cv )
#define enc_gte_lm(lm) (((lm) & gte_mask_lm ) << gte_shift_lm )
#define enc_gte_cmd(cmd) (((cmd) & gte_mask_cmd) << gte_shift_cmd)
/* Composite: all six GTE fields + the COP2/CO base. */
#define enc_gte_cmdw(sf, mx, v, cv, lm, cmd) ( \
@@ -394,11 +363,11 @@ enum { _C2_TX_SUBS_ = 0
* (the perspective divide happens regardless of `sf`).
*
* If we emit a strictly-spec-compliant word (`sf=0`, reserved bits clear),
* PCSX-Redux's GTE checks those bits more strictly than the silicon does and RTPT silently no-ops.
* The floor's screen coordinates come out as raw projection-of-rotation (Z never divided),
* PCSX-Redux's GTE checks those bits more strictly than the silicon does and RTPT silently no-ops
* the floor's screen coordinates come out as raw projection-of-rotation (Z never divided),
* `nclip` ends up wrong, and the triangle is culled.
*
* So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern.
* So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern everyone has shipped for 25 years.
* NCLIP / OP / MVMVA stay spec-clean — their reserved bits really are zero in the original PsyQ source.
* --------------------------------------------------------------------------
*/
@@ -409,92 +378,11 @@ enum { _C2_TX_SUBS_ = 0
#define gte_cmdw_nclip (gte_cmd_base | enc_gte_cmd(gte_cmd_nclip))
#define gte_cmdw_op (gte_cmd_base | enc_gte_cmd(gte_cmd_op ))
#define gte_cmdw_outer_product gte_cmdw_op /* "outer product" -- NOCASH/Sdk terminology */
#define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology.
* RGA(Lengyel): the GTE OP is a 3D signed-16-bit D x IR cross, not a generic RGA exterior product.
* The wedge alias is the 3D complement interpretation of the same 3 scalars (MAC1..MAC3). */
#define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology */
#define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva))
/* MVMVA with sf=0 (no shift, full-integer), cv=3 (no translation), v=3 (IR vector input).
* Reads input from IR1/2/3 (loaded via mtc2 rt, C2_IRx). MAC1/2/3 = RT row · IR (full product, no >>12).
* Per PSX-SPX: SAR (sf*12) with sf=0 = SAR 0 = no shift. */
#define gte_cmdw_mvmva_sf0_ir (gte_cmd_base | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_cmd(gte_cmd_mvmva))
/* MVMVA with sf=1 (>>12 shift, 4.12 fixed-point), cv=3 (no translation), v=3 (IR): for ApplyMatrixLV.
* Reads input from IR1/2/3 (loaded via mtc2 rt, C2_IRx). MAC1/2/3 = (RT row · IR) >> 12.
* Per PSX-SPX: SAR (sf*12) with sf=1 = SAR 12 = arithmetic right-shift by 12.
* This matches the libgte C-side ApplyMatrixLV output (R*pos >> 12). */
#define gte_cmdw_mvmva_ir (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_cmd(gte_cmd_mvmva))
/* MVMVA: sf=0, mx=3 (Light matrix), v=3 (IR), cv=3 (no TR).
* For pass1 of the C11 two-pass decomposition. Reads L matrix.
* Since L matrix is typically zero, pass1 contributes 0 to the combine. */
#define gte_cmdw_mvmva_sf0_mx3_v3_cv3 (gte_cmd_base | enc_gte_sf(0) | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
/* MVMVA: sf=1 (>>12), mx=3 (Light matrix), v=2 (V0), cv=0 (with TR).
* Matches the C11 ApplyMatrixLV pass 2 command word (0x49E012) exactly.
* The combine is (pass1 << 3) + pass2. */
#define gte_cmdw_mvmva_pass2_c11 (gte_cmd_base | enc_gte_sf(1) | enc_gte_v(2) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
/* MVMVA: sf=0, mx=3, v=2, cv=0. Matches the C11 pass 1 command. */
#define gte_cmdw_mvmva_pass1_c11 (gte_cmd_base | enc_gte_v(2) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
#define gte_cmdw_mvmva_no_tr gte_cmdw_mvmva_ir
/* MVMVA pass 2 — C11 ApplyMatrixLV command.
* Decoded: op_cop2 | CO | fake_cmd=4 | sf=1 (>>12) | mx=0 (RT matrix) | v=3 (IR) | cv=3 (no translation) | lm=0 | cmd=MVMVA.
* Reads (RT row · IR) >> 12 into MAC1/2/3. Per-field composition (no opaque literal)
* keeps the bit layout visible at the call site + matches the libgte C-side byte-exact. */
#define gte_cmdw_mvmva_c11_pass2 (gte_cmd_base | enc_gte_fake_cmd(4) | enc_gte_sf(1) | enc_gte_v(3) | enc_gte_mx(0) | enc_gte_cv(3) | enc_gte_cmd(gte_cmd_mvmva))
/* MVMVA: sf=1 (>>12), mx=0 (RT matrix), v=0 (V0), cv=3 (no TR). */
#define gte_cmdw_mvmva_sf1_mx0_v0_cv3 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_v(0) | enc_gte_mx(0) | enc_gte_cmd(gte_cmd_mvmva))
/* RTPS with sf=1 (12-bit shift, no translation): matches the output of libgte's
* ApplyMatrixLV when the GTE pipeline expects R*pos >> 12. The shift produces
* values like (-270, 710, 1713) which match the C11 reference path. */
#define gte_cmdw_rtps_sf1 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_cmd(gte_cmd_rtps))
/* SQR / GPF cosmetic-bits compat helpers.
* Each command's `_compat` macro ORs in the `fake_cmd` field value libgte happens to emit.
* The hardware ignores these bits (per PSX-SPX line 48). */
#define gte_cmdw_sqr_fake_sig enc_gte_fake_cmd(0x0A)
#define gte_cmdw_gpf_fake_sig enc_gte_fake_cmd(0x19)
/* SQR — Square Vector.
* PSX-SPX `geometrytransformationenginegte.md` §"SQR":
* [MAC1,MAC2,MAC3] = [IR1*IR1, IR2*IR2, IR3*IR3] SHR (sf*12)
* [IR1,IR2,IR3] = [MAC1,MAC2,MAC3] (saturated to 0x7FFF when lm=1)
* Sourced verbatim from libgte msc02 VectorNormal disassembly at 0x800160b0:
* 0x4AA00428 = gte_cmd_base | gte_cmdw_sqr_compat | enc_gte_lm(1) | enc_gte_cmd(0x28)
* bit 19 sf=0
* bit 10 lm=1
* bits 5-0 cmd=0x28=SQR
* bits 24-20 = 0x0A (libgte "nonsense SDK command number" signature) */
#define gte_cmdw_sqr (gte_cmd_base | enc_gte_cmd(gte_cmd_sqr) | enc_gte_lm(1) | gte_cmdw_sqr_fake_sig)
/* GPF — General-purpose Interpolation.
* PSX-SPX `geometrytransformationenginegte.md` §"GPF":
* [MAC1,MAC2,MAC3] = (([IR1,IR2,IR3] * IR0) + [MAC1,MAC2,MAC3]) SAR (sf * 12)
* [IR1,IR2,IR3] = [MAC1,MAC2,MAC3]
* Sourced verbatim from libgte msc02 VectorNormal disassembly at 0x8001613c:
* 0x4B90003D = gte_cmd_base | gte_cmdw_gpf_compat | enc_gte_cmd(0x3D)
* bit 19 sf = 0
* bit 10 lm = 0
* bits 5-0 cmd = 0x3D = GPF
* bits 24-20 = 0x19 (libgte "nonsense SDK command number" signature) */
#define gte_cmdw_gpf (gte_cmd_base | enc_gte_cmd(gte_cmd_gpf) | gte_cmdw_gpf_fake_sig)
/* Mask to round LZCR (leading-zero/ones count, range 1..32 per PSX-SPX cop2r31)
* down to even. The normalize_v3s4 half-shift logic computes (31 - LZCR) >> 1;
* clearing bit 0 ensures the subtraction result is always odd,
* so the >> 1 division is consistent (no 0.5 loss). */
enum {
gte_lzcr_even_mask = 0xFFFE, /* all bits except bit 0 */
};
#define gte_cmdw_rotate_translate_perspective_single gte_cmdw_rtps
#define gte_cmdw_rotate_translate_perspective_triple gte_cmdw_rtpt
/* RGA(Lengyel): RTPS/RTPT consume the matrix expansion of a rigid transformation (rotation matrix + translation vector) loaded into the RT/TR control registers.
* For unitized points the same result equals the motor antiproduct; the GTE executes the LA form, not a symbolic antiproduct. */
/* PsyQ compatibility bits for AVSZ3 (Bits 20, 22, 24 must be set) */
#define gte_cmdw_psyq_avsz3_compat (0x15 << 20)
@@ -545,6 +433,7 @@ enum {
#define gte_lw_v2_z(base) enc_gte_lw(gte_in_v2_z, (base), GTE_Z_Offset)
/* gte_load_vN(r_ptr, base) — placeholder-punned lwc2 loaders
*
* Emits `.word` constants encoding `lwc2 $N, off(<base>)` for the chosen GTE vector register, where `<base>` is the GPR number you pass in
* (typically one of R_T4..R_T9 for the standard "3-pointer" pattern).
*
+93 -214
View File
@@ -12,57 +12,68 @@
#endif
#pragma region Tape Drive
/* -----------------------------------------------------------------------------------------------------------
/* -----------------------------------------------------------------------------
* TAPE DRIVE ABI
* -----------------------------------------------------------------------------------------------------------
* Note(Ed): One of the main purposes of this codebase is to help me learn this,
* as such the information below may not* be entirely realized or finalized conceptually.
* -----------------------------------------------------------------------------------------------------------
* This ABI and its associated legos were directly inspired by researching the work of
* Timothy Lottes and Onat Türkçüoğlu; along with many others. It's the simplest bootstrap of a
* directly executed chain of assemby arrays (Atoms) that terminate with a yield sequence to the next atom.
* These eventually lead to a terminal atom for the tape which is defined below as "tape_exit".
* -----------------------------------------------------------------------------
* Note(Ed): One of the main purposes of this codebase is to help me
* learn this, as such the information below may be entirely realized
* or finalized conceptually.
* -----------------------------------------------------------------------------
* This ABI and its associated legos were directly inspired by researching
* the work of Timothy Lottes and Onat Türkçüoğlu; along with many others.
* It's the simplest bootstrap of a a directly executed chain of assemby
* arrays (Atoms) that terminate with a yield sequence to the next atom.
* These eventually lead to a terminal atom for the tape which is defined
* below as "tape_exit".
*
* It behaves as one of the simplest runtime harnesses ontop of a host-enviornment's execution engine
* to author and compose programs with. From here various conventions can be further applied.
* To make things easier to understand it may be better to focus on what this ABI does not have.
* It does not have have any branching within the tape but relative branches within atoms or between atoms.
* Branching nearly is always downstream. Automatic stack usage is non-existent.
* Push/Pop, FIFO, or Arena/Bump data structures are used by atoms explicitly.
* In it's current form with the C11 macro DSL, the user also has fullfill manual register allocation per atom.
* This behaves as one of the simplest runtime harnesses ontop of a
* host-enviornment's execution engine to author and compose programs with.
* From here various conventions can be further applied.
* To make things easier to understand it may be better to focus on what this
* ABI does not have. It does not have have any branching within the tape but
* relative branches between atoms. Branching nearly is always downstream.
* Stack usage is non-existent. Push/Pop, FIFO, or Arena/Bump data structures
* are used by atoms explicitly. In it's current form withe C11 macro dsl,
* the user also has to do manual register allocation per atom.
*
* One of the remarkable things about utilizing this ABI is its essentially interopable with CPUs, GPUs, FPGA,
* or, basically anything from the 5th generation consoles and onward.
* The ABI directly reflects how all computational hardware must be architected in order to execute
* digital logic effectively on current era tech.
* On the PS1 we don't have access to a few features like multi-threading, speculative execution, or L3 cache;
* but, we can set the foundation for legoing whats required for eventually expanding this ABI's paradigm
* and core atoms to take those newer hardware features into account. For example, you can easily expand
* this to support wave-based execution model on a PS2 or PS3. Not having a stack or
* automatic register allocation means the user cannot ignore excessive argument shuffle across workload or
* waves and thier phases. Crossing ABI boundaries to other runtimes that do has obviouss penalties.
* One of the remarkable things about utilizing this abi is its essentially
* interopable with CPUs, GPUs, FPGA, or, basically anything
* from the 5th generation consoles and onward.
* The ABI directly reflects how all computational hardware must be architected
* in order to execute digital logic effectively on current era tech.
* On the PS1 we don't have access to a few features like multi-threading,
* speculative execution, or L3 cache; but, we can set the foundation for legoing
* whats required baseline wise for eventually expanding the harness and core atoms
* to take those newer hardware features into account. For example, you can easily
* expand this to support wave-based execution model on a PS2 or PS3.
* Not having a stack or automatic register allocation means the user can't ignore
* excessive argument shuffle across workload or waves and thier phases.
* Crossing ABI boundaries to other runtimes that do has an obviouss penalties.
*
* Learning data-oriented code becomes a natural progression. Your not fighting a stack-based procedural
* paradigm that wants to argument shuffle. There is no ambiguity due to the lack of constraints, for example,
* on how the user may "call" a procedure in traditional random dispatch runtimes. The user does have to
* hammer down "rules" or patterns for massaging the compiler to dissolve those call frames; just to get
* the asesmbly into its desired form. The form is obvious, and once the user gets to author these compoonents
* it becomes a game of tetris.
* Learning data-oreinted code becomes a natural progression. Your not fighting
* a stack-based procedural paradigm that wants to argument shuffle on the stack
* by lack of constraints on how the user may "call" a procedure. The user doesn't
* have to hammer down "rules" or patterns to know how to massage the compiler
* to get the asesmbly into its natural form. The form is obvious, and once
* the user gets to author their compoonents it becomes a game of tetris.
*
* Another feature is this ABI is very compatible with bootstrapping and developing simple toolchains built off
* of bit-packed annotated command streams the user can directly author, maintatain, and immediately execute.
* That being like a color forth, or maybe something more familar like an immediate mode library
* for various systems such as GUIs. This can make the tetris less of a chore with some helpful policy
* generation for allocation of registers, helping to choose resuable components, designing DSL on the fly, etc.
* -----------------------------------------------------------------------------------------------------------
* TODO(Ed): We need pretty ascii diagrams and proper guides, articles, etc.
* -----------------------------------------------------------------------------------------------------------
* For now this ideation has just started functioning. I'm abusing C11 & a lua metaprogram to help establish
* a hybrid toolchain to ideate on a traditional text-based authoring UX for this paradigm.
* If pcsx-redux provides viable hot-reload and persistent data storage beyond save-states
* (just copying ram to filesystem), I can author a color forth to mess around with.
* With either an editor in-emulator or on the actual machine itself. Assembly is tedius,
* but I think this codebase most likely has a pretty ergonomic flavor worst case...
* Another feature is this ABI is very compatible with bootstrapping and developing
* simple toolchains built off of bit-packed annotated command streams the user can
* directly author, maintatain, and immediately execute. That being a color forth.
* This can make the tetris less of a chore with some helpful policy generation for
* allocation of registers, helping to choose resuable components, designing DSL on
* the fly, etc.
* -----------------------------------------------------------------------------
* TODO(Ed): We ned pretty ascii diagrams and proper guides, articles, etc.
* -----------------------------------------------------------------------------
* For now this thing is just functioning and I'm abusing C11 + a lua metaprogram
* to help establish a hybrid toolchain to ideate on a traditional text-based
* authoring UX for this paradigm.
* If pcsx-redux gets me viable hot-reload and persistent data storage beyond
* save-states (just copying ram to filesystem). I can author a color forth to
* mess around with, with an editor in-emulator or on the actual machine itself.
* Assembly is tedius, but I think this codebase most likely has some of the most,
* ergonomic you can come across..
* */
/* Register Allocation Info */
enum {
@@ -100,62 +111,42 @@ enum {
// S 0-7
};
typedef U2 Reg; // Register parameter used with atom or atom component procedures
typedef U4 const MipsCode; // Underlying type to mips asm words.
typedef Slice_(MipsCode);
typedef U4 const MipsAtom;
typedef Slice_(MipsAtom);
// Sometimes a user will define a bundle of atoms that represent a procedure of work as:
// MipsAtom* <identifier>[...];
// Unfortuantely if using slice_from_array it will make the slice's pointer: MipsAtom** so this enforce its defined as MipsAtom*
// TODO(Ed): Alternatively we can make the MipsAtom an opaque pointer to the atom... so that the blow returns 'MipsAtom'.
#define atombundle_from_array(array) (Slice_MipsAtom){.ptr=array[0],.len=Array_len(array)}
// Underlying type to an ptr to an array of mips asm words that must terminate with an ac_yield.
typedef U4 const MipsAtom; // Underlying type to an array of mips asm words that must terminate with an ac_yield.
#define MipsAtom_(sym) MipsCode sym [] align_(4) =
// Used for atoms with value-args
// internal MipsAtom* X_proc(AtomArena_R aa, args) MipsAtom_Proc_(X, aa, { body })
// expands to:
// internal MipsAtom* X_proc(AtomArena_R aa, args) { MipsCode atom_comp_code[] align_(4) = { body }; return atomarena_push(aa, slice_from_array(MipsCode, atom_comp_code)); }
// The atom name is derived by the Lua metaprogram from the preceding
// `MipsAtom* X_proc(...)` declaration (backward walk from the macro site,
// strips the `_proc` suffix).
#define MipsAtom_Proc_(aa, ...) { MipsCode atom_comp_code[] align_(4) = __VA_ARGS__; return atomarena_push(aa, slice_from_array(MipsCode, atom_comp_code)); }
// Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names).
// MipsAtomComp_(ac_X) { body }
// expands to:
// MipsCode ac_X[] align_(4) = { body };
#define MipsAtomComp_(sym) MipsCode sym [] align_(4) =
// Used for components with value-args (mandatory `ab` (atom-builder) arg).
// FI_ void ac_X(MipsAtomBuilder_R ab, args) MipsAtomComp_Proc_(ab, { body })
// Used for components with value-args (e.g., ac_format_f3_color).
// FI_ Slice_MipsCode ac_X(args) MipsAtomComp_Proc_(ac_X, { body })
// expands to:
// FI_ void ac_X(MipsAtomBuilder_R ab, args) {
// MipsCode atom_comp_code[] align_(4) = { body };
// atombuilder_push(ab, slice_from_array(MipsCode, atom_comp_code));
// }
// The body must NOT include mac_yield() (the parent atom yields).
// The component name is derived by the Lua metaprogram from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration (backward walk from the macro site).
// Inline-only callers (the generated `mac_<name>` aliases) skip the `ab` arg via metaprogram filtering; escape callers (ac_<name> invoked as a function) pass a long-lived builder.
#define MipsAtomComp_Proc_(ab, ...) { MipsCode atom_comp_code[] align_(4) = __VA_ARGS__; atombuilder_push(ab, slice_from_array(MipsCode, atom_comp_code)); }
// FI_ Slice_MipsCode ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return slice_from_array(MipsCode, ac_X); }
#define MipsAtomComp_Proc_(sym, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; return slice_from_array(MipsCode, sym); }
/* Line-table anchor: gcc only adds a file to the .debug_line file table when the
file contains line-numbered content. Files containing only:
- `MipsAtomComp_` static-array declarations, or
- `MipsAtomComp_Proc_` (force-inline) function bodies whose line info gets
attributed to the call site at the include point are otherwise omitted from the file table,
which breaks the DWARF injection when it tries to resolve atom-component provenance paths.
/* Line-table anchor: gcc only adds a file to the .debug_line file table when the contains line-numbered content.
Files containing only atoms and atom components.
Place `ATOM_FILE_LINE_MARKER();` once at file scope in any `.atom.c` that defines atoms.
Macro expands to a file-scope `internal U4 const` declaration keeps the file in the line table.
The constant is in `.rodata` so the linker may eliminate it.
Two-level concat + `__LINE__` suffix makes the identifier unique per call site
(identifier embeds the source line, so duplicates across `#include`d files don't collide). */
The macro expands to a file-scope `internal U4 const` declaration keeps the file in the line table.
The constant is in `.rodata` and unreferenced; the linker may eliminate it.
The two-level concat + `__LINE__` suffix makes the identifier unique per call site
(the identifier embeds the source line, so duplicates across `#include`d files don't collide). */
#define ATOM_FILE_DEBUGGER_LINE_MARKER(file_name) internal U4 const tmpl(atom_file_debugger_line_marker,file_name) = 0
typedef Slice_MipsAtom Tape;
typedef Slice_(MipsAtom); typedef Slice_MipsAtom Tape;
/* The 'Exit' Atom */
atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(R_RA), nop };
atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(rret_addr), nop };
// TODO(Ed): When we have a substantial workload/throughput, profile each of these to see impact at ABI boundaries.
@@ -199,15 +190,13 @@ FI_ void tape_run_a02_s07(Tape tape) { register U4* tape_ptr rgcc(R_TapePtr) = u
typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; };
FI_ void tb_init(TapeBuilder* tb, FArena* arena) { tb->ptr = arena->start; tb->used = 0; }
FI_ TapeBuilder tb_make_old( FArena* arena) { return (TapeBuilder){ arena->start, 0 }; }
FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ u4_(mem.ptr), mem.len, 0 }; } /* capacity in elements (matches used units) */
FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ mem.ptr, mem.len, 0 }; }
FI_ void tb_emit(TapeBuilder* tb, MipsAtom* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
FI_ void tb_emit(TapeBuilder* tb, MipsCode* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; }
#define tb_emit_(atom) tb_emit(& tb, atom)
#define tb_data_(field, data) tb_data(& tb, u4_(data))
FI_ void tb_emit_bundle(TapeBuilder_R tb, Slice_MipsAtom atoms) { mem_copy(u4_(tb->ptr), u4_(atoms.ptr), S_slice(atoms)); tb->used += atoms.len; }
FI_ Tape tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Tape){ C_(U4*,tb->ptr), tb->used }; }
FI_ Tape tb_slice(TapeBuilder tb) { return (Tape){ C_(U4*,tb.ptr), tb.used }; }
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit))
@@ -242,145 +231,35 @@ atom_dbg_skip MipsAtomComp_(ac_yield_tail) {
add_ui_self(R_TapePtr, S_(MipsCode)),
jump_reg( R_AtomJmp), nop,
};
#pragma endregion Macro Atom Components
#pragma region Atom Builder
#pragma region Mips Atom Builder
// This helps with runtime procedural authoring of mips atoms.
typedef Struct_(FMipsAtom512) { U4 data[512]; U4 used; };
// FArena Related
typedef Relative_(FArena) Struct_(AtomBuilder) { U4 start; U4 capacity; U4 used; };
typedef Relative_(FArena) Struct_(MipsAtomBuilder) { U4 start; U4 capacity; U4 used; };
// Whatever the builder is writting to should most likely coresspond
// to something that can fit within instruction cache?
// Usual way to resolve an atom after the bulder is done.
#define atom_from_atombuilder(ab) C_(MipsAtom*, (ab).start)
FI_ void atombuilder_push(AtomBuilder_R ab, Slice_MipsCode code) {
assert(ab->capacity - ab->used - code.len);
U4 dest = ab->start + ab->used * S_(MipsCode); U4 size = S_slice(code);
mem_copy(dest, u4_(code.ptr), size); ab->used += size;
FI_ void atombuilder_unroll(MipsAtomBuilder_R ab, Slice_MipsCode_R code) {
assert(ab->capacity - ab->used - code->len);
mem_copy(ab->start, u4_(code->ptr), code->len);
mem_bump(ab->start, ab->capacity, & ab->used, code->len);
}
#define atombuilder_push_mac(ab, mac) atombuilder_push(ab, slice_arg_from_array(Slice_MipsCode, mac))
#define atombuilder_unroll_mac(ab, mac) atombuilder_unroll(ab, slice_arg_from_array(Slice_MipsCode, mac))
// When done authoring, utilize this to cap-off the atom (if not utilizing a MipsAtom_Proc).
FI_ void atombuilder_end(AtomBuilder_R ab) { atombuilder_push(ab, slice_from_array(MipsCode, ac_yield)); }
// When done authoring, utilize this to cap-off the atom
FI_ void atombuilder_end(MipsAtomBuilder_R ab) {
mem_copy(ab->start, u4_(ac_yield), S_(ac_yield));
mem_bump(ab->start, ab->capacity, & ab->used, S_(ac_yield));
}
FI_ void tb_emit_atombuilder(TapeBuilder_R tb, AtomBuilder_R ab) { tb_emit(tb, atom_from_atombuilder(ab[0])); }
#define mipsatom_from_builder(ab) (Slice_MipsCode){ab.start, ab.used}
#pragma endregion Mips Atom Builder
#pragma region Atom Arena
// Just a dedicated FArena that is meant to mem_copy and return atom definitions made with MipsAtom_Proc_
typedef Relative_(FArena) Struct_(AtomArena) { U4 start; U4 capacity; U4 used; };
#define atomarena_unused_start(ab) ((ab).start + (ab).used)
FI_ void atomarena_init(AtomArena_R arena, Slice mem) { assert(arena != nullptr);
arena->start = u4_(mem.ptr);
arena->capacity = mem.len;
arena->used = 0;
}
FI_ AtomArena atomarena_make(Slice mem) { AtomArena a; atomarena_init(& a, mem); return a; }
FI_ MipsAtom* atomarena_push(AtomArena_R aa, Slice_MipsCode code) {
assert(aa->capacity - aa->used - code.len);
U4 dest = atomarena_unused_start(aa[0]); U4 size = S_slice(code);
mem_copy(dest, u4_(code.ptr), size); aa->used += size;
return C_(MipsAtom*, dest);
}
FI_ void atomarena_reset(AtomArena_R aa) { aa->used = 0; }
#pragma endregion Atom Arena
#pragma region RegFile (Register File Allocator)
// A specialized allocator utilized to help the user track which registers are bound to values
// that must be preserved for the arena's bounds.
// TODO(Ed): Technically we can do this at comp-time with the metaprogram, but we may have namespace conflicts.
// Unless we follow a convention for #define <Scope_Prefix> or something per register allocation boundary.
/* ABI + tape reserves that are never handed out by alloc. */
U4 const regfile_abi_mask =
(1u << R_0) | (1u << R_AT) |
(1u << R_K0) | (1u << R_K1) |
(1u << R_GP) | (1u << R_SP) |
(1u << R_FP) | (1u << R_RA) |
(1u << R_T8) | (1u << R_T9); /* AtomJmp + TapePtr */
typedef Struct_(RegFile) {
A2_U2 GPR;
A2_U2 GTE;
};
#define regfile(pin_mask) {.GPR={u4_lo(pin_mask), u4_hi(pin_mask)} }
FI_ void regfile_init(RegFile_R rf) {
/* pack the 32-bit ABI mask into the two U2s */
rf->GPR[0] = u4_lo(regfile_abi_mask);
rf->GPR[1] = u4_hi(regfile_abi_mask);
rf->GTE[0] = rf->GTE[1] = 0;
}
FI_ RegFile regfile_make(void) { RegFile rf; regfile_init(& rf); return rf; }
typedef Struct_(RegFile_RInfo) {
U2_R section;
U2 mask;
B2 occupied;
};
FI_ RegFile_RInfo regfile_rinfo(A2_U2 file, Reg r_id) {
U2 s_id = r_id >> 4;
U2_R section = & file[s_id];
U2 mask = u2_(1u << (r_id & 15));
B2 occupied = (section[0] & mask) != 0;
return (RegFile_RInfo){section, mask, occupied};
}
FI_ Reg regfile__alloc_helper(A2_U2 file, Reg r_id) {
Reg result = 0; RegFile_RInfo info = regfile_rinfo(file, r_id);
if (info.occupied == false) {
info.section[0] |= info.mask;
result = r_id;
}
return result;
}
I_ Reg regfile_alloc(RegFile_R rf) {
U2 allocated = 0;
for index_iter(Reg, r_id, R_T0, <=, R_T7) {
allocated = regfile__alloc_helper(rf->GPR, r_id); Jmp_nZero_(allocated,resolved);
}
allocated = regfile__alloc_helper(rf->GPR, R_V0); Jmp_nZero_(allocated,resolved);
allocated = regfile__alloc_helper(rf->GPR, R_V1);
assert(allocated != 0);
resolved: return allocated;
}
FI_ Reg regfile_pin(RegFile_R rf, Reg r_id) {
RegFile_RInfo info = regfile_rinfo(rf->GPR, r_id);
assert(info.occupied == false);
info.section[0] |= info.mask;
return r_id;
}
FI_ void regfile_pin_mask(RegFile_R rf, U4 mask) {
B4 occupied = u4_r(rf->GPR)[0] & mask;
assert(occupied == false);
u4_r(rf->GPR)[0] |= mask;
}
FI_ void regfile_free_mask(RegFile_R rf, U4 mask) {
if (regfile_abi_mask & mask) return;
u4_r(rf->GPR)[0] &= ~mask;
}
FI_ void regfile_free_reg(RegFile_R rf, Reg r_id) {
/* never free the ABI set */
if (regfile_abi_mask & (1u << r_id)) return;
RegFile_RInfo info = regfile_rinfo(rf->GPR, r_id);
info.section[0] &= ~info.mask;
}
FI_ void regfile_reset(RegFile_R rf) {
rf->GPR[0] = u4_lo(regfile_abi_mask);
rf->GPR[1] = u4_hi(regfile_abi_mask);
}
FI_ void regfile_reset_mask(RegFile_R rf, U4 mask) {
rf->GPR[0] = u4_lo(mask);
rf->GPR[1] = u4_hi(mask);
}
#pragma endregion RegFileArena (Register File Allocator)
#pragma region Mips Atom Procs
#pragma endregion Mips Atom Procs
#pragma region Baked Mips Atoms
// These atoms are resolved at compile time and are (usually) statically linked readonly data.
+5 -31
View File
@@ -9,43 +9,17 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(math_atom_c);
#pragma region MACs (Mips Atom Component)
// FI_ Slice_MipsCode ac_load_imm
FI_ Slice_MipsCode ac_load_v2s2(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_half( rs_x, r_base, offset + O_(V3_S2,x)),
load_half( rs_y, r_base, offset + O_(V3_S2,y)),
FI_ Slice_MipsCode ac_load_v2s2(U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v2s2, {
load_half( rs_x, r_base, O_(V3_S2,x)),
load_half( rs_y, r_base, O_(V3_S2,y)),
})
FI_ Slice_MipsCode ac_store_v2s2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
FI_ Slice_MipsCode ac_store_v2s2(U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v2s2, {
store_half(rt_x, base, offset + O_(V2_S2,x)),
store_half(rt_y, base, offset + O_(V2_S2,y)),
})
FI_ Slice_MipsCode ac_load_v3s4(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 rs_z, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_word( rs_x, r_base, offset + O_(V3_S4,x)),
load_word( rs_y, r_base, offset + O_(V3_S4,y)),
load_word( rs_z, r_base, offset + O_(V3_S4,z)),
})
// TODO(Ed): we could generate these mappings properly..
#define ac_load_p3s4 ac_load_v3s4
#define mac_load_p3s4 mac_load_v3s4
FI_ Slice_MipsCode ac_store_v3s4(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_z, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
store_word(rt_x, base, offset + O_(V3_S4,x)),
store_word(rt_y, base, offset + O_(V3_S4,y)),
store_word(rt_z, base, offset + O_(V3_S4,z)),
})
// TODO(Ed): we could generate these mappings properly..
#define ac_store_p3s4 ac_store_v3s4
#define mac_store_p3s4 mac_store_v3s4
FI_ Slice_MipsCode ac_sub_v3s4(AtomBuilder_R ab, U4 rds_x, U4 rds_y, U4 rds_z, U4 rt_x, U4 rt_y, U4 rt_z) atom_dbg_skip MipsAtomComp_Proc_(ab, {
sub_s(rds_x, rds_x, rt_x),
sub_s(rds_y, rds_y, rt_y),
sub_s(rds_z, rds_z, rt_z),
})
FI_ Slice_MipsCode ac_store_rects2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
FI_ Slice_MipsCode ac_store_rects2(U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rects2, {
store_half(rt_x, base, offset + O_(Rect_S2,x)),
store_half(rt_y, base, offset + O_(Rect_S2,y)),
store_half(rt_width, base, offset + O_(Rect_S2,width)),
+7 -63
View File
@@ -7,24 +7,11 @@
#define max(A, B) (((A) > (B)) ? (A) : (B))
#define clamp_bot(X, B) max(X, B)
/* Convention
<Type> ## <Width> _ <Component Type> ## <Component Width>
For types with compound data (Ex: Rotation Matrix & Translation):
<TypeA> ## <TypeB> ## <Width> _ <ComponentTypeA> ## <ComponentWidthA> ## <ComponentTypeB> ## <ComponentWidthB>
A: Array
V: Vector
R: Range
M: Matrix
T: Translation
*/
enum {
v3s2_byteoff = 3, // log2(8), used with shift_left_logical op for index via byte offset.
};
typedef Array_(U1, 2);
typedef Array_(U2, 2);
typedef Array_(U4, 2);
typedef Array_(S2, 2);
typedef Array_(S2, 3);
@@ -39,43 +26,23 @@ typedef Struct_(Extent2_S4) { S4 width; S4 height; };
typedef Struct_(V2_U1) { U1 x; U1 y; };
typedef Struct_(V2_S2) { S2 x; S2 y; };
typedef Struct_(V2_S4) { S4 x; S4 y; };
typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; }; // PSY-Q: SVECTOR
typedef Struct_(V3_S4) { S4 x; S4 y; S4 z; S4 pad; }; // PSY-Q: VECTOR. RGA(Lengyel): Euclidean vector or direction. A zero-weight RGA point is stored as a V3_S4 with the implicit weight dropped.
typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; };
typedef Struct_(V3_S4) { S4 x; S4 y; S4 z; S4 pad; };
typedef Struct_(V4_S2) { S2 x; S2 y; S2 z; S2 w; };
typedef Struct_(V4_S4) { S4 x; S4 y; S4 z; S4 w; };
// typedef Struct_(P3_S4) { S4 x; S4 y; S4 z; S4 w1; }; // RGA(Lengyel): Affine point with implicit weight one. Storage alias of V3_S4. Use P3_S4 when the value is a point.
typedef V3_S4 P3_S4;
typedef Struct_(R1_U2) { U2 p0; U2 p1; };
typedef Struct_(R1_S2) { S2 p0; S2 p1; };
typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; }; // Range-2 Signed 2-Byte (16-bit)
typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; }; // Range-2 Signed 4-Byte (32-bit)
typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; };
typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; };
typedef Struct_(Rect_S2) { S2 x; S2 y; S2 width; S2 height; };
typedef Struct_(Rect_S4) { S4 x; S4 y; S4 width; S4 height; };
typedef Struct_(MT3_S2S4) { A3x3_S2 m; A3_S4 t; }; // PSY-Q: MATRIX. RGA(Lengyel): Matrix expansion of a rigid transformation. GTE utilizes this representation; corresponding motor not constructed here.
typedef Struct_(M3_S2) { A3x3_S2 m; A3_S4 t; };
/* RGA(Lengyel) reserved names (deferred):
* P4_S4 - future flat point with explicit weight (Lengyel/TML FlatPoint3D analog).
* B3_S4 - future 3D bivector (callers store a Complement(Wedge(...)) as a V3_S4).
* Mo8_S4 - future motor. Not introduced until a course operation actually needs composition, interpolation, or inversion. */
typedef Array_(V2_U1, 2);
typedef Array_(V2_S2, 2);
typedef Array_(V2_S2, 3);
typedef Array_(V2_S2, 4);
#define r1u2(p0,p1) (R1_U2){p0,p1}
enum {
fp_one = (1 << 12),
};
#define v3s4_fp_one() v3s4(fp_one, fp_one, fp_one)
#define v2s2(x,y) (V2_S2){x,y}
#define v3s2(x,y,z) (V3_S2){x,y,z,0}
#define v3s4(x,y,z) (V3_S4){x,y,z,0}
@@ -94,28 +61,5 @@ FI_ void add_a3s4_fp(A3_S4_R out_a, A3_S4 b) {
(out_a[0])[2] += b[2] >> 1;
}
FI_ void sub_a3s4(A3_S4_R out_a, A3_S4 b) {
(out_a[0])[0] -= b[0];
(out_a[0])[1] -= b[1];
(out_a[0])[2] -= b[2];
}
FI_ void sub_a3s4_fp(A3_S4_R out_a, A3_S4 b) {
(out_a[0])[0] -= b[0] >> 1;
(out_a[0])[1] -= b[1] >> 1;
(out_a[0])[2] -= b[2] >> 1;
}
FI_ void mul_a3s4(A3_S4_R out_a, A3_S4 b) {
(out_a[0])[0] *= b[0];
(out_a[0])[1] *= b[1];
(out_a[0])[2] *= b[2];
}
FI_ void add_v3s4 (V3_S4_R out_a, V3_S4 b) { add_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
FI_ void add_v3s4_fp(V3_S4_R out_a, V3_S4 b) { add_a3s4_fp(C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
FI_ void sub_v3s4 (V3_S4_R out_a, V3_S4 b) { sub_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
FI_ void sub_v3s4_fp(V3_S4_R out_a, V3_S4 b) { sub_a3s4_fp(C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
FI_ void mul_v3s4 (V3_S4_R out_a, V3_S4 b) { mul_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
FI_ void add_v3s4 (V3_S4_R out_a, V3_S4 b) { add_a3s4 (pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
FI_ void add_v3s4_fp(V3_S4_R out_a, V3_S4 b) { add_a3s4_fp(pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
+13 -22
View File
@@ -18,7 +18,7 @@ I_ U4 align_pow2(U4 x, U4 b) {
#define align_struct(type_width) ((U4)(((type_width) + 3) & ~3))
FI_ void mem_bump(U4 cap, U4*R_ used, U4 amount) {
FI_ void mem_bump(U4 start, U4 cap, U4*R_ used, U4 amount) {
assert(amount <= (cap - used[0]));
used[0] += amount;
}
@@ -58,13 +58,13 @@ typedef Struct_(Str8) { UTF8* ptr; U4 len; };
typedef Struct_(Slice_Str8) { Str8* ptr; U4 len; };
#define slit(string_literal) (Str8){ (UTF8*) string_literal, S_(string_literal) - 1 }
typedef Struct_(Slice) { B1* ptr; U4 len; }; // Untyped Slice (byte-addressable; .len in elements)
FI_ Slice slice_ut_(U4 ptr, U4 len) { return (Slice){(B1*)ptr, len}; }
typedef Struct_(Slice) { U4 ptr, len; }; // Untyped Slice
FI_ Slice slice_ut_(U4 ptr, U4 len) { return (Slice){ptr, len}; }
#define Slice_(type) Struct_(tmpl(Slice,type)) { type* ptr; U4 len; }
typedef Slice_(B1);
#define slice_assert(s) do { assert((s).ptr != 0); assert((s).len > 0); } while(0)
#define slice_end(slice) ((slice).ptr + S_slice(slice) / S_(B1)) /* byte-ptr arithmetic; .len is in elements per slice convention */
#define slice_end(slice) ((slice).ptr + (slice).len)
#define S_slice(s) ((s).len * S_((s).ptr[0]))
#define slice_ut(ptr,len) slice_ut_(u4_(ptr), u4_(len))
@@ -72,30 +72,23 @@ typedef Slice_(B1);
#define slice_to_ut(s) slice_ut_(u4_((s).ptr), S_slice(s))
#define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter)
#define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = Array_decl(type,__VA_ARGS__), .len = Array_len( Array_decl(type,__VA_ARGS__)) }
#define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = Array_len(array) }
#define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = array_decl(type,__VA_ARGS__), .len = array_len( array_decl(type,__VA_ARGS__)) }
#define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = S_(array) }
FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(u4_(s.ptr), s.len); }
FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(s.ptr, s.len); }
#define slice_zero(s) slice_zero_(slice_to_ut(s))
FI_ void slice_copy_(Slice dest, Slice src) {
assert(S_slice(dest) >= S_slice(src));
assert(dest.len >= src.len);
slice_assert(dest);
slice_assert(src);
mem_copy(u4_(dest.ptr), u4_(src.ptr), S_slice(src));
mem_copy(dest.ptr, src.ptr, src.len);
}
#define slice_copy(dest, src) do { \
static_assert(T_same(dest, src)); \
slice_copy_(slice_to_ut(dest), slice_to_ut(src)); \
} while(0)
FI_ Slice slice_bump(U4_R used, U4 start, U4 len, U4 amount) {
assert(len - used[0] - amount);
U4 ptr = start + used[0]; used[0] += amount;
return slice_ut(ptr, amount);
}
typedef Slice_(U1);
typedef Slice_(U4);
#pragma endregion Slice
@@ -105,19 +98,18 @@ typedef Slice_(U4);
typedef Opt_(farena) { U4 alignment, type_width; };
typedef Struct_(FArena) { U4 start, capacity, used; };
FI_ void farena_init(FArena_R arena, Slice mem) { assert(arena != nullptr);
arena->start = u4_(mem.ptr);
arena->start = mem.ptr;
arena->capacity = mem.len;
arena->used = 0;
}
FI_ FArena farena_make(Slice mem) { FArena a; farena_init(& a, mem); return a; }
FI_ Slice farena_bump(FArena_R a, U4 amount) { return slice_bump(& a->used, a->start, a->capacity, amount); }
I_ Slice farena_push(FArena_R arena, U4 amount, Opt_farena o) {
I_ Slice farena_push(FArena_R arena, U4 amount, Opt_farena o) {
if (amount == 0) { return (Slice){}; }
U4 desired = amount * (o.type_width == 0 ? 1 : o.type_width);
U4 to_commit = align_pow2(desired, o.alignment ? o.alignment : MEM_ALIGNMENT_DEFAULT);
U4 ptr = arena->start + arena->used;
mem_bump(arena->capacity, & arena->used, to_commit);
return (Slice){ (B1*)ptr, to_commit };
mem_bump(arena->start, arena->capacity, & arena->used, to_commit);
return (Slice){ ptr, to_commit };
}
FI_ void farena_reset (FArena_R arena) { arena->used = 0; }
FI_ void farena_rewind(FArena_R arena, U4 save_point) {
@@ -125,7 +117,6 @@ FI_ void farena_rewind(FArena_R arena, U4 save_point) {
arena->used -= save_point - arena->start;
}
FI_ U4 farena_save(FArena arena) { return arena.used; }
FI_ U4 farena_unused_start(FArena arena) { return arena.start + arena.used; }
#define farena_push_(arena, amount, ...) farena_push((arena), (amount), opt_(farena, __VA_ARGS__))
#define farena_push_type(arena, type, ...) C_(type*, farena_push((arena), 1, opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr)
#define farena_push_array(arena, type, amount, ...) (tmpl(Slice,type)){ C_(type*, farena_push((arena), (amount), opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr), (amount) }
+15 -22
View File
@@ -1,25 +1,18 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "gen/macs.h"
# include "gen/offsets.h"
# include "bios.h"
# include "mips.h"
# include "gen/macs.h"
# include "gen/offsets.h"
# include "lottes_tape.h"
#endif
ATOM_FILE_DEBUGGER_LINE_MARKER(mips_atom_c);
#pragma region MACs (Mips Atom Components)
FI_ Slice_MipsCode ac_load_word_imm(AtomBuilder_R ab, Reg dst, U4 imm)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_upper_i(dst, u4_hi(imm)),
or_i_self( dst, u4_lo(imm)),
})
#pragma endregion MACs (Mips Atom Components)
#pragma region Baked Atoms
enum {
bios_flushcache = 0x44,
bios_table_addr = 0xA0,
};
/* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0).
* Sequence (per MIPS ABI; arguments in arg registers, RA pushed to stack):
* 1. sp -= 8; sw $ra, 4($sp) ; save RA
@@ -31,14 +24,14 @@ atom_dbg_skip MipsAtomComp_Proc_(ab, {
* 6. sp += 8
*/
internal MipsAtom_(mips_flush_icache) {
add_ui(R_SP, R_SP, -MipsStackAlignment), // sp -= 8
store_word(R_RA, R_SP, S_(U4)), // sw $ra, 4($sp)
add_ui(R_V0, R_0, bios_flushcache), // addiu $a0, $0, 0x44
add_ui(R_T0, R_0, bios_table_addr), // addiu $t0, $0, 0xA0
jump_link(R_T0, R_RA), nop, // jalr $t0, $ra, BD slot
load_word(R_RA, R_SP, S_(U4)), // lw $ra, 4($sp)
jump_reg(R_RA), // jr $ra
add_ui(R_SP, R_SP, MipsStackAlignment), // sp += 8 (BD)
add_ui(rstack_ptr, rstack_ptr, -MipsStackAlignment), // sp -= 8
store_word(rret_addr, rstack_ptr, S_(U4)), // sw $ra, 4($sp)
add_ui(rret_0, rdiscard, bios_flushcache), // addiu $a0, $0, 0x44
add_ui(rtmp_0, rdiscard, bios_table_addr), // addiu $t0, $0, 0xA0
jump_link(rtmp_0, rret_addr), nop, // jalr $t0, $ra, BD slot
load_word(rret_addr, rstack_ptr, S_(U4)), // lw $ra, 4($sp)
jump_reg(rret_addr), // jr $ra
add_ui(rstack_ptr, rstack_ptr, MipsStackAlignment), // sp += 8 (BD)
mac_yield(),
};
+49 -58
View File
@@ -136,31 +136,31 @@ enum {
/* Semantic Aliases for MIPS Registers (O32 ABI) */
// , rdiscard = R_0 /* Hardwired to 0 */
// , rasm_tmp = R_AT /* Assembler temporary (destroyed by some assembler pseudoinstructions!) */
// , rret_0 = R_V0 /* Function return value */
// , rret_1 = R_V1 /* Second return value (e.g., 64-bit) */
// , rarg_0 = R_A0 /* First function argument */
// , rarg_1 = R_A1 /* Second function argument */
// , rarg_2 = R_A2 /* Third function argument */
// , rarg_3 = R_A3 /* Fourth function argument */
// , rtmp_0 = R_T0 /* Temporary (Caller saved) */
// , rtmp_1 = R_T1 /* Temporary (Caller saved) */
// , rtmp_2 = R_T2 /* Temporary (Caller saved) */
// , rtmp_3 = R_T3 /* Temporary (Caller saved) */
// , rtmp_4 = R_T4 /* Temporary (Caller saved) — common GTE base pointer */
// , rtmp_9 = R_T9 /* Temporary (Caller saved) — common GTE base pointer */
// , rstatic_0 = R_S0 /* Static (Callee saved, preserved across calls) */
// , rstatic_1 = R_S1
// , rstatic_2 = R_S2
// , rstatic_3 = R_S3
// , rstatic_4 = R_S4
// , rstatic_5 = R_S5
// , rstatic_6 = R_S6
// , rstatic_7 = R_S7
// , rsaved_0 = R_S0 /* Alias for rstatic_0 (alternate vocabulary) */
// , rstack_ptr = R_SP /* Stack Pointer */
// , rret_addr = R_RA /* Return Address (populated by JAL) */
, rdiscard = R_0 /* Hardwired to 0 */
, rasm_tmp = R_AT /* Assembler temporary (destroyed by some assembler pseudoinstructions!) */
, rret_0 = R_V0 /* Function return value */
, rret_1 = R_V1 /* Second return value (e.g., 64-bit) */
, rarg_0 = R_A0 /* First function argument */
, rarg_1 = R_A1 /* Second function argument */
, rarg_2 = R_A2 /* Third function argument */
, rarg_3 = R_A3 /* Fourth function argument */
, rtmp_0 = R_T0 /* Temporary (Caller saved) */
, rtmp_1 = R_T1 /* Temporary (Caller saved) */
, rtmp_2 = R_T2 /* Temporary (Caller saved) */
, rtmp_3 = R_T3 /* Temporary (Caller saved) */
, rtmp_4 = R_T4 /* Temporary (Caller saved) — common GTE base pointer */
, rtmp_9 = R_T9 /* Temporary (Caller saved) — common GTE base pointer */
, rstatic_0 = R_S0 /* Static (Callee saved, preserved across calls) */
, rstatic_1 = R_S1
, rstatic_2 = R_S2
, rstatic_3 = R_S3
, rstatic_4 = R_S4
, rstatic_5 = R_S5
, rstatic_6 = R_S6
, rstatic_7 = R_S7
, rsaved_0 = R_S0 /* Alias for rstatic_0 (alternate vocabulary) */
, rstack_ptr = R_SP /* Stack Pointer */
, rret_addr = R_RA /* Return Address (populated by JAL) */
/* --- MIPS CPU Opcodes (Bits 31-26) --- */
@@ -259,22 +259,22 @@ enum { _BitOffsets = 0
, SHAMT_SHIFT = 6 /* Shift Amount */
, FC_SHIFT = 0
/* IMM_MASK is the 16-bit two's-complement truncation for the immediate field.
* It is NOT a range guard — it is load-bearing for negative branch offsets
* (the metaprogram emits raw signed offsets; the mask truncates them to the
* 16-bit representation the hardware expects). The static analysis
* `immediate_field_width` check validates ranges at build time. */
/* Bit Masks to prevent overflow into adjacent fields */
, OPCODE_MASK = 0x3F
, REG_MASK = 0x1F
, SHAMT_MASK = 0x1F /* Shift Amount */
, FC_MASK = 0x3F
, IMM_MASK = 0xFFFF
};
#define enc_op(op) ((op) << OPCODE_SHIFT)
#define enc_rs(rs) ((rs) << RS_SHIFT)
#define enc_rt(rt) ((rt) << RT_SHIFT)
#define enc_rd(rd) ((rd) << RD_SHIFT)
#define enc_shamt(shamt) ((shamt) << SHAMT_SHIFT)
#define enc_fc(fc) ((fc) << FC_SHIFT)
#define enc_imm(imm) ((imm) & IMM_MASK)
#define enc_op(op) (((op) & OPCODE_MASK) << OPCODE_SHIFT)
#define enc_rs(rs) (((rs) & REG_MASK) << RS_SHIFT)
#define enc_rt(rt) (((rt) & REG_MASK) << RT_SHIFT)
#define enc_rd(rd) (((rd) & REG_MASK) << RD_SHIFT)
#define enc_shamt(shamt) (((shamt) & SHAMT_MASK) << SHAMT_SHIFT)
#define enc_fc(fc) (((fc) & FC_MASK) << FC_SHIFT)
#define enc_imm(imm) (((imm) & IMM_MASK))
/* MIPS R-Type Instruction Format (Register-to-Register) */
#define enc_r(op, rs, rt, rd, shamt, fc) (enc_op(op) | enc_rs(rs) | enc_rt(rt) | enc_rd(rd) | enc_shamt(shamt) | enc_fc(fc))
@@ -318,10 +318,7 @@ enum { _BitOffsets = 0
#define load_half(rt, base, off) enc_i(op_lh, (base), (rt), (off))
#define load_byte_u(rt, base, off) enc_i(op_lbu, (base), (rt), (off))
#define load_half_u(rt, base, off) enc_i(op_lhu, (base), (rt), (off))
#define LdSlot_
#define store_word(rt, base, off) enc_i(op_sw, (base), (rt), (off))
#define add_ui(rt, rs, imm) enc_i(op_addiu, (rs), (rt), (imm))
#define and_i(rt, rs, imm) enc_i(op_andi, (rs), (rt), (imm))
// #define and_si and_i
@@ -351,12 +348,6 @@ enum { _BitOffsets = 0
#define shift_lright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_srl)
#define shift_aright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_sra)
/* Shift Variable — register-shift forms.
* shift_lleft_var(rd, rt, rs) → sllv rd, rt, rs (shamt in low 5 bits of rs)
* shift_aright_var(rd, rt, rs) → srav rd, rt, rs */
#define shift_lleft_var(rd, rt, rs) enc_r(op_special, (rs), (rt), (rd), 0, fc_sllv)
#define shift_aright_var(rd, rt, rs) enc_r(op_special, (rs), (rt), (rd), 0, fc_srav)
#define shift_lleft_self(rd_rt, shamt) enc_r(op_special, R_0, (rd_rt), (rd_rt), (shamt), fc_sll)
#define mask_upper(rd, rt, shamt) shift_lleft(rd, rt, shamt), shift_lright(rd, rt, shamt)
@@ -375,21 +366,20 @@ enum { _BitOffsets = 0
* WARNING: `jump(off)` CANNOT BE USED for within-atom jumps in the current pipeline.
* The MIPS j opcode encodes `(target_addr >> 2)` in its 26-bit immediate field; an ABSOLUTE byte address, not a relative word offset.
* The metaprogram computes `off` as a relative word offset (`target_word_idx - branch_word_idx - 1`), which the assembler/linker does NOT resolve.
*
* `jump(off)` is only safe when the BUILD PIPELINE owns the absolute position of the emitted code — i.e. when: s
* - the build emits a symbol-relative `.word` expression that the linker resolvess via `R_MIPS_26`, OR
* - the code is hand-assembled with explicit absolute targets, OR a custom post-build patcher resolves the 26-bit field.
* TODO(Ed): Review this.. technically we can resolve aboslute jumps on baked atoms? (Even proedurally generated ones...)
*/
#define jump(off) enc_i(op_j, R_0, R_0, (off))
// Annotate an instruction as filling a branch-delay slot.
#define BdSlot_
/* jump_rel off — unconditional relative jump (the within-atom-safe `jump`).
* MIPS I R3000A has no "branch always" opcode. The idiom for an unconditional relative jump is `beq $0, $0, off`. */
* MIPS I R3000A has no "branch always" opcode. The idiom for an unconditional relative jump is `beq $0, $0, off`.
*/
#define jump_rel(off) branch_equal(R_0, R_0, (off))
/* call_addr off — jump-and-link to immediate address.
*
* Same WARNING as `jump(off)` above: the jal opcode also encodes an absolute 26-bit target.
* For within-atom calls, the current pipeline has no equivalent always-taken call-and-link idiom.
* Workaround: `branch_link` (always-taken branch + explicit `la $ra, next_word_addr; jr $ra`), or just use `call_reg($tmp)` after loading the target into a register.
@@ -407,7 +397,13 @@ enum { _BitOffsets = 0
* sub_s / sub_u → sub / subu
* mult_s / mult_u → mult / multu (writes HI/LO; result in LO)
* div_s / div_u → div / divu (LO = quot, HI = rem)
*/
*
* NOTE: dsl.h defines `add_s`/`sub_s`/`mut_s`/`gt_s`/etc. as _Generic-based signed integer-arithmetic helpers for U1/U2/U4.
* Those live in a different conceptual layer (generic arithmetic on DSL types) and would collide with the instruction encoders here.
* The `#undef` below lets the gas-style names below win; if a file needs both, the dsl.h versions can be reached via their long forms
* (e.g. `def_signed_op`-style or the underlying `add_s1/s2/s4`). */
#undef add_s
#undef sub_s
#define add_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_add)
#define add_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_addu)
#define sub_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_sub)
@@ -417,7 +413,6 @@ enum { _BitOffsets = 0
#define div_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_div)
#define div_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_divu)
// TODO(Ed): Change convention of 'self' to ds for (destination is source)?
#define add_u_self(rd_rs, rt) add_u(rd_rs, rd_rs, rt)
/* --- Arithmetic I-type (immediate) --- */
@@ -460,13 +455,9 @@ enum { _BitOffsets = 0
#define shift_amount(rd, rt, n) shift_lleft(rd, rt, n)
/* nop — sll $0, $0, 0 */
#define nop shift_lleft(R_0, R_0, 0)
#define nop shift_lleft(rdiscard, rdiscard, 0)
#define nop2 nop, nop
// li_s — load signed 16-bit immediate into GPR (addiu rt, $0, imm — sign-extends).
#define li_s(rt, imm) add_ui((rt), R_0, (imm))
// #define load_imm_s(rt, imm) add_ui((rt), R_0, (imm))
#define load_imm_1w(rt, imm) add_ui((rt), R_0, (imm))
#define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm))
+79 -92
View File
@@ -9,35 +9,6 @@
ATOM_FILE_DEBUGGER_LINE_MARKER(pad_atom_c);
#pragma region MACs (Mips Atom Components)
FI_ Slice_MipsCode ac_pad_set_centered_axes(AtomBuilder_R ab, Reg state, Reg scratch) atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_upper_i(scratch, (PadAxis_Centered >> 16) & 0xFFFF),
or_i_self( scratch, PadAxis_Centered & 0xFFFF),
// mac_load_word_imm(scratch, PadAxis_Centered),
store_word( scratch, state, O_(PadState,axes)),
})
FI_ Slice_MipsCode ac_pad_set_id_byte(AtomBuilder_R ab, Reg state, Reg r_id, U1 id_value) atom_dbg_skip MipsAtomComp_Proc_(ab, {
add_ui( r_id, R_0, id_value),
store_byte(r_id, state, O_(PadState,id)),
})
FI_ Slice_MipsCode ac_pad_set_status(AtomBuilder_R ab, U4 r_tmp, U1 r_state, U4 pad_status) atom_dbg_skip MipsAtomComp_Proc_(ab, {
add_ui( r_tmp, R_0, pad_status),
store_word(r_tmp, r_state, O_(PadState,status)),
})
/* Invert r_buttons (active-low → active-high) and store to PadState.buttons.
* r_buttons must already be loaded (the caller is responsible for filling the load-delay slot of
* the preceding load_half_u with an instruction that doesn't read r_buttons). */
FI_ Slice_MipsCode ac_pad_store_inverted_buttons(AtomBuilder_R ab, U1 r_buttons, U1 r_pad_state) atom_dbg_skip MipsAtomComp_Proc_(ab, {
nor_u( r_buttons, r_buttons, R_0),
store_half( r_buttons, r_pad_state, O_(PadState,buttons)),
})
#pragma endregion MACs (Mips Atom Components)
#pragma region Baked Atoms
/* ----- pad_bios_snapshot -----
@@ -55,16 +26,16 @@ FI_ Slice_MipsCode ac_pad_store_inverted_buttons(AtomBuilder_R ab, U1 r_buttons,
* byte_swap16(x) = (x >> 8) | (x << 8); nor(x, R_0) = ~x. store_half truncates to 16 bits so the upper-16 mask is implicit in the store.
*
* Register use (atom-local; no wave-context touched):
* R_T0 = raw base : Kept throughout; axes loads read raw[4..7] from R_T0.
* R_T1 = state base : Kept throughout; all stores go through R_T1.
* R_T2 = raw[0] status : Alive across the disc/pending/id dispatch, then dead.
* R_T3 = raw[1] id : Alive across the id dispatch, then dead.
* R_T4 = scratch : Shifts, compares, immediate loads, store values.
* R_T5 = scratch : Parallel lui + ori for the 0x80808080 axes constant + byte-swap target.
* R_T0 = raw base (kept throughout; axes loads read raw[4..7] from R_T0)
* R_T1 = state base (kept throughout; all stores go through R_T1)
* R_T2 = raw[0] status (alive across the disc/pending/id dispatch, then dead)
* R_T3 = raw[1] id (alive across the id dispatch, then dead)
* R_T4 = scratch (shifts, compares, immediate loads, store values)
* R_T5 = scratch (parallel lui+ori for the 0x80808080 axes constant + byte-swap target)
*/
enum {
R_PadRaw = R_T0 atom_reg atom_type(U1),
R_PadState = R_T1 atom_reg atom_type(PadState*),
R_PadState = R_T1 atom_reg,
R_RawStatus = R_T2 atom_reg,
R_RawId = R_T3 atom_reg,
};
@@ -73,8 +44,8 @@ typedef Struct_(Binds_PadBiosSnapshot) {
PadState* state;
};
internal MipsAtom_(pad_bios_snapshot) atom_info(atom_bind(Binds_PadBiosSnapshot)
, atom_reads( R_PadRaw, R_PadState, R_RawStatus, R_RawId)
, atom_writes(R_PadRaw, R_PadState, R_RawStatus, R_RawId)
, atom_reads( R_PadRaw, R_PadState, R_RawStatus, R_RawId, R_T4, R_T5, R_TapePtr)
, atom_writes(R_PadRaw, R_PadState, R_RawStatus, R_RawId, R_T4, R_T5, R_TapePtr)
) {
/* === Bind consumption: T0 = raw, T1 = state, advance R_TapePtr by 8. */
load_word(R_PadRaw, R_TapePtr, O_(Binds_PadBiosSnapshot,raw)),
@@ -82,98 +53,111 @@ internal MipsAtom_(pad_bios_snapshot) atom_info(atom_bind(Binds_PadBiosSnapshot)
add_ui_self( R_TapePtr, S_(Binds_PadBiosSnapshot)),
/* === Read raw[0] (status) + raw[1] (id) */
load_byte_u(R_RawStatus, R_PadRaw, O_(PadBiosRaw,status)),
load_byte_u(R_RawId, R_PadRaw, O_(PadBiosRaw,id)),
load_byte_u(R_RawStatus, R_PadRaw, 0),
load_byte_u(R_RawId, R_PadRaw, 1),
atom_label(snap_root) /* === Case 1: Disconnected (status == 0xFF). */
add_ui(R_T4, R_0, PadRawStatus_Timeout), branch_ne(R_RawStatus, R_T4, atom_offset(snap_root, skip_disconnected)),
add_ui(R_T4, R_0, 0xFF), branch_ne(R_RawStatus, R_T4, atom_offset(snap_root, skip_disconnected)),
/* BD-slot: pre-compute PadStatus_Disconnected. Branch reads R_T4=0xFF in EX before this WB completes.
* If branch NOT taken (fall through to pending/id_dispatch), R_T4 is overwritten by the next case body's add_ui — harmless. */
atom_label(disconnected) /* === Disconnected body. */
mac_pad_set_status(R_T4, R_PadState, PadStatus_Disconnected),
store_half( R_0, R_PadState, O_(PadState,buttons)),
mac_pad_set_centered_axes(R_PadState, R_T4),
mac_pad_set_id_byte(R_PadState, R_RawId, PadRawStatus_Timeout),
/* R_T4 = PadStatus_Disconnected from snap_root BD-slot. */
store_word(R_T4, R_PadState, O_(PadState,status)),
store_half(R_0, R_PadState, O_(PadState,buttons)),
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
store_word( R_T4, R_PadState, O_(PadState,left_x)),
store_byte( R_RawId, R_PadState, O_(PadState,id)),
jump_rel(atom_offset(disconnected, snap_end)),
/* BD-slot: load next atom's entry point (replaces the nop).
* Always jumps to snap_end, where mac_yield_tail() transfers control to R_AtomJmp without re-loading it. */
* The unconditional branch always jumps to snap_end, where mac_yield_tail()
* transfers control to R_AtomJmp without re-loading it. */
mac_yield_load(),
atom_label(skip_disconnected)
/* === Case 2: Pending (status == 0 && id == 0)
* Combined check: if (status | id) != 0 then skip to id_dispatch. Falls through to the Pending case only when both are zero. */
* Combined check: if (status | id) != 0 then skip to id_dispatch.
* Falls through to the Pending case only when both are zero. */
or_u_self(R_RawStatus, R_RawId), branch_ne(R_RawStatus, R_0, atom_offset(case_2, id_dispatch)),
/* BD-slot: pre-compute PadStatus_Pending. Branch reads R_RawStatus in EX before this WB completes.
* If branch NOT taken (fall through to id_dispatch), R_T4 is overwritten by the digital/analog body add_ui - harmless. */
* If branch NOT taken (fall through to id_dispatch), R_T4 is overwritten by the digital/analog body add_ui harmless. */
atom_label(pending) /* === Pending body (status=0, id=0 — pre-IRQ-empty buffer). */
mac_pad_set_status(R_T4, R_PadState, PadStatus_Pending),
store_half( R_0, R_PadState, O_(PadState,buttons)),
mac_pad_set_centered_axes(R_PadState, R_T4),
store_byte(R_RawId, R_PadState, O_(PadState,id)),
atom_label(pending) /* === Pending body */
/* R_T4 = PadStatus_Pending from case_2 BD-slot. */
store_word(R_T4, R_PadState, O_(PadState,status)),
store_half(R_0, R_PadState, O_(PadState,buttons)),
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
store_word( R_T4, R_PadState, O_(PadState,left_x)),
store_byte( R_RawId, R_PadState, O_(PadState,id)),
jump_rel(atom_offset(pending, snap_end)),
mac_yield_load(),
atom_label(id_dispatch) /* === Case 3-6: ID dispatch */
add_ui(R_T4, R_0, PadRawId_Digital), branch_ne(R_RawId, R_T4, atom_offset(id_dispatch, try_analog_stick)),
add_ui(R_T4, R_0, 0x41), branch_ne(R_RawId, R_T4, atom_offset(id_dispatch, try_analog_stick)),
/* BD-slot: pre-compute PadStatus_Digital. Branch reads R_RawId in EX before this WB completes.
* If branch NOT taken (fall through to try_analog_stick), R_T4 is overwritten by the analog body add_ui. */
/* === Digital body (status, buttons normalize, axes=0x80, id, branch.
* R_T5 holds the 0x80808080 axes constant (loaded into the load-delay slot of the buttons-load).
* R_T5 is then "dead" — only consumed at the analog_pad range check downstream. */
mac_pad_set_status(R_T4, R_PadState, PadStatus_Digital),
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw, buttons)), /* R_T4 = raw_buttons; */
mac_load_word_imm(R_T5, PadAxis_Centered), /* fills the buttons-load's delay slot (doesn't read R_T4) */
// load_upper_i(R_T5, PadAxis_Centered_Hi), or_i_self(R_T5, PadAxis_Centered_Lo),
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
store_word(R_T5, R_PadState, O_(PadState, axes)), /* single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y) */
mac_pad_set_id_byte(R_PadState, R_T4, PadRawId_Digital),
/* === Digital body (status, buttons normalize, axes=0x80, id, branch. */
/* R_T4 = PadStatus_Digital from id_dispatch BD-slot. */
store_word( R_T4, R_PadState, O_(PadState,status)),
load_half_u(R_T4, R_PadRaw, 2 * S_(U1)),
/* Fill R_T4's load-delay slot with the 0x80808080 axes constant into R_T5
* (R_T5 is dead on this path; it's only consumed at the analog_pad range check). */
load_upper_i(R_T5, 0x8080), or_i_self(R_T5, 0x8080),
nor_u( R_T4, R_T4, R_0), /* raw_buttons is already in host bit order; no swap needed */
store_half( R_T4, R_PadState, O_(PadState,buttons)),
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
store_word( R_T5, R_PadState, O_(PadState,left_x)),
add_ui( R_T4, R_0, 0x41),
store_byte( R_T4, R_PadState, O_(PadState,id)),
jump_rel(atom_offset(id_dispatch, snap_end)),
mac_yield_load(),
atom_label(try_analog_stick) /* === Case 4: AnalogStick (id == 0x53)*/
add_ui(R_T4, R_0, PadRawId_AnalogStick), branch_ne(R_RawId, R_T4, atom_offset(try_analog_stick, try_analog_pad)),
add_ui(R_T4, R_0, 0x53), branch_ne(R_RawId, R_T4, atom_offset(try_analog_stick, try_analog_pad)),
/* BD-slot: pre-compute PadStatus_AnalogStick. Branch reads R_RawId in EX before this WB completes.
* If branch NOT taken (fall through to try_analog_pad), R_T4 is overwritten by the analog_pad body add_ui. */
atom_label(analog_stick) /* === AnalogStick body
* R_T5 holds left_xy (loaded into the load-delay slot of the buttons-load via the left-axis load_half_u).
* R_T4 holds right_xy (loaded into the load-delay slot of the left-load).
* R_T5 is then "dead" — reused for the id-byte value load in mac_pad_write_id_byte.
* The buttons invert+store happens BEFORE R_T4 is overwritten by the right_xy load. */
mac_pad_set_status(R_T4, R_PadState, PadStatus_AnalogStick),
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,buttons)), /* R_T4 = raw_buttons; delay slot at the next instruction */
load_half_u( R_T5, R_PadRaw, O_(PadBiosRaw,left)), /* fills the buttons-load's delay slot (doesn't read R_T4) */
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,right)), /* fills R_T5's load-delay slot (doesn't read R_T5); overwrites R_T4 (was buttons) with right_xy */
store_half( R_T5, R_PadState, O_(PadState, left)),
store_half( R_T4, R_PadState, O_(PadState, right)),
mac_pad_set_id_byte(R_PadState, R_T5, PadRawId_AnalogStick),
* Axes are loaded as two halfwords: raw[6..7] → left_xy (sh at offset 8), raw[4..5] → right_xy (sh at offset 10).
* R_T5 holds left_xy / id-value in turn (it's dead on this path — only consumed at the analog_pad range check). */
/* R_T4 = PadStatus_AnalogStick from try_analog_stick BD-slot. */
store_word( R_T4, R_PadState, O_(PadState,status)),
load_half_u( R_T4, R_PadRaw, 2 * S_(U1)), /* R_T4 = raw_buttons */
load_half_u( R_T5, R_PadRaw, 6 * S_(U1)), /* R_T5 = left_xy; fills R_T4's load-delay slot (doesn't read R_T4) */
nor_u( R_T4, R_T4, R_0), /* R_T4 = ~raw_buttons */
store_half( R_T4, R_PadState, O_(PadState,buttons)),
load_half_u( R_T4, R_PadRaw, 4 * S_(U1)), /* R_T4 = right_xy; fills R_T5's load-delay slot */
store_half( R_T5, R_PadState, O_(PadState,left_x)), /* R_T5 settled, store left_xy */
store_half( R_T4, R_PadState, O_(PadState,right_x)),
add_ui( R_T5, R_0, 0x53), /* R_T5 = id value (clobbers left_xy, already stored) */
store_byte( R_T5, R_PadState, O_(PadState,id)),
jump_rel(atom_offset(analog_stick, snap_end)),
mac_yield_load(),
atom_label(try_analog_pad) /* === Case 5-6: AnalogPad (id & 0xF0 == 0x70) */
and_i( R_T4, R_RawId, PadRawId_AnalogPadMask),
add_ui( R_T5, R_0, PadRawId_AnalogPadValue),
and_i( R_T4, R_RawId, 0xF0),
add_ui( R_T5, R_0, 0x70),
branch_ne(R_T4, R_T5, atom_offset(try_analog_pad, try_unsupported)),
/* BD-slot: pre-compute PadStatus_AnalogPad. Branch reads R_T4 in EX before this WB completes.
* If branch NOT taken (fall through to try_unsupported), R_T4 is overwritten by the unsupported body add_ui. */
atom_label(analog_pad) /* === AnalogPad body
* Same shape as AnalogStick with AnalogPad status. R_T5 holds left_xy (it's dead on this path).
* The id byte is raw id from the BIOS buffer (R_RawId already holds raw[1]).
* Buttons invert + store happens before R_T4 is overwritten by the right_xy load. */
mac_pad_set_status(R_T4, R_PadState, PadStatus_AnalogPad),
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,buttons)), /* R_T4 = raw_buttons; delay slot at the next instruction */
load_half_u( R_T5, R_PadRaw, O_(PadBiosRaw,left)), /* fills the buttons-load's delay slot (doesn't read R_T4) */
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
load_half_u(R_T4, R_PadRaw, O_(PadBiosRaw,right)), /* fills R_T5's load-delay slot (doesn't read R_T5); overwrites R_T4 with right_xy */
store_half( R_T5, R_PadState, O_(PadState, left)),
store_half( R_T4, R_PadState, O_(PadState, right)),
store_byte( R_RawId, R_PadState, O_(PadState, id)),
* Same shape as AnalogStick with AnalogPad status. R_T5 holds left_xy (it's dead on this path). */
/* R_T4 = PadStatus_AnalogPad from try_analog_pad BD-slot. */
store_word( R_T4, R_PadState, O_(PadState,status)),
load_half_u(R_T4, R_PadRaw, 2 * S_(U1)), /* R_T4 = raw_buttons */
load_half_u(R_T5, R_PadRaw, 6 * S_(U1)), /* R_T5 = left_xy; fills R_T4's load-delay slot */
nor_u( R_T4, R_T4, R_0), /* R_T4 = ~raw_buttons */
store_half( R_T4, R_PadState, O_(PadState,buttons)),
load_half_u(R_T4, R_PadRaw, 4 * S_(U1)), /* R_T4 = right_xy; fills R_T5's load-delay slot */
store_half( R_T5, R_PadState, O_(PadState,left_x)), /* R_T5 settled, store left_xy */
store_half( R_T4, R_PadState, O_(PadState,right_x)),
store_byte( R_RawId, R_PadState, O_(PadState,id)),
jump_rel(atom_offset(analog_pad, snap_end)),
mac_yield_load(),
@@ -182,8 +166,11 @@ atom_label(try_unsupported) /* === Case 7: Unsupported — fall through from the
add_ui( R_T4, R_0, PadStatus_Unsupported),
store_word(R_T4, R_PadState, O_(PadState,status)),
store_half(R_0, R_PadState, O_(PadState,buttons)),
mac_pad_set_centered_axes(R_PadState, R_T4),
mac_pad_set_id_byte(R_PadState, R_RawId, PadUnknownId_Sentinel),
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
store_word( R_T4, R_PadState, O_(PadState,left_x)),
add_ui( R_T4, R_0, 0xFF), /* 0xFF sentinel: "unknown id" */
store_byte( R_T4, R_PadState, O_(PadState,id)),
/* Fall through to snap_end. */
atom_label(no_jump_fallthrough)
-78
View File
@@ -1,78 +0,0 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "dsl.h"
# include "gcc_asm.h"
# include "mips.h"
# include "bios.h"
# include "pad.h"
#endif
/* Uses ONE 8-byte frame allocated via the compiler's standard prologue.
* 4 wasted-arg words for B(12h) InitPAD2 are at [SP+0..15] but are not explicitly allocated.
* Compiler handles the MIPS O32 "wasted stack" convention for us by treating the B-call as a 4-arg call.
*
* The buffer pointers are passed as arguments so the compiler keeps them in callee-saved registers;
* The B(12h) asm volatile block does NOT clobber those registers (it clobbers only the volatile GPRs + B-table arg registers explicitly).
* The C-level writes after the call re-load the pointers from their callee-saved homes.
*
* The clobber list for both B-calls names the full BIOS destroy set documented in kernelbios.md:167-174 (R1..R15, R24..R25, R31, HI/LO).
* The kernel-ABI "volatile GPRs" subset is clb_mem_drain; the rest of the destroy set is enumerated explicitly here. */
NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
{
/* Pin raw0 + raw1 to $a0 + $a1 via rgcc; the B(12h) call uses these directly.
* The `(void)` casts mark them as unread after the call so the compiler doesn't need to move them back. */
register PadBiosRaw* p0 rgcc(R_A0) = raw0;
register PadBiosRaw* p1 rgcc(R_A1) = raw1;
(void)p0; (void)p1;
// TODO(Ed): Properly annotate the raw values in the inline asm instructions.
// Use enums.
/* B(12h) InitPAD2(raw0, 0x22, raw1, 0x22)
* $a0 = raw0 (rgcc-bound; survives the sequence below)
* $a1 = raw1 (preserved into $a2 before $a1 is overwritten)
* $a2 = raw1 (moved from $a1; survives $a1's overwrite)
* $a3 = 0x22 (immediate)
* $t1 = 0x12 (function number)
* $t2 = 0xB0 (BIOS B-table address) */
asm volatile(
asm_words(
or_u( R_A2, R_A1, R_0), /* $a2 = $a1 = raw1 */
add_ui( R_A1, R_0, bios_pad_buffer_size), /* $a1 = 0x22 */
add_ui( R_A3, R_0, bios_pad_buffer_size), /* $a3 = 0x22 */
add_ui( R_T1, R_0, bios_init_pad_2), /* $t1 = 0x12 */
add_ui( R_T2, R_0, bios_btable_addr), /* $t2 = 0xB0 */
call_reg(R_T2), /* jalr $t2, $ra */
nop /* BD slot */
)
asm_rpins, r_use(p0), r_use(p1)
asm_clobber:
rlit(R_AT),
rlit(R_V0), rlit(R_V1),
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
rlit(R_RA),
clb_mem_drain
);
/* The C-level writes re-load the pointers via the parameter names and write 0xFF to each
* buffer's status byte to mark the initial-state hazard documented in kernelbios.md:1621-1624. */
u1_v(raw0)[0] = 0xFF;
u1_v(raw1)[0] = 0xFF;
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */
asm volatile(
asm_words(
add_ui( R_T1, R_0, bios_start_pad_2), /* $t1 = 0x13 */
add_ui( R_T2, R_0, bios_btable_addr), /* $t2 = 0xB0 (re-load) */
call_reg(R_T2), /* jalr $t2, $ra */
nop /* BD slot */
)
asm_clobber:
rlit(R_AT),
rlit(R_V0), rlit(R_V1),
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
rlit(R_RA),
clb_mem_drain
);
}
+36 -78
View File
@@ -1,30 +1,28 @@
#ifdef INTELLISENSE_DIRECTIVES
# pragma once
# include "dsl.h"
# include "math.h"
#endif
/* PSX button bit positions — 1:1 with PSX-SPX docs at docs/psx-spx/docs/controllersandmemorycards.md:405-421.
* Wire is active-low (0 = pressed).
* The decoder atom computes buttons = (~raw_buttons) & 0xFFFF;
* active-low-to-active-high inversion is applied bit-by-bit. */
typedef Enum_(U2, PadBtns) {
Bit_(Pad_Select, 0),
Bit_(Pad_L3, 1),
Bit_(Pad_R3, 2),
Bit_(Pad_Start, 3),
Bit_(Pad_Up, 4),
Bit_(Pad_Right, 5),
Bit_(Pad_Down, 6),
Bit_(Pad_Left, 7),
Bit_(Pad_L2, 8),
Bit_(Pad_R2, 9),
Bit_(Pad_L1, 10),
Bit_(Pad_R1, 11),
* The decoder atom computes buttons = (~raw_buttons) & 0xFFFF; the active-low-to-active-high inversion is applied bit-by-bit. */
enum {
Bit_(Pad_Select, 0),
Bit_(Pad_L3, 1),
Bit_(Pad_R3, 2),
Bit_(Pad_Start, 3),
Bit_(Pad_Up, 4),
Bit_(Pad_Right, 5),
Bit_(Pad_Down, 6),
Bit_(Pad_Left, 7),
Bit_(Pad_L2, 8),
Bit_(Pad_R2, 9),
Bit_(Pad_L1, 10),
Bit_(Pad_R1, 11),
Bit_(Pad_Triangle, 12),
Bit_(Pad_Circle, 13),
Bit_(Pad_Cross, 14),
Bit_(Pad_Square, 15),
Bit_(Pad_Circle, 13),
Bit_(Pad_Cross, 14),
Bit_(Pad_Square, 15),
};
enum {
@@ -34,22 +32,18 @@ enum {
Pad1 = 1 << PadId_Offset,
};
/* =============================================================================
#define pad0_(btn_id) (btn_id << Pad0)
#define pad1_(btn_id) (btn_id << Pad1)
/* ============================================================
* BIOS pad-buffer subsystem: docs/psx-spx/docs/kernelbios.md (B(12h) + B(13h))
* ============================================================================= */
* ============================================================ */
enum {
PAD_BIOS_RAW_SIZE = 0x22,
};
// BIOS pad buffer layout (docs/psx-spx/docs/kernelbios.md (InitPAD2 returns 0x22 = 34 bytes per port)).
// Bytes 0..7 are the named snapshot region; bytes 8..33 are reserved (the BIOS writes the buffer raw; we only read bytes 0..7 via O_(PadBiosRaw, ...)).
typedef Struct_(PadBiosRaw) {
U1 status; /* offset 0 (PadRawStatus_Ok / PadRawStatus_Timeout) */
U1 id; /* offset 1 (PadRawId_Digital / PadRawId_AnalogStick / 0x7x AnalogPad) */
U2 buttons; /* offset 2-3 (active-low 16-bit button map) */
V2_U1 right; /* offset 4-5 (right stick x, y) */
V2_U1 left; /* offset 6-7 (left stick x, y) */
U1 reserved[PAD_BIOS_RAW_SIZE - 8]; /* offset 8..33 */
U1 bytes[PAD_BIOS_RAW_SIZE];
};
typedef Enum_(U4, PadStatus) {
@@ -62,54 +56,18 @@ typedef Enum_(U4, PadStatus) {
PadStatus_Invalid,
};
/* Distinct from the game-facing PadStatus enum: PadRawStatus_Ok and PadRawStatus_Timeout are raw BIOS values;
* PadStatus_* are game-facing post-decode states. PadUnknownId_Sentinel is written by the decoder
* when the controller id does not match any known controller type.
* PadAxisCentered_Word: Four-byte 0x80 pattern used to clear / center
* four byte axes at PadState.left_x through PadState.right_y. */
typedef Enum_(U1, PadRawStatus) {
PadRawStatus_Ok = 0x00,
PadRawStatus_Timeout = 0xFF,
};
typedef Enum_(U1, PadRawId) {
PadRawId_Digital = 0x41,
PadRawId_AnalogStick = 0x53,
PadRawId_AnalogPadMask = 0xF0,
PadRawId_AnalogPadValue = 0x70,
};
typedef Enum_(U1, PadUnknownId) {
PadUnknownId_Sentinel = 0xFF,
};
typedef Enum_(U4, PadAxisCentered) {
PadAxis_Centered_Hi = 0x8080,
PadAxis_Centered_Lo = 0x8080,
PadAxis_Centered = 0x80808080U,
};
typedef Enum_(U1, PadDeadZone) {
PadDeadZone_LowBound = 0x70, /* left_x < LowBound → active; delta = 0x80 - left_x > 0 (rightward pull) */
PadDeadZone_Center = 0x80, /* analog rest position; left_x == Center → delta = 0 (no rotation) */
PadDeadZone_HighBound = 0x90, /* left_x > HighBound → active; delta = 0x80 - left_x < 0 (leftward pull) */
};
typedef Struct_(PadAxes) {
V2_U1 left; /* offset 8-9 */
V2_U1 right; /* offset 10-11 */
};
// Field order is chosen so that the 4 axes (left_x, left_y, right_x, right_y)
// form a contiguous 4-byte block at offset 8, allowing a single `store_word` to clear-or-write all 4 axes in one MIPS instruction.
/* PadState — per-port normalized runtime state.
* Field order is chosen so that the 4 axes (left_x, left_y, right_x, right_y)
* form a contiguous 4-byte block at offset 8, allowing a single `store_word` to clear-or-write all 4 axes in one MIPS instruction.
* The struct size stays 12 bytes (unchanged from the prior order,
* which left the C compiler to insert 1 byte of trailing pad to reach the 4-byte struct alignment). */
typedef Struct_(PadState) {
PadStatus status; /* offset 0, (U4) */
PadBtns buttons; /* offset 4, */
U1 id; /* offset 6, */
byte_pad(1); /* offset 7, explicit pad to align the axes block */
union {
A2_V2_U1 axes; /* offset 8-11 store_target (4-byte aligned)*/
struct {
V2_U1 left; /* offset 8-9 */
V2_U1 right; /* offset 10-11 */
};
};
PadStatus status; /* offset 0, size 4 (U4) */
U2 buttons; /* offset 4, size 2 */
U1 id; /* offset 6, size 1 */
U1 pad; /* offset 7, size 1 — explicit pad to align the axes block */
U1 left_x; /* offset 8, size 1 — store_word target (4-byte aligned) */
U1 left_y; /* offset 9, size 1 */
U1 right_x; /* offset 10, size 1 */
U1 right_y; /* offset 11, size 1 */
};
internal void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1);
+5 -23
View File
@@ -64,9 +64,9 @@ typedef Struct_(Tile) {
Linear Algebra
*/
MT3_S2S4* mt3s2s4_rotation (V3_S2* vec, MT3_S2S4* mat) asm("RotMatrix");
MT3_S2S4* mt3s2s4_translation(MT3_S2S4* mat, V3_S4* vec) asm("TransMatrix");
MT3_S2S4* mt3s2s4_scale (MT3_S2S4* mat, V3_S4* vec) asm("ScaleMatrix");
M3_S2* m3s2_rotation (V3_S2* vec, M3_S2* mat) asm("RotMatrix");
M3_S2* m3s2_translation(M3_S2* mat, V3_S4* vec) asm("TransMatrix");
M3_S2* m3s2_scale (M3_S2* mat, V3_S4* vec) asm("ScaleMatrix");
// Rotation, Translation, Perspective
@@ -99,23 +99,5 @@ FI_ S4 rtp_avg_nclip_a4_v3s2(
);
}
void gte_matrix_set_rotation (MT3_S2S4* mat) asm("SetRotMatrix");
void gte_matrix_set_translation(MT3_S2S4* mat) asm("SetTransMatrix");
// Einheit, Metrication to unit vector. "Normalization", not Orthogonal "Normal, Normalis". Directionalization.
// RGA(Lengyel): Normalize the bulk of a zero-weight direction. This is not finite-point unitization (which forces w=1).
S4 normalize_v3s4(V3_S4* v0, V3_S4* v1) asm("VectorNormal");
// RGA(Lengyel): Apply the matrix expansion of a rigid transformation.
// Motor antiproduct is equivalent for unitized points; LA form is what GTE consumes.
V3_S4* mul_m3s2_v3s4(MT3_S2S4* m, V3_S4* v, V3_S4* result) asm("ApplyMatrixLV");
// RGA(Lengyel): Store the full translation column. The motor translator would store half this displacement in m.xyz.
MT3_S2S4* trans_m3s2(MT3_S2S4* m, V3_S4* off) asm("TransMatrix");
MT3_S2S4* gte_comp_coord_m3s2(MT3_S2S4* m0, MT3_S2S4* m1, MT3_S2S4* result) asm("CompMatrixLV");
// RGA(Lengyel): Complement(Wedge(a,b)), i.e. the Euclidean 3D complement of the exterior product, stored as a V3_S4.
// The underlying GTE OP is a specialized signed-16-bit D x IR command; the wedge interpretation is a 3D dual of the same 3 scalars.
void cross_v3s4(V3_S4* v0, V3_S4* v1, V3_S4* result) asm("OuterProduct12");
void gte_matrix_set_rotation (M3_S2* mat) asm("SetRotMatrix");
void gte_matrix_set_translation(M3_S2* mat) asm("SetTransMatrix");
-11
View File
@@ -15,8 +15,6 @@
#define WORD_COUNT(name, count) enum { words_##name = (count) };
WORD_COUNT(nop, 1)
WORD_COUNT(atom_label, 0)
WORD_COUNT(atom_offset, 0)
WORD_COUNT(load_upper_i, 1)
WORD_COUNT(jump_reg, 1)
WORD_COUNT(jump_link, 1)
@@ -56,15 +54,6 @@ WORD_COUNT(gte_sw, 1)
WORD_COUNT(gte_cmdw_rtpt, 1)
WORD_COUNT(gte_cmdw_nclip, 1)
WORD_COUNT(gte_avg_sort_z3, 1)
WORD_COUNT(gte_cmdw_sqr, 1)
WORD_COUNT(gte_cmdw_gpf, 1)
WORD_COUNT(shift_lleft_var, 1)
WORD_COUNT(shift_aright_var, 1)
WORD_COUNT(li_s, 1)
WORD_COUNT(and_i, 1)
WORD_COUNT(add_si, 1)
WORD_COUNT(branch_lt_zero, 1)
WORD_COUNT(sub_s, 1)
WORD_COUNT(sub_u, 1)
WORD_COUNT(nop2, 2)
-13
View File
@@ -1,13 +0,0 @@
#ifdef INTELLISENSE_DIRECTIVES
#pragma once
#endif
// Auto-generated by ps1_meta.lua (passes/auto_reg.lua) — DO NOT EDIT
// Directory: C:\projects\Pikuma\ps1\code\hello_camera
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.c
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.h
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.atom.c
// Per-phase register allocations resolved by the lua pass.
// R_<Sym>_Code = <chosen GPR's _Code constant> for every marker in this directory.
#define R_GpTmp_Code R_V0_Code
+1 -19
View File
@@ -8,7 +8,7 @@
#pragma region hello_camera
// --- atom: pad_input_cube_rotation (60 words) ---
// --- atom: pad_apply_input (60 words) ---
#define _atom_offset_dpad_left_exit_dpad_left 6
#define _atom_offset_dpad_right_exit_dpad_right 6
@@ -26,24 +26,6 @@ enum {
atom_offset_end_low_exit_stick = _atom_offset_end_low_exit_stick,
};
// --- atom: pad_input_cam (40 words) ---
#define _atom_offset_left_x_exit_left_x 3
#define _atom_offset_right_x_exit_right_x 3
#define _atom_offset_up_y_exit_up_y 3
#define _atom_offset_down_y_exit_down_y 3
#define _atom_offset_cross_z_exit_cross_z 3
#define _atom_offset_circle_z_exit_circle_z 3
enum {
atom_offset_left_x_exit_left_x = _atom_offset_left_x_exit_left_x,
atom_offset_right_x_exit_right_x = _atom_offset_right_x_exit_right_x,
atom_offset_up_y_exit_up_y = _atom_offset_up_y_exit_up_y,
atom_offset_down_y_exit_down_y = _atom_offset_down_y_exit_down_y,
atom_offset_cross_z_exit_cross_z = _atom_offset_cross_z_exit_cross_z,
atom_offset_circle_z_exit_circle_z = _atom_offset_circle_z_exit_circle_z,
};
// --- atom: cube_g4_face (76 words) ---
#define _atom_offset_cull_cube_g4_face_exit 41
+102 -595
View File
@@ -17,7 +17,6 @@
# include "duffle/psyq.atom.c"
# include "gen/offsets.h"
# include "gen/macs.h"
# include "gen/auto_reg.h"
# include "hello_camera.h"
#endif
@@ -25,8 +24,8 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c);
#pragma region MACs (Mips Atom components)
FI_ Slice_MipsCode ac_put_disp_env(AtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ab, {
FI_ Slice_MipsCode ac_put_disp_env(U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_put_disp_env, {
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
@@ -36,8 +35,8 @@ MipsAtomComp_Proc_(ab, {
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
})
I_ Slice_MipsCode ac_put_draw_env(AtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ab, {
FI_ Slice_MipsCode ac_put_draw_env(U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_put_draw_env, {
/*
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
* References:
@@ -51,18 +50,18 @@ MipsAtomComp_Proc_(ab, {
* (binary; the PutDrawEnv implementation builds the 16-word DR_ENV from the user's DRAWENV struct and emits it via GP0 GPU commands.)
*
* Word indices (libpsyx PutDrawEnv / SetDrawEnv order):
* tag = (length << 24) | addr — 16-word packet (1 tag + 15 code)
* code[0] = DrawMode (dfe=1, dtd=0, tpage=0) — must come first per libpsyx
* code[1] = TextureWindow (tw=(0,0)) — bare-cmd word; GPU uses current state
* code[2] = DrawArea top-left (clip.x=0, clip.y=240)
* code[3] = DrawArea bottom-right (clip.x+w=320, clip.y+h=480)
* code[4] = DrawOffset (ofs=(0,0)) — bare-cmd word
* code[5] = Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit
* code[6] = Initial-bg-color (isbg=1, r=7, g=7, b=7)
* code[7] = DrawMode (isbg=1, tpage=0) — re-asserts DrawMode with isbg
* code[8..10] = padding (NOP) — 3 words to fill the packet
* code[11..12] = TextureWindow bottom-right — defaults to (0,0,0,0)
* code[13..14] = padding (NOP) — completes the 16-word packet
* tag = (length << 24) | addr — 16-word packet (1 tag + 15 code)
* code[0] = DrawMode (dfe=1, dtd=0, tpage=0) — must come first per libpsyx
* code[1] = TextureWindow (tw=(0,0)) — bare-cmd word; GPU uses current state
* code[2] = DrawArea top-left (clip.x=0, clip.y=240)
* code[3] = DrawArea bottom-right (clip.x+w=320, clip.y+h=480)
* code[4] = DrawOffset (ofs=(0,0)) — bare-cmd word
* code[5] = Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit
* code[6] = Initial-bg-color (isbg=1, r=7, g=7, b=7)
* code[7] = DrawMode (isbg=1, tpage=0) — re-asserts DrawMode with isbg
* code[8..10] = padding (NOP) — 3 words to fill the packet
* code[11..12] = TextureWindow bottom-right — defaults to (0,0,0,0)
* code[13..14] = padding (NOP) — completes the 16-word packet
*/
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port), /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */
mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port), /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */
@@ -91,445 +90,6 @@ MipsAtomComp_Proc_(ab, {
#pragma endregion MACs
#pragma region Atom Procs
// Modular Atoms
enum {
// TODO(Ed): We can resolve scratch at anytime its fixed to a specific address.
R_ResolveScratch = R_T4 atom_reg atom_type(U4*),
#define R_ResolveScratch_Code R_T4_Code
};
typedef Struct_(Binds_ResolveLookAt) {
MT3_S2S4* look_at;
P3_S4* eye;
P3_S4* target;
V3_S4* up_in;
};
/* ─── ResolveLookAtScratch — offset schema for the resolve_look_at bundle's */
typedef Struct_(ResolveLookAtScratch) {
V3_S4 fwd; /* offset +0 (16 bytes — 4 S4 fields incl. internal pad) */
V3_S4 uz; /* offset +16 (16 bytes) */
V3_S4 right; /* offset +32 (16 bytes) */
V3_S4 ux; /* offset +48 (16 bytes) */
V3_S4 up; /* offset +64 (16 bytes) */
V3_S4 uy; /* offset +80 (16 bytes) */
P3_S4 eye; /* offset +96 (16 bytes; storage alias of V3_S4) */
P3_S4 target; /* offset +112 (16 bytes; storage alias of V3_S4) */
V3_S4 up_in; /* offset +128 (16 bytes) */
};
/* ─── resolve_look_at bundle chain atoms ──────────────────────────── */
typedef Struct_(Binds_ResolveLookAtSub) {
P3_S4* target; /* U4 (C-side P3_S4* — read by atom 0 directly; NOT a scratchpad address) */
P3_S4* eye; /* U4 (C-side P3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */
V3_S4* up_in; /* U4 (C-side V3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */
ResolveLookAtScratch* scratchpad;
};
/* Atom 0 in the bundle: input_and_sub. Stages C-side inputs into the scratchpad and computes fwd = target - eye.
* Staging work:
* * Stage eye.x/y/z → scratch (for atom 6's translation column)
* * Stage up_in.x/y/z → scratch (for atom 2's outer-product operand)
* * Compute fwd = target - eye, store fwd.x/y/z → scratch+0/+4/+8 (for atom 1)
* GPR codes (assigned by resolve_look_at_init):
* r_target_ptr : R_T0
* r_eye_ptr : R_T1
* r_up_in_ptr : R_T2
* r_scratch : R_T4 (R_ResolveScratch; wave-context carrier)
* r_tmp0 : R_T3 (stage eye/up_in + load eye.y)
* r_tmp1 : R_T5 (stage eye/up_in + load eye.z)
* r_tmp2 : R_T6 (stage eye/up_in + load target.x)
* r_tmp3 : R_T7 (stage eye/up_in + load target.y)
* R_AT : hardcoded (load eye.y / eye.z / target.z)
* R_V0 : hardcoded (load eye.z / target.z)
* Pool cost: 8 GPRs + R_T4 (carrier) + R_AT + R_V0 (hardcoded) = 11 GPRs.
*/
internal MipsAtom* resolve_look_at__input_and_sub_proc(AtomArena_R aa,
// TODO(Ed): We can resolve scratch at anytime its fixed to a specific address.
U4 r_scratch
, U4 r_target_ptr,U4 r_eye_ptr, U4 r_up_in_ptr
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2, U4 r_tmp3
) MipsAtom_Proc_(aa, {
load_word(r_target_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,target)),
load_word(r_eye_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)),
load_word(r_up_in_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)),
load_word(r_scratch, R_TapePtr, O_(Binds_ResolveLookAtSub,scratchpad)),
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtSub)),
// Stage eye.x/y/z into the scratchpad (atom 6 reads these for the translation column).
mac_load_p3s4( r_tmp0, r_tmp1, r_tmp2, r_eye_ptr, 0),
mac_store_p3s4(r_tmp0, r_tmp1, r_tmp2, r_scratch, O_(ResolveLookAtScratch,eye)),
/* Stage up_in.x/y/z into the scratchpad. */
mac_load_p3s4( r_tmp0, r_tmp1, r_tmp2, r_up_in_ptr, 0),
mac_store_p3s4(r_tmp0, r_tmp1, r_tmp2, r_scratch, O_(ResolveLookAtScratch,up_in)),
/* Compute fwd = target - eye. */
mac_load_p3s4(r_tmp0, r_tmp1, r_tmp2, r_target_ptr, 0),
mac_load_p3s4(r_tmp3, R_AT, R_V0, r_eye_ptr, 0),
mac_sub_v3s4(
r_tmp0, r_tmp1, r_tmp2,
r_tmp3, R_AT, R_V0),
mac_store_v3s4(r_tmp0, r_tmp1, r_tmp2, r_scratch, O_(ResolveLookAtScratch,fwd)),
mac_yield()
})
/* Atom 2: cross uz × up_in → right. */
internal MipsAtom* resolve_look_at__cross_uz_up_in_to_right_proc(AtomArena_R aa, U4 r_scratch
, U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */
, U4 r_d /* load b.x */
, U4 r_f, U4 r_g, U4 r_h /* r_f = &right (out ptr), r_g = &uz, r_h = &up_in */
) MipsAtom_Proc_(aa, {
/* FIX: build packed RT22+RT33 with proper sign extension. */
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
add_si(r_h, r_scratch, O_(ResolveLookAtScratch,up_in)), /* r_h = &up_in */
add_si(r_f, r_scratch, O_(ResolveLookAtScratch,right)), /* r_f = &right (out) */
nop,
/* Load a (uz).x/y/z into r_a/r_b/r_c. */
load_word(r_a, r_g, O_(V3_S4,x)),
load_word(r_b, r_g, O_(V3_S4,y)),
load_word(r_c, r_g, O_(V3_S4,z)),
nop,
/* Load b (up_in).x/y/z into r_d + R_AT/R_V0 (R_AT/R_V0 are hardcoded scratch). */
load_word(r_d, r_h, O_(V3_S4,x)),
load_word(R_AT, r_h, O_(V3_S4,y)),
load_word(R_V0, r_h, O_(V3_S4,z)),
nop,
/* Save the two RT control-register slots OP will clobber. We reuse
* r_g/r_h (scratch pointers, no longer needed) as the save targets. */
gte_mv_from_ctrl_r(r_g, gte_cr_RT11), /* r_g = C2 r0 (RT11|RT12) */
gte_mv_from_ctrl_r(r_h, gte_cr_RT22), /* r_h = C2 r4 (RT22|RT33) */
/* Load uz.x/uz.y/uz.z into COP2 control registers.
* OP reads D1 = RT11 from $0.low, D2 = RT22 from $2.high, D3 = RT33 from $4.high.
* RT22 is in BOTH $2.high AND $4.low (shared bit position). OP reads from $2.high.
* So set RT22 via ctc2 r_b, $2 (sets $2.high = a.y.high = RT22, $2.low = a.y.low = RT13).
* Then set RT33 via ctc2 r_c, $4 (sets $4.high = a.z.high = RT33, $4.low = a.z.low).
* The $2 and $4 writes don't clobber each other (separate registers).
* The 2nd ctc2 DOES clobber $4.low (becomes a.z.low, NOT a.y.high), but since OP
* reads RT22 from $2.high (which the 2nd ctc2 doesn't touch), D2 is still a.y.high.
* This is libpsyx's OuterProduct12 convention EXACTLY. */
gte_mv_to_ctrl_r(r_b, gte_cr_RT13), /* $2 = r_b = a.y. RT13=a.y.low, RT22=a.y.high. */
gte_mv_to_ctrl_r(r_c, gte_cr_RT22), /* $4 = r_c = a.z. RT22=a.z.low, RT33=a.z.high. */
/* Load uz into the RT diagonal. */
gte_mv_to_ctrl_r(r_a, gte_cr_RT11), /* D1 = RT11 = uz.x (low 16 of $0, sign-extended by OP). */
nop2, /* CTC2 retirement (CPU→COP2 2-slot delay) */
/* Load up_in into IR (the second operand for OP). */
gte_mv_to_data_r(r_d, C2_IR1), /* IR1 = up_in.x */
gte_mv_to_data_r(R_AT, C2_IR2), /* IR2 = up_in.y */
gte_mv_to_data_r(R_V0, C2_IR3), /* IR3 = up_in.z */
nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
gte_cmdw_outer_product, /* OP: MAC1/2/3 = uz × up_in
* MAC1 = IR3*D2 - IR2*D3 = up_in.z*uz.y.high - up_in.y*uz.z.high
* MAC2 = IR1*D3 - IR3*D1 = up_in.x*uz.z.high - up_in.z*uz.x
* MAC3 = IR2*D1 - IR1*D2 = up_in.y*uz.x - up_in.x*uz.y.high
* For up_in = (0, -fp_one, 0):
* MAC1 = 0 - (-fp_one)*uz.z.high = fp_one*uz.z.high
* MAC2 = 0 - 0 = 0
* MAC3 = (-fp_one)*uz.x - 0 = -fp_one*uz.x */
/* Restore the RT slots we clobbered. */
gte_mv_to_ctrl_r(r_g, gte_cr_RT11), /* restore C2 r0 (RT11|RT12) */
gte_mv_to_ctrl_r(r_h, gte_cr_RT22), /* restore C2 r4 (RT22|RT33) */
/* mfc2 MAC1/2/3 → r_a/r_b/r_c (out.x/y/z). */
gte_mv_from_data_r(r_a, C2_MAC1),
gte_mv_from_data_r(r_b, C2_MAC2),
gte_mv_from_data_r(r_c, C2_MAC3),
nop, /* MFC2 retirement */
/* Right-shift MAC by 12 to convert from GTE's S12.20 fixed-point scale back to libpsyx OuterProduct12 convention (S12.0, fp_one=4096=1<<12).
* Without this, MAC values (~16M for unit-vector cross products) overflow the GTE's 16-bit IR registers when atom 3 normalizes via mtc2. */
shift_aright(r_a, r_a, 12),
shift_aright(r_b, r_b, 12),
shift_aright(r_c, r_c, 12),
/* Store out.x/y/z to r_f (out ptr = scratch+32). */
store_word(r_a, r_f, O_(V3_S4,x)),
store_word(r_b, r_f, O_(V3_S4,y)),
store_word(r_c, r_f, O_(V3_S4,z)),
mac_yield()
})
/* Atom 4: cross uz × ux → up. */
internal MipsAtom* resolve_look_at__cross_uz_ux_to_up_proc(AtomArena_R aa, U4 r_scratch
, U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */
, U4 r_d /* load b.x */
, U4 r_f, U4 r_g, U4 r_h /* r_f = &up (out ptr), r_g = &uz, r_h = &ux */
) MipsAtom_Proc_(aa, {
/* Compute the three scratch pointers from r_scratch. */
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
add_si(r_h, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_h = &ux */
add_si(r_f, r_scratch, O_(ResolveLookAtScratch,up)), /* r_f = &up (out) */
nop,
/* Load a (uz).x/y/z into r_a/r_b/r_c. */
load_word(r_a, r_g, O_(V3_S4,x)),
load_word(r_b, r_g, O_(V3_S4,y)),
load_word(r_c, r_g, O_(V3_S4,z)),
nop,
/* Load b (ux).x/y/z into r_d + R_AT/R_V0. */
load_word(r_d, r_h, O_(V3_S4,x)),
load_word(R_AT, r_h, O_(V3_S4,y)),
load_word(R_V0, r_h, O_(V3_S4,z)),
nop,
/* OP reads D1/D2/D3 from RT11/RT22/RT33 ($0/$2/$4), not V0/V1/V2.
* Mirror atom 1: cfc2 RT save, ctc2 RT diagonal from uz, mtc2 IR from ux,
* ctc2 RT restore. */
/* Save the two RT control-register slots OP will clobber (reusing
* r_g/r_h — they're no longer needed as scratch pointers). */
gte_mv_from_ctrl_r(r_g, gte_cr_RT11), /* r_g = C2 $0 (RT11|RT12) */
gte_mv_from_ctrl_r(r_h, gte_cr_RT22), /* r_h = C2 $4 (RT22|RT33) */
/* Load uz into the RT diagonal — same packing as atom 1.
* OP reads D1 = RT11 from $0.low, D2 = RT22 from $2.high, D3 = RT33 from $4.high.
* RT22 is shared between $2.high and $4.low — the ctc2 sequence to $2 then $4
* sets RT22 to uz.y.high (via $2), then to uz.z.low (via $4). OP reads
* RT22 from $2.high which the second ctc2 doesn't touch, so D2 stays uz.y.high.
* (This is libpsyx OuterProduct12 convention EXACTLY.) */
gte_mv_to_ctrl_r(r_b, gte_cr_RT13), /* $2 = uz.y. RT13=uz.y.low, RT22=uz.y.high. */
gte_mv_to_ctrl_r(r_c, gte_cr_RT22), /* $4 = uz.z. RT22=uz.z.low, RT33=uz.z.high. */
gte_mv_to_ctrl_r(r_a, gte_cr_RT11), /* $0 = uz.x. RT11=uz.x. */
nop2, /* CTC2 retirement (CPU→COP2 2-slot delay) */
/* Load ux into the IR registers (the second operand for OP). */
gte_mv_to_data_r(r_d, C2_IR1), /* IR1 = ux.x */
gte_mv_to_data_r(R_AT, C2_IR2), /* IR2 = ux.y */
gte_mv_to_data_r(R_V0, C2_IR3), /* IR3 = ux.z */
nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
gte_cmdw_outer_product,
/* Restore the RT slots we clobbered. */
gte_mv_to_ctrl_r(r_g, gte_cr_RT11), /* restore C2 $0 (RT11|RT12) */
gte_mv_to_ctrl_r(r_h, gte_cr_RT22), /* restore C2 $4 (RT22|RT33) */
gte_mv_from_data_r(r_a, C2_MAC1),
gte_mv_from_data_r(r_b, C2_MAC2),
gte_mv_from_data_r(r_c, C2_MAC3),
nop,
/* Right-shift MAC by 12 to convert from GTE's S12.20 scale back to libpsyx
* OuterProduct12 convention (S12.0, fp_one=4096). See atom 1 for rationale. */
shift_aright(r_a, r_a, 12),
shift_aright(r_b, r_b, 12),
shift_aright(r_c, r_c, 12),
store_word(r_a, r_f, O_(V3_S4,x)),
store_word(r_b, r_f, O_(V3_S4,y)),
store_word(r_c, r_f, O_(V3_S4,z)),
mac_yield()
})
typedef Struct_(Binds_ResolveLookAtPopAndTrans) {
U4 look_at; /* U4 (MT3_S2S4* — destination matrix address) */
};
/* Atom 6 in the bundle: write look_at->m[][] from ux/uy/uz, then compute the translation column t[] = R * (-eye).
*
* GPR codes (assigned by resolve_look_at_init):
* r_look_at : MT3_S2S4* (popped from tape; output matrix destination)
* r_pux : pointer to ux (offset O_(ResolveLookAtScratch,ux))
* r_puy : pointer to uy (offset O_(ResolveLookAtScratch,uy))
* r_puz : pointer to uz (offset O_(ResolveLookAtScratch,uz))
* r_peye : pointer to eye (offset O_(ResolveLookAtScratch,eye))
* r_tmp0/1/2 : atom-local scratch (load + MVMVA + store temps)
*
* 4 pointer regs (r_pux/r_puy/r_puz/r_peye) are DEDICATED — they hold the scratch addresses for the entire body.
* They are computed in-body via `add_si(r_px, r_scratch, O_(ResolveLookAtScratch, field))` so no tape-data pointer is needed.
*
* Struct layout (per duffle/math.h):
* MT3_S2S4 { A3x3_S2 m; A3_S4 t; } → m[][] is S2 packed (9 × 2 = 18 bytes at offset 0)
* t[0/1/2] is S4 (3 × 4 = 12 bytes at offset 18)
*
* Translation column: GTE MVMVA with the world rotation matrix pre-set
* (helper emits set_gte_world before the bundle, per the bundle design).
* MVMVA computes R * pos (with cv=0/mx=0/sf=0/v=0); MAC1/2/3 = R * (-eye).
* Pool cost: r_look_at (1) + r_scratch (R_T4 carrier) + 4 ptr regs + 3 tmp regs = 9 GPRs.
*/
internal MipsAtom* resolve_look_at__populate_proc(AtomArena_R aa
, U4 r_look_at
, U4 r_scratch
, U4 r_pux, U4 r_puy, U4 r_puz
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2
) MipsAtom_Proc_(aa, {
/* Pop look_at* (the matrix output) — advance R_TapePtr by 4 bytes. */
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
/* Compute the 3 scratch pointers in their dedicated GPRs (eye isn't needed by 6a — 6b reads it). */
add_si(r_pux, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_pux = &ux */
add_si(r_puy, r_scratch, O_(ResolveLookAtScratch,uy)), /* r_puy = &uy */
add_si(r_puz, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_puz = &uz */
nop,
/* ── m[0] = (S2)ux ── */
load_word(r_tmp0, r_pux, O_(V3_S4,x)),
load_word(r_tmp1, r_pux, O_(V3_S4,y)),
load_word(r_tmp2, r_pux, O_(V3_S4,z)),
nop,
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[0][0])),
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[0][1])),
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[0][2])),
/* ── m[1] = (S2)uy ── */
load_word(r_tmp0, r_puy, O_(V3_S4,x)),
load_word(r_tmp1, r_puy, O_(V3_S4,y)),
load_word(r_tmp2, r_puy, O_(V3_S4,z)),
nop,
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[1][0])),
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[1][1])),
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[1][2])),
/* ── m[2] = (S2)uz ── */
load_word(r_tmp0, r_puz, O_(V3_S4,x)),
load_word(r_tmp1, r_puz, O_(V3_S4,y)),
load_word(r_tmp2, r_puz, O_(V3_S4,z)),
nop,
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[2][0])),
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[2][1])),
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[2][2])),
/* Zero t[0..2] — atom 6c writes the final values here. */
store_word(R_0, r_look_at, O_(MT3_S2S4,t[0])),
store_word(R_0, r_look_at, O_(MT3_S2S4,t[1])),
store_word(R_0, r_look_at, O_(MT3_S2S4,t[2])),
mac_yield()
})
/* Atom 6b in the bundle: matrix-vector product off = R * (-eye) >> 12.
* Uses RTPS with V0 loaded from scratch via lwc2. The RT matrix is
* pre-loaded by atom 6a.5 (resolve_look_at__load_rt).
* Stores off to scratch+96 (overwriting the packed pos).
*
* GPR codes (assigned by resolve_look_at_init):
* r_scratch : R_ResolveScratch (R_T4) — scratch base
* r_peye : pointer to eye (slot +96, reused as off destination)
* r_tmp0/1/2: -eye + GTE transfer scratch
*
* Pool cost: r_scratch (carrier) + 1 ptr reg + 3 tmp regs = 5 GPRs.
*/
internal MipsAtom* resolve_look_at__matrix_vector_proc(AtomArena_R aa
, U4 r_scratch
, U4 r_peye
, U4 r_look_at
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2
) MipsAtom_Proc_(aa, {
/* === EXACT C11 ApplyMatrixLV replication ===
* The C11 does:
* 1. ctc2 RT matrix (5 ctc2s to C2[0..4])
* 2. lw v.x/y/z from memory
* 3. S15 decomposition (negu + sra 15 + negu + andi 0x7FFF + negu)
* 4. mtc2 HIGH bits to IR1/2/3, nop, MVMVA pass1 (sf=0, mx=0, v=3, cv=3)
* 5. mfc2 MACs
* 6. mtc2 LOW bits to IR1/2/3, nop, MVMVA pass2 (sf=1, mx=0, v=3, cv=3)
* 7. mfc2 MACs
* 8. Combine: (pass1 << 3) + pass2
*
* For S16-fitting pos (|pos| < 32768), pos >> 15 = 0, so pass1 = 0.
* The combine simplifies: result = 0 + pass2 = pass2.
* So we skip the S15 decomposition and just do pass 2 directly.
* We still use v=3 (IR input) and mx=0 (RT matrix) like the C11. */
/* Pop look_at* from tape. */
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
/* r_peye = &eye (slot +96, reused as off destination). */
add_si(r_peye, r_scratch, O_(ResolveLookAtScratch,eye)),
nop,
/* === Load RT matrix from look_at into C2[0..4] via ctc2 ===
* Exact s ame sequence as set_gte_mt3s2s4 / C11's ApplyMatrixLV. */
load_word( r_tmp0, r_look_at, 0), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT11),
load_word( r_tmp0, r_look_at, 4), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT12),
load_word( r_tmp0, r_look_at, 8), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT13),
load_word( r_tmp0, r_look_at, 12), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT21),
load_half_u(r_tmp0, r_look_at, 16), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT22),
nop2, /* CTC2 retirement (2 slots × 5 ctc2s) */
/* Load pos = -eye after the matrix load releases r_tmp0. */
load_word(r_tmp0, r_peye, O_(P3_S4,x)),
load_word(r_tmp1, r_peye, O_(P3_S4,y)),
load_word(r_tmp2, r_peye, O_(P3_S4,z)),
nop,
sub_u(r_tmp0, R_0, r_tmp0), /* pos.x = -eye.x */
sub_u(r_tmp1, R_0, r_tmp1),
sub_u(r_tmp2, R_0, r_tmp2),
/* === mtc2 pos (as S16) to IR1/2/3 ===
* The GTE takes low 16 bits. pos fits in S16. For negative pos, the
* 32-bit sign-extended value's low 16 bits = correct S16. */
/* Mask pos to 16 bits to be safe. For S16-fitting pos, pos & 0xFFFF
* gives the correct S16 value (sign bit preserved). */
/* r_tmp0/1/2 already have pos values. */
gte_mv_to_data_r(r_tmp0, C2_IR1),
gte_mv_to_data_r(r_tmp1, C2_IR2),
gte_mv_to_data_r(r_tmp2, C2_IR3),
nop2, /* MTC2 retirement (2 slots) */
/* === MVMVA pass 2 — C11 ApplyMatrixLV command ===
* sf=1, mx=0 (RT), v=3 (IR), cv=3. Reads RT × IR >> 12. */
gte_cmdw_mvmva_c11_pass2,
nop, /* GTE interlock */
/* === mfc2 MAC1/2/3 → r_tmp0/1/2 === */
gte_mv_from_data_r(r_tmp0, C2_MAC1),
gte_mv_from_data_r(r_tmp1, C2_MAC2),
gte_mv_from_data_r(r_tmp2, C2_MAC3),
nop,
/* === Store off → scratch+96 (overwriting pos) === */
store_word(r_tmp0, r_peye, O_(V3_S4,x)),
store_word(r_tmp1, r_peye, O_(V3_S4,y)),
store_word(r_tmp2, r_peye, O_(V3_S4,z)),
mac_yield()
})
/* Atom 6c in the bundle: copy scratch+96 (off, written by atom 6b) → look_at->t[].
* Uses mac_trans_matrix component (m->t = v, libgte TransMatrix semantics = struct copy).
*
* GPR codes (assigned by resolve_look_at_init):
* r_look_at : MT3_S2S4* (popped from tape; output matrix destination)
* r_scratch : R_ResolveScratch (R_T4) — scratch base
* r_off_ptr : pointer to off (= &scratch.eye, reused slot)
* r_tmp0 : transfer reg for mac_trans_matrix
*
* Pool cost: r_look_at (1) + r_scratch (carrier) + r_off_ptr + 1 clobber = 4 GPRs.
*/
I_ MipsAtom* resolve_look_at__trans_matrix_proc(AtomArena_R aa
, U4 r_look_at, U4 r_scratch, U4 r_off_ptr
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2
) MipsAtom_Proc_(aa, {
/* Pop look_at* from tape. */
// load_word(r_Vlook_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
// add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
/* r_off_ptr = &off (= &scratch.eye since atom 6b overwrote eye with off). */
add_si(r_off_ptr, r_scratch, O_(ResolveLookAtScratch,eye)),
nop,
/* Copy off → look_at.t[] (mac_trans_matrix: m->t = v). */
mac_trans_mt3s3s4(r_look_at, r_off_ptr, r_tmp0, r_tmp1, r_tmp2),
mac_yield()
})
#pragma endregion Atom Procs
#pragma region Baked Atoms
enum {
@@ -545,112 +105,112 @@ internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
) {
/* display[0] = (0, 0, 320, 240); rest of struct zeroed. */
add_ui(R_ScreenX, R_0, ScreenRes_X), add_ui(R_ScreenY, R_0, ScreenRes_Y),
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area.width) + O_(DoubleBuffer,display[0])),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,display_area) + O_(DoubleBuffer,display[0])),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + O_(DoubleBuffer,display[0])),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + O_(DoubleBuffer,display[0])),
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area.width) + OA_(DoubleBuffer,display,0)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,display_area) + OA_(DoubleBuffer,display,0)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + OA_(DoubleBuffer,display,0)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,0)),
/* display[1] = (0, 240, 320, 240); rest of struct zeroed. */
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area) + O_(DoubleBuffer,display[1])),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + O_(DoubleBuffer,display[1])),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + O_(DoubleBuffer,display[1])),
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area) + OA_(DoubleBuffer,display,1)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + OA_(DoubleBuffer,display,1)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,1)),
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + O_(DoubleBuffer,draw[0])), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
mac_store_v2s2( R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + O_(DoubleBuffer,draw[0])), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + OA_(DoubleBuffer,draw,0)), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
mac_store_v2s2( R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + OA_(DoubleBuffer,draw,0)), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + O_(DoubleBuffer,draw[1])),
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + OA_(DoubleBuffer,draw,1)),
/* draw[0].texture_window = (0, 0, 0, 0); two word-zeroes cover the full 8-byte tw field. */
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + O_(DoubleBuffer,draw[0])),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + O_(DoubleBuffer,draw[0])),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + OA_(DoubleBuffer,draw,0)),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,0)),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,drawing_offset[0].x) + O_(DoubleBuffer,draw[1])),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + O_(DoubleBuffer,draw[1])),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + O_(DoubleBuffer,draw[1])),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,drawing_offset[0].x) + OA_(DoubleBuffer,draw,1)),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + OA_(DoubleBuffer,draw,1)),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,1)),
/* draw[0].texture_page = 10 (gp0_tpage_default). C11 SetDefDrawEnv at C11_only.elf:0x8001273C writes the same 0x0A. . */
add_ui(R_T0, R_0, gp0_tpage_default),
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + O_(DoubleBuffer,draw[0])),
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + O_(DoubleBuffer,draw[1])),
add_ui(R_T0, R_0, gp0_tpage_default),
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,0)),
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,1)),
/* draw[0] control bytes: flag_dither=1, flag_draw_on_display=1 (the dfe bit per psx-spx; libpsyx sets it via `SetDefDrawEnv`'s conditional at C11_only.elf:0x80012728), enable_auto_clear=1. Each byte is named;
* the previous `store_word(R_0, ..., +20)` overwrote all four with zero. */
add_ui(R_T0, R_0, 1),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + O_(DoubleBuffer,draw[0])),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + O_(DoubleBuffer,draw[0])),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + O_(DoubleBuffer,draw[0])),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + O_(DoubleBuffer,draw[1])),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + O_(DoubleBuffer,draw[1])),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + O_(DoubleBuffer,draw[1])),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + OA_(DoubleBuffer,draw,0)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + OA_(DoubleBuffer,draw,0)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,0)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + OA_(DoubleBuffer,draw,1)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + OA_(DoubleBuffer,draw,1)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,1)),
/* draw[0].initial_bg_color = (r=7, g=7, b=7). */
add_ui(R_T0, R_0, 7),
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + O_(DoubleBuffer,draw[0])),
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + O_(DoubleBuffer,draw[1])),
add_ui(R_T0, R_0, 7),
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,0)),
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,1)),
mac_yield(),
};
/* gp_screen_init's GPR setup. Tests the mixed user-pinning + auto-reg pattern:
* - R_IO_BaseAddr = R_T4 (user-pinned via atom_reg; pre-existing)
* - R_GP1_Offset = R_T2 (user-pinned via atom_reg; NEW -- for GPIO_PORT1_OFFSET)
* - R_ScreenX = R_T5 (user-pinned via atom_reg; used as a transfer and GTE setup reg)
* - R_GpTmp = auto-allocated by the lua pass and used for several GPU transfers;
* the C preprocessor resolves it to the chosen free pool GPR.
*
* For gp_screen_init, the auto-reg pool exclusions are:
* user_pinned (from the corpus register_alias_registry) : R_T0..R_T7 (all 8 user-pinned across hello_camera.atom.c)
* body-parsed physical registers : aliases resolve through the registry;
* the body uses R_ScreenX, not raw R_T5
* source_pool after both subtractions : {R_V0, R_V1} only
* R_GpTmp gets R_V0 (the first-fit choice). Its repeated GPU-transfer use proves that the
* auto-reg allocation is active while the R_ScreenX references prove the pinned alias is used.
* R_TapePtr (R_T9), R_AtomJmp (R_T8), R_AT are excluded from the POOL by construction in
* passes/auto_reg.lua -- see the "obvious exclusions" comment block at the top of that file.
*/
enum {
R_IO_BaseAddr = R_T4 atom_reg, /* Caller-pinned: IO_BASE_ADDR = 0x1F800000 */
R_GP1_Offset = R_T2 atom_reg, /* Caller-pinned: GPIO_PORT1_OFFSET = 0x10 */
atom_auto_reg(gp_screen_init, R_GpTmp), /* Auto-allocated scratch; resolved to a free pool GPR by the lua pass. C-preprocessor expands to R_GpTmp = R_GpTmp_Code with an atom_auto_reg trailing comment. */
#define R_IO_BaseAddr_Code R_T4_Code
#define R_GP1_Offset_Code R_T2_Code
};
internal MipsAtom_(gp_screen_init) atom_info(atom_phase(screen_init), atom_reads(R_IO_BaseAddr)) {
store_word(R_0, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(00h) Reset */
mac_gcmd_push(gp1_word_ResetCmdBuffer(), R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(01h) ClearFIFO; uses pinned R_ScreenX as the transfer reg. */
mac_gcmd_push(gp1_word_AcknowledgeIRQ(), R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(02h) AckIRQ; uses pinned R_ScreenX as the transfer reg. */
mac_gcmd_push(gp1_word_DisplayOn(), R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(03h) Display ON; uses pinned R_ScreenX as the transfer reg. */
mac_gcmd_push(gp1_word_dma_to_gpu(), R_GpTmp, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(04h) DMADirection=2 (CPU->GPU). libpsyx's per-frame PutDrawEnv/DrawOTag use DMA2; without this the DMA queue never drains. Uses auto-allocated R_GpTmp. */
mac_gcmd_push(gp1_word_StartDisplayArea(), R_GpTmp, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(05h) StartDisplayArea (X=0, Y=0); uses auto-allocated R_GpTmp. */
mac_gcmd_push(gp1_word_ResetCmdBuffer(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(01h) ClearFIFO */
mac_gcmd_push(gp1_word_AcknowledgeIRQ(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(02h) AckIRQ */
mac_gcmd_push(gp1_word_DisplayOn(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(03h) Display ON */
mac_gcmd_push(gp1_word_dma_to_gpu(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(04h) DMADirection=2 (CPUGPU). libpsyx's per-frame PutDrawEnv/DrawOTag use DMA2; without this the DMA queue never drains. */
mac_gcmd_push(gp1_word_StartDisplayArea(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(05h) StartDisplayArea (X=0, Y=0) */
/* GP1: DisplayMode + Display Ranges. */
mac_gcmd_push(gp1_word_display_mode_320x240_15bit_ntsc, R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
mac_gcmd_push(gp1_word_horizontal_range_ntsc, R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
mac_gcmd_push(gp1_word_vertical_range_ntsc, R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
/* GP1: DisplayMode + Display Ranges */
mac_gcmd_push(gp1_word_display_mode_320x240_15bit_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
mac_gcmd_push(gp1_word_horizontal_range_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
mac_gcmd_push(gp1_word_vertical_range_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
/* GTE: SetGeomOffset (OFX, OFY) — ScreenRes_CenterX, ScreenRes_CenterY. */
load_upper_i(R_ScreenX, ScreenRes_CenterX), gte_mv_to_ctrl_r(R_ScreenX, gte_cr_OFX_Code),
load_upper_i(R_ScreenX, ScreenRes_CenterY), gte_mv_to_ctrl_r(R_ScreenX, gte_cr_OFY_Code),
load_upper_i(R_T5, ScreenRes_CenterX), gte_mv_to_ctrl_r(R_T5, gte_cr_OFX_Code),
load_upper_i(R_T5, ScreenRes_CenterY), gte_mv_to_ctrl_r(R_T5, gte_cr_OFY_Code),
/* GTE: SetGeomScreen (H) — CR26 (per PSX-SPX / libpsyx), value is the raw projection-plane distance, NOT shifted. */
add_ui(R_ScreenX, R_0, ScreenZ), gte_mv_to_ctrl_r(R_ScreenX, gte_cr_H_Code),
add_ui(R_T5, R_0, ScreenZ), gte_mv_to_ctrl_r(R_T5, gte_cr_H_Code),
/* GP1: DisplayEnable — bit 0 = 0 (Display ON). */
mac_gcmd_push(gp1_word_DisplayOn(), R_GpTmp, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* Uses auto-allocated R_GpTmp. */
mac_gcmd_push(gp1_word_DisplayOn(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
mac_yield(),
};
/* ----- pad_apply_input -----
* Reads pad[0].buttons + pad[0].left_x;
* Applies the input-semantics deltas to cube_rot.y + floor_rot.y:
* - D-pad Left: cube_rot.y += 30, floor_rot.y += 5
* - D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5
* - Analog stick X (dead zone 0x70..0x90):
* cube delta = (0x80 - left_x) >> 2 (range approx -32..+32)
* floor delta = (0x80 - left_x) >> 5 (range approx -4..+4)
* - D-pad + analog deltas add when used together.
*
* Convention:
* pad_state = 0 means no buttons active.
* The fail-safe zero-button value flows through unchanged, so a disconnected/fresh pad produces no rotation.
* The branch_le_zero pattern below matches the existing pad_input_demo convention (atom body lines 248/257).
*
* Signed-delta trick:
* load_byte_u zero-extends left_x to 32 bits; sub_u from 0x80 wraps to a SIGNED two's-complement value in the negative range;
* shift_aright (sra) then correctly sign-extends the shift for both positive (left_x < 0x80) and negative (left_x > 0x80) cases.
* Digital pads publish left_x = 0x80 → delta = 0 → no rotation, so the analog step is naturally a no-op for digital controllers.
*/
typedef Struct_(Binds_PadApplyInput) {
PadState* state;
V3_S2* cube_rot;
V3_S2* floor_rot;
PadState* state;
V3_S2* cube_rot;
V3_S2* floor_rot;
};
enum {
R_PadStateT5 = R_T5 atom_reg,
R_CubeRot = R_T1 atom_reg,
R_FloorRot = R_T2 atom_reg,
};
internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyInput)
internal MipsAtom_(pad_apply_input) atom_info(atom_bind(Binds_PadApplyInput)
, atom_reads(R_T0, R_CubeRot, R_FloorRot, R_T3, R_T4, R_PadStateT5, R_TapePtr)
, atom_writes( R_CubeRot, R_FloorRot)
) {
@@ -665,7 +225,7 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
// Note(Ed): Potential op with delay slot?
/* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)),
and_i(R_T3, R_T0, pad0_(Pad_Left)), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)),
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
add_si( R_T4, R_T4, 30),
@@ -675,7 +235,7 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
atom_label(exit_dpad_left)
/* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)),
and_i(R_T3, R_T0, pad0_(Pad_Right)), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)),
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
add_si( R_T4, R_T4, -30),
@@ -686,21 +246,21 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
/* Analog left-stick X: dead zone 0x70..0x90.
* Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)),
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left_x)),
/* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly).
* set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */
add_ui(R_T4, R_0, PadDeadZone_HighBound), set_lt_u(R_T4, R_T3, R_T4), branch_ne(R_T4, R_0, atom_offset(dead_zone_low_check, dead_low_active)),
add_ui(R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_low_active */
add_ui(R_T4, R_0, 0x70), set_lt_u(R_T4, R_T3, R_T4), branch_ne(R_T4, R_0, atom_offset(dead_zone_low_check, dead_low_active)),
add_ui(R_T4, R_0, 0x80), /* BD-slot: pre-load 0x80 for dead_low_active */
atom_label(dead_check_upper)
/* left_x >= 0x70 → check upper bound. */
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), /* reload */
add_ui( R_T4, R_0, PadDeadZone_HighBound),
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left_x)), /* reload */
add_ui( R_T4, R_0, 0x90),
/* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)),
add_ui( R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_high_active */
add_ui( R_T4, R_0, 0x80), /* BD-slot: pre-load 0x80 for dead_high_active */
jump_rel(atom_offset(dead_zone_skip, exit_stick)),
mac_yield_load(),
@@ -713,7 +273,8 @@ atom_label(dead_low_active)
/* R_T4 = cube_delta */
shift_aright(R_T4, R_T3, 2),
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), nop,
load_half( R_T0, R_CubeRot, O_(V3_S2,y)),
nop,
add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
/* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap;
@@ -734,7 +295,8 @@ atom_label(dead_high_active)
/* delta = 0x80 - left_x (signed negative). */
shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), nop,
load_half( R_T0, R_CubeRot, O_(V3_S2,y)),
nop,
add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
@@ -753,63 +315,7 @@ atom_label(exit_stick)
};
enum {
R_Cam = R_T4 atom_reg,
R_CamPadState = R_T5 atom_reg,
};
typedef Struct_(Binds_PadInputCam) {
PadState* state;
Camera* cam;
};
internal MipsAtom_(pad_input_cam) atom_info(atom_bind(Binds_PadInputCam)
, atom_reads( R_Cam, R_CamPadState, R_TapePtr)
, atom_writes(R_Cam)
) {
/* Bind pop: state → R_CamPadState (R_T5), cam → R_Cam (R_T4), advance R_TapePtr by 8. */
load_word(R_CamPadState, R_TapePtr, O_(Binds_PadInputCam,state)),
load_word(R_Cam, R_TapePtr, O_(Binds_PadInputCam,cam)),
add_ui_self( R_TapePtr, S_(Binds_PadInputCam)),
/* Load pad[0].buttons into R_T0; nop fills the load-delay slot. */
load_word(R_T0, R_CamPadState, O_(PadState,buttons)),
load_word(R_T1, R_Cam, O_(Camera,pos.x)), // BD-Slot.
// D-pad Left → cam.pos.x -= 50. and_i fulfills BD-slot for load on R_Cam.
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(left_x, exit_left_x)), mac_yield_load(),
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
atom_label(exit_left_x)
/* D-pad Right → cam.pos.x += 50. Reuses R_T1 from Left. */
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(right_x, exit_right_x)), nop,
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
atom_label(exit_right_x)
/* D-pad Up → cam.pos.y -= 50. Load pos.y BEFORE the andi. */
load_word(R_T1, R_Cam, O_(Camera,pos.y)),
and_i(R_T3, R_T0, Pad_Up), branch_le_zero(R_T3, atom_offset(up_y, exit_up_y)), nop,
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
atom_label(exit_up_y)
/* D-pad Down → cam.pos.y += 50. Reuses R_T1 from Up. */
and_i(R_T3, R_T0, Pad_Down), branch_le_zero(R_T3, atom_offset(down_y, exit_down_y)), nop,
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
atom_label(exit_down_y)
/* D-pad Cross → cam.pos.z -= 50. Load pos.z BEFORE the andi. */
load_word(R_T1, R_Cam, O_(Camera,pos.z)),
and_i(R_T3, R_T0, Pad_Cross), branch_le_zero(R_T3, atom_offset(cross_z, exit_cross_z)), nop,
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
atom_label(exit_cross_z)
/* D-pad Circle → cam.pos.z += 50. Reuses R_T1 from Cross. */
and_i(R_T3, R_T0, Pad_Circle), branch_le_zero(R_T3, atom_offset(circle_z, exit_circle_z)), nop,
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
atom_label(exit_circle_z)
mac_yield_tail(),
};
enum {
R_PrimCursor = R_T7 atom_reg atom_type(U4*), /* Output cursor (primitive buffer) */
R_PrimCursor = R_T7 atom_reg atom_type(U4*), /* VRAM output cursor (primitive buffer) */
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2*), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
R_VertBase = R_T5 atom_reg atom_type(V3_S2*), /* Base address of the vertex array */
R_OtBase = R_T6 atom_reg atom_type(U4*), /* Base address of the Ordering Table */
@@ -818,6 +324,7 @@ enum {
#define R_VertBase_Code R_T5_Code
#define R_OtBase_Code R_T6_Code
};
typedef Struct_(Binds_CubeTri) {
U4 PrimCursor;
V4_S2* FaceCursor;
@@ -837,7 +344,7 @@ internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_
mac_yield()
};
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
internal
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
@@ -854,9 +361,9 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
/* BD-slot: Write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
* harmless because the OT entry that points to this prim is created later. */
/* BD-slot: write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
* harmless because the OT entry that points to this prim is created later, only on the body path. */
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
@@ -872,7 +379,7 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_G4)),
mac_insert_ot_tag_g4(R_OtBase, R_PrimCursor),
mac_format_g4_color(R_PrimCursor,
/* c0 magenta */ 0xFF, 0x00, 0xFF,
/* c1 yellow */ 0xFF, 0xFF, 0x00,
@@ -913,14 +420,14 @@ MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
, atom_writes(R_PrimCursor, R_FaceCursor)
) {
mac_load_tri_indices(R_FaceCursor, R_T0, R_T1, R_T2),
mac_load_tri_indices( R_FaceCursor, R_T0, R_T1, R_T2),
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
gte_cmdw_nclip,
/* Culling (Branch forward if Backface) */
gte_mv_from_data_r(R_T0, C2_MAC0),
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
/* Format Primitive */
mac_gte_store_f3(R_PrimCursor),
@@ -932,7 +439,7 @@ MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
mac_format_f3_color(R_PrimCursor, 0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_F3)), /* Insert into Ordering Table Linked List */
mac_insert_ot_tag_f3(R_OtBase, R_PrimCursor), /* Insert into Ordering Table Linked List */
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
// Note(Ed): No bounds checking, should be checked before atom runs.
// end: branch(bounds_chk)
+110 -312
View File
@@ -1,7 +1,7 @@
#pragma region Vendors
#include <stdio.h>
#include <stdlib.h>
// #include <assert.h>
#include <assert.h>
// #include "libgpu.h"
// #include "libetc.h"
// #include "libgte.h"
@@ -26,12 +26,10 @@
#include "duffle/dsl.atom.h"
#include "duffle/lottes_tape.h"
#include "duffle/bios.h"
#include "duffle/psyq.h"
#pragma endregion Duffle Headers
#pragma region Duffle TUs
#include "duffle/pad.c"
#include "duffle/math.atom.c"
#include "duffle/mips.atom.c"
#include "duffle/gte.atom.c"
@@ -43,7 +41,6 @@
#pragma region Hello Camera Headers
# include "gen/macs.h"
# include "gen/offsets.h"
# include "gen/auto_reg.h"
#include "hello_camera.h"
#pragma endregion Hello Camera Headers
@@ -52,16 +49,9 @@
#include "hello_camera.atom.c"
#pragma endregion Hello Joypad TUs
enum {
Scratchpad_Loc = 0x1F800000,
};
#define C_scratch(type) C_(type, Scratchpad_Loc)
enum {
Scratchpad_Len = 1024,
MemTape_Len = 512,
ResolveLookAtArena_Words = 1024,
ResolveLookAtArena_Size = ResolveLookAtArena_Words * S_(MipsCode),
enum {
Scratchpad_Len = 1024,
MemTape_Len = 512,
};
typedef Struct_(SMemory) {
PrimitiveArena primitives;
@@ -71,10 +61,7 @@ typedef Struct_(SMemory) {
U4 MemTape[MemTape_Len];
MT3_S2S4 tform_world;
MT3_S2S4 tform_view;
Camera cam;
M3_S2 tform_world;
Ent_Cube cube;
Ent_Floor floor;
@@ -82,18 +69,11 @@ typedef Struct_(SMemory) {
PadBiosRaw pad_raw[2];
PadState pad[2];
// TODO(Ed): We don't need this we can just cast at any point an address to a desired view of scratchpad, we have the address.
U4_V scratchpad; // d-cache
U1 resolve_look_at_mem[ResolveLookAtArena_Size];
MipsAtom* resolve_look_at_atom_addrs[10];
};
global SMemory smem;
extern SMemory smem;
#define pad0_btn_(btn) btn & smem.pad[0].buttons
#define pad1_btn_(btn) btn & smem.pad[1].buttons
I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
gknown PrimitiveArena* pa = & smem.primitives;
gknown B1* buf = (B1*) r_(smem.primitives.buf)[smem.active_buf_id];
@@ -104,230 +84,75 @@ I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
}
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
I_ void resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4* up_in) {
// RGA(Lengyel): Build matrix expansion of a rigid transformation. Corresponding motor is not constructed; we write the LA form for GTE.
// Preconditions: eye != target, up_in not collinear with (target - eye).
V3_S4 right, up, forward;
V3_S4 ux, uy, uz;
V3_S4 pos, off;
forward = target[0]; sub_v3s4(& forward, eye[0]); // RGA(Lengyel): Affine point - point = zero-weight direction.
normalize_v3s4(& forward, & uz); // RGA(Lengyel): Normalize the direction bulk. Not finite-point unitization.
cross_v3s4(& uz, up_in, & right); normalize_v3s4(& right, & ux); // RGA(Lengyel): Complement(Wedge(forward, up_in)) -> right axis.
cross_v3s4(& uz, & ux, & up); normalize_v3s4(& up, & uy); // RGA(Lengyel): Complement(Wedge(forward, right)) -> up axis.
// RGA(Lengyel): matrix expansion of the world-to-camera rotation (basis rows).
look_at->m[0][0] = ux.x; look_at->m[0][1] = ux.y; look_at->m[0][2] = ux.z;
look_at->m[1][0] = uy.x; look_at->m[1][1] = uy.y; look_at->m[1][2] = uy.z;
look_at->m[2][0] = uz.x; look_at->m[2][1] = uz.y; look_at->m[2][2] = uz.z;
pos = eye[0]; mul_v3s4(& pos, v3s4(-1,-1,-1)); // RGA(Lengyel): -eye in world coordinates (spatial bulk only; implicit weight is dropped).
// RGA(Lengyel): R * (-eye) is the full matrix translation column.
// Motor translator would store half this displacement in m.xyz; GTE consumes full column.
mul_m3s2_v3s4(look_at, & pos, & off);
trans_m3s2( look_at, & off);
}
FI_ void camera_look_at_c11(Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at_c11(& c->look_at, & c->pos, target, up_in); }
/* Pre-build all 7 chain atoms of the resolve_look_at bundle into the static arena.
* 4 unique procs in hello_camera.atom.c (chain atoms 0, 2, 4, 6); atoms 1, 3, 5
* share the GENERIC normalize_v3s4_proc from gte.atom.c
* 0: resolve_look_at__input_and_sub_proc
* 1: normalize_v3s4_proc (fwd → uz; offsets 0, 16)
* 2: resolve_look_at__cross_uz_up_in_to_right_proc
* 3: normalize_v3s4_proc (right → ux; offsets 32, 48)
* 4: resolve_look_at__cross_uz_ux_to_up_proc
* 5: normalize_v3s4_proc (up → uy; offsets 64, 80)
* 6: resolve_look_at__populate_and_translate_proc
*/
internal void resolve_look_at_init(void) {
/* Wrap the static arena in a MipsAtomBuilder. */
AtomArena ab = atomarena_make(slice_ut_arr(smem.resolve_look_at_mem));
TapeBuilder tb = tb_make(slice_ut_arr(smem.resolve_look_at_atom_addrs));
U4 pin_mask = regfile_abi_mask | (1 << R_ResolveScratch);
RegFile rf = regfile(pin_mask);
U4 r_target_ptr = regfile_alloc(& rf);
U4 r_eye_ptr = regfile_alloc(& rf);
U4 r_up_in_ptr = regfile_alloc(& rf);
U4 r_tmp0 = regfile_alloc(& rf);
U4 r_tmp1 = regfile_alloc(& rf);
U4 r_tmp2 = regfile_alloc(& rf);
U4 r_tmp3 = regfile_alloc(& rf);
smem.resolve_look_at_atom_addrs[0] = resolve_look_at__input_and_sub_proc(& ab,
R_ResolveScratch,
r_target_ptr, r_eye_ptr, r_up_in_ptr,
r_tmp0, r_tmp1, r_tmp2, r_tmp3);
/* === ATOM 1: normalize fwd→uz === */
U2 src_offset = O_(ResolveLookAtScratch, fwd);
U2 dst_offset = O_(ResolveLookAtScratch, uz);
smem.resolve_look_at_atom_addrs[1] = normalize_v3s4_proc(& ab,
src_offset, dst_offset, RegUse_(normalize_v3s4_proc){
.scratch = R_ResolveScratch,
.src_ptr = R_T0,
.dst_ptr = R_T1,
.recip_est = R_T6,
.norm = R_T7,
.shift = R_V0,
.src_x = R_T2,
.t3 = R_T3,
.t4 = R_T5,
.t5 = R_V1,
});
/* === ATOM 2: cross uz×up_in→right === */
U4 r_a_2 = R_T0;
U4 r_b_2 = R_T1;
U4 r_c_2 = R_T2;
U4 r_d_2 = R_T3;
U4 r_f_2 = R_T5; /* out ptr (HARDCODED in body: scratch+32) */
U4 r_g_2 = R_T6; /* a ptr = scratch+16 */
U4 r_h_2 = R_T7; /* b ptr = scratch+128 */
smem.resolve_look_at_atom_addrs[2] = resolve_look_at__cross_uz_up_in_to_right_proc(& ab,
R_ResolveScratch,
r_a_2, r_b_2, r_c_2, r_d_2, r_f_2, r_g_2, r_h_2);
/* === ATOM 3: normalize right→ux === */
src_offset = O_(ResolveLookAtScratch, right);
dst_offset = O_(ResolveLookAtScratch, ux);
smem.resolve_look_at_atom_addrs[3] = normalize_v3s4_proc(& ab,
src_offset, dst_offset, RegUse_(normalize_v3s4_proc){
.scratch = R_ResolveScratch,
.src_ptr = R_T0,
.dst_ptr = R_T1,
.recip_est = R_T6,
.norm = R_T7,
.shift = R_V0,
.src_x = R_T2,
.t3 = R_T3,
.t4 = R_T5,
.t5 = R_V1,
});
/* === ATOM 4: cross uz×ux→up === */
U4 r_a_4 = R_T0;
U4 r_b_4 = R_T1;
U4 r_c_4 = R_T2;
U4 r_d_4 = R_T3;
U4 r_f_4 = R_T5; /* out ptr (HARDCODED: scratch+64) */
U4 r_g_4 = R_T6; /* a ptr = scratch+16 */
U4 r_h_4 = R_T7; /* b ptr = scratch+48 */
smem.resolve_look_at_atom_addrs[4] = resolve_look_at__cross_uz_ux_to_up_proc(& ab,
R_ResolveScratch,
r_a_4, r_b_4, r_c_4, r_d_4, r_f_4, r_g_4, r_h_4);
/* === ATOM 5: normalize up→uy === */
src_offset = O_(ResolveLookAtScratch, up);
dst_offset = O_(ResolveLookAtScratch, uy);
smem.resolve_look_at_atom_addrs[5] = normalize_v3s4_proc(& ab,
src_offset, dst_offset,
RegUse_(normalize_v3s4_proc){
.scratch = R_ResolveScratch,
.src_ptr = R_T0,
.dst_ptr = R_T1,
.recip_est = R_T6,
.norm = R_T7,
.shift = R_V0,
.src_x = R_T2,
.t3 = R_T3,
.t4 = R_T5,
.t5 = R_V1,
});
/* === ATOM 6a: populate (m[][] from ux/uy/uz, t[]=0) === */
U4 r_look_at_6a = R_T0; /* tape pop → look_at* */
U4 r_scratch_6a = R_ResolveScratch;
U4 r_pux_6a = R_T1;
U4 r_puy_6a = R_T3;
U4 r_puz_6a = R_T5;
U4 r_tmp0_6a = R_T2;
U4 r_tmp1_6a = R_T6;
U4 r_tmp2_6a = R_V0;
smem.resolve_look_at_atom_addrs[6] = resolve_look_at__populate_proc(& ab,
r_look_at_6a, r_scratch_6a,
r_pux_6a, r_puy_6a, r_puz_6a,
r_tmp0_6a, r_tmp1_6a, r_tmp2_6a);
/* === ATOM 6a.5: set_gte_mt3s2s4 (BAKED — ctc2 RT matrix) ===
* This is a BAKED atom from gte.atom.c. Its body hardcodes R_T3 as
* the matrix pointer (popped from tape). It does NOT need GPR
* assignment from us — it has its own internal GPR usage.
* We just take its address. */
smem.resolve_look_at_atom_addrs[7] = (MipsAtom*) & set_gte_mt3s2s4;
/* === ATOM 6b: matrix_vector (RT * (-eye) >> 12) ===
* Uses mac_apply_matrix_lv component macro which internally uses
* r_t0 for the RT matrix load + V0 load, then r_t0/r_t1/r_t2
* for the mfc2/store. We pass our GPRs. */
U4 r_scratch_6b = R_ResolveScratch;
U4 r_peye_6b = R_T1; /* scratch+96 (packed V0 dst, then off dst) */
U4 r_look_at_6b = R_T0; /* tape pop → look_at* */
U4 r_tmp0_6b = R_T2;
U4 r_tmp1_6b = R_T3;
U4 r_tmp2_6b = R_T5;
smem.resolve_look_at_atom_addrs[8] = resolve_look_at__matrix_vector_proc(& ab,
r_scratch_6b, r_peye_6b, r_look_at_6b,
r_tmp0_6b, r_tmp1_6b, r_tmp2_6b);
/* === ATOM 6c: trans_matrix (off → look_at->t[]) === */
U4 r_look_at_6c = R_T0; /* tape pop → look_at* */
U4 r_scratch_6c = R_ResolveScratch;
U4 r_off_ptr_6c = R_T1; /* &scratch.eye (= off dst) */
U4 r_tmp0_6c = R_T2;
smem.resolve_look_at_atom_addrs[9] = resolve_look_at__trans_matrix_proc(& ab,
r_look_at_6c, r_scratch_6c, r_off_ptr_6c, r_tmp0_6c, R_T3, R_T4);
/* Sanity check: arena didn't overflow. */
assert(ab.used <= ResolveLookAtArena_Size);
}
/* Emit the resolve_look_at bundle into the tape. Called once per frame from update().
* The 7 chain atoms are pre-built at init time (resolve_look_at_init) and referenced by address via smem.resolve_look_at_atom_addrs[].
* Per-frame work: 7 tb_emit (atom pointer emissions) + 5 tb_data (C-side pointers for atom 0 + look_at for atom 6).
/* Uses ONE 8-byte frame allocated via the compiler's standard prologue.
* The 4 wasted-arg words for B(12h) InitPAD2 live at [SP+0..15] but are not explicitly allocated.
* The compiler handles the MIPS O32 "wasted stack" convention for us by treating the B-call as a 4-arg call.
*
* Binds_ contract (the field-name labels are for human readability):
* Atom 0 input_and_sub target(4) eye(4) up_in(4) scratch_base(4) = 4 words
* Atoms 1-5 (no tape data — atom uses r_scratch + offset internally)
* Atom 6 populate_and_translate look_at(4) = 1 word
* ----
* 5 tb_data words total per frame.
*/
I_ void resolve_look_at(
TapeBuilder_R tb
, MT3_S2S4* look_at
, P3_S4* eye
, P3_S4* target
, V3_S4* up_in
){
tb_emit(tb, smem.resolve_look_at_atom_addrs[0]); {
tb_data(tb, u4_(target));
tb_data(tb, u4_(eye));
tb_data(tb, u4_(up_in));
tb_data(tb, u4_(smem.scratchpad));
}
* The buffer pointers are passed as arguments so the compiler keeps them in callee-saved registers;
* The B(12h) asm volatile block does NOT clobber those registers (it clobbers only the volatile GPRs + the B-table arg registers explicitly).
* The C-level writes after the call re-load the pointers from their callee-saved homes.
*
* The clobber list for both B-calls names the full BIOS destroy set documented in kernelbios.md:167-174 (R1..R15, R24..R25, R31, HI/LO).
* The kernel-ABI "volatile GPRs" subset is clb_system; the rest of the destroy set is enumerated explicitly here. */
NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
{
/* Pin raw0 + raw1 to $a0 + $a1 via rgcc; the B(12h) call uses these directly.
* The `(void)` casts mark them as unread after the call so the compiler doesn't need to move them back. */
register PadBiosRaw* p0 rgcc(R_A0) = raw0;
register PadBiosRaw* p1 rgcc(R_A1) = raw1;
(void)p0; (void)p1;
tb_emit(tb, smem.resolve_look_at_atom_addrs[1]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[2]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[3]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[4]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[5]); { }
// TODO(Ed): Properly annotate the raw values in the inline asm instructions.
// Use enums.
tb_emit(tb, smem.resolve_look_at_atom_addrs[6]); {
tb_data(tb, u4_(look_at));
}
tb_emit(tb, smem.resolve_look_at_atom_addrs[7]); {
tb_data(tb, u4_(look_at));
}
tb_emit(tb, smem.resolve_look_at_atom_addrs[8]); {
tb_data(tb, u4_(look_at));
}
tb_emit(tb, smem.resolve_look_at_atom_addrs[9]); {
// tb_data(tb, u4_(look_at));
}
/* B(12h) InitPAD2(raw0, 0x22, raw1, 0x22)
* $a0 = raw0 (rgcc-bound; survives the sequence below)
* $a1 = raw1 (preserved into $a2 before $a1 is overwritten)
* $a2 = raw1 (moved from $a1; survives $a1's overwrite)
* $a3 = 0x22 (immediate)
* $t1 = 0x12 (function number)
* $t2 = 0xB0 (BIOS B-table address) */
asm volatile(
asm_words(
or_u( rarg_2, rarg_1, rdiscard), /* $a2 = $a1 = raw1 */
add_ui( rarg_1, rdiscard, 0x22), /* $a1 = 0x22 */
add_ui( rarg_3, rdiscard, 0x22), /* $a3 = 0x22 */
add_ui( rtmp_1, rdiscard, 0x12), /* $t1 = 0x12 */
add_ui( rtmp_2, rdiscard, 0xB0), /* $t2 = 0xB0 */
call_reg(rtmp_2), /* jalr $t2, $ra */
nop /* BD slot */
)
asm_rpins, r_use(p0), r_use(p1)
asm_clobber:
rlit(R_AT),
rlit(R_V0), rlit(R_V1),
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
rlit(R_RA),
clb_mem_drain
);
/* The C-level writes re-load the pointers via the parameter names and write 0xFF to each
* buffer's status byte to mark the initial-state hazard documented in kernelbios.md:1621-1624. */
u1_v(raw0)[0] = 0xFF;
u1_v(raw1)[0] = 0xFF;
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */
asm volatile(
asm_words(
add_ui( rtmp_1, rdiscard, 0x13), /* $t1 = 0x13 */
add_ui( rtmp_2, rdiscard, 0xB0), /* $t2 = 0xB0 (re-load) */
call_reg(rtmp_2), /* jalr $t2, $ra */
nop /* BD slot */
)
asm_clobber:
rlit(R_AT),
rlit(R_V0), rlit(R_V1),
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
rlit(R_RA),
clb_mem_drain
);
}
GCC_OPTIMIZATION_DISABLE
@@ -335,25 +160,21 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
{
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
// Pad Input
if (1) // Pad Input
{
tb.used = 0; tb_scope_run(& tb) {
// Grab latest state from bios.
/* BIOS-owned polling: per-frame snapshot of both ports. */
tb_emit_(pad_bios_snapshot);
tb_data_(raw, & smem.pad_raw[0]);
tb_data_(raw, & smem.pad_raw[0]);
tb_data_(state, & smem.pad[0]);
// tb_emit_(pad_bios_snapshot);
// tb_data_(raw, & smem.pad_raw[1]);
// tb_data_(state, & smem.pad[1]);
tb_emit_(pad_input_cam);
tb_data_(state, & smem.pad[0]);
tb_data_(cam, & smem.cam);
// tb_emit_(pad_input_cube_rotation);
// tb_data_(state, & smem.pad[0]);
// tb_data_(cube_rot, & smem.cube.rot);
// tb_data_(floor_rot, & smem.floor.rot);
tb_emit_(pad_bios_snapshot);
tb_data_(raw, & smem.pad_raw[1]);
tb_data_(state, & smem.pad[1]);
/* Per-frame rotation apply: consume pad[0].buttons + pad[0].left_x */
tb_emit_(pad_apply_input);
tb_data_(state, & smem.pad[0]);
tb_data_(cube_rot, & smem.cube.rot);
tb_data_(floor_rot, & smem.floor.rot);
}
}
@@ -380,31 +201,15 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
A2_S2 p; //???
S4 flag; //????
B4 use_c11_path = false;
if (use_c11_path) {
camera_look_at_c11(& smem.cam, & smem.cube.pos, & v3s4(0, -fp_one, 0));
}
if (use_c11_path == false)
{
tb.used = 0; tb_scope_run(& tb) {
resolve_look_at(& tb, & smem.cam.look_at, & smem.cam.pos, & smem.cube.pos, & v3s4(0, -fp_one, 0));
}
}
// Draw cube
if (1)
{
mt3s2s4_rotation (& smem.cube.rot, & smem.tform_world);
mt3s2s4_translation(& smem.tform_world, & smem.cube.pos);
mt3s2s4_scale (& smem.tform_world, & smem.cube.scale);
// Combine world and look_at matrix.
gte_comp_coord_m3s2(& smem.cam.look_at, & smem.tform_world, & smem.tform_view);
gte_matrix_set_rotation (& smem.tform_view);
gte_matrix_set_translation(& smem.tform_view);
// gte_matrix_set_rotation (& smem.tform_world);
// gte_matrix_set_translation(& smem.tform_world);
m3s2_rotation (& smem.cube.rot, & smem.tform_world);
m3s2_translation(& smem.tform_world, & smem.cube.pos);
m3s2_scale (& smem.tform_world, & smem.cube.scale);
gte_matrix_set_rotation (& smem.tform_world);
gte_matrix_set_translation(& smem.tform_world);
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
U4 prim_cursor = prim_base + pa->used;
@@ -425,22 +230,16 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
tb_data(& tb, u4_(& pa->used));
tb_data(& tb, prim_base);
}
tape_run_a02_s07(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
tape_run(tb_slice(tb));
// smem.cube.rot.y += 30;
}
// Draw floor
if (1)
{
mt3s2s4_rotation (& smem.floor.rot, & smem.tform_world);
mt3s2s4_translation(& smem.tform_world, & smem.floor.pos);
mt3s2s4_scale (& smem.tform_world, & smem.floor.scale);
// Combine world and look_at matrix.
gte_comp_coord_m3s2(& smem.cam.look_at, & smem.tform_world, & smem.tform_view);
gte_matrix_set_rotation (& smem.tform_view);
gte_matrix_set_translation(& smem.tform_view);
m3s2_rotation (& smem.floor.rot, & smem.tform_world);
m3s2_translation(& smem.tform_world, & smem.floor.pos);
m3s2_scale (& smem.tform_world, & smem.floor.scale);
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
U4 prim_cursor = prim_base + pa->used;
@@ -450,11 +249,11 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
// Prepare the tape. (Push protocol to tape)
tb.used = 0; tb_scope(& tb) {
// tb_emit(& tb, set_gte_mt3s2s4);
// tb_data(& tb, u4_(& smem.tform_view));
tb_emit(& tb, set_gte_world);
tb_data(& tb, u4_(& smem.tform_world));
tb_emit(& tb, rbind_floor_f3_face);
// TODO(Ed): Just use a single context struct ref?
// TODO(Ed): Just use a single context struct ref
tb_data(& tb, prim_cursor);
tb_data(& tb, u4_(smem.floor.faces));
tb_data(& tb, u4_(smem.floor.verts));
@@ -467,7 +266,7 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
tb_data(& tb, u4_(& pa->used));
tb_data(& tb, prim_base);
}
tape_run_a02_s07(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
tape_run(tb_slice(tb));// Fire off the tape.
// C-side state (pa->used) has already been updated by the tape!
// smem.floor.rot.y += 5;
@@ -491,14 +290,25 @@ void gp_display_frame(DoubleBuffer* screen_buf, S4* active_buf_id, U4* ordering_
}
GCC_OPTIMIZATION_DISABLE
void hot_reload_entry(void)
{
smem.primitives.used = 0;
while (1) {
gknown S4* active_buf_id = & smem.active_buf_id;
gknown U4* ordering_buf = r_(smem.ordering_tbl)[active_buf_id[0]];
gknown PrimitiveArena* pa = & smem.primitives;
update(pa, ordering_buf);
render();
gp_display_frame(& smem.screen_buf, active_buf_id, ordering_buf, pa);
}
}
int main(void)
{
smem = (SMemory){0};
// TODO(Ed): remove this field we don't need it in smem.
smem.scratchpad = C_(U4_V, Scratchpad_Loc);
smem.scratchpad = C_(U4_V, 0x1F800000);
// smem.primitives.used = 0;
// smem.active_buf_id = 0;
smem.cam.pos = v3s4(500, -1000, -1500);
/*Persistent Entity Setup*/{
ent_cube128_init(& smem.cube.verts, & smem.cube.faces); {
Ent_Cube* cube = & smem.cube;
@@ -518,10 +328,6 @@ int main(void)
reset_graph(0);
/* Direct BIOS: poll both ports during VBlank. */
pad_bios_init_start(& smem.pad_raw[0], & smem.pad_raw[1]);
/* Pre-build the resolve_look_at bundle atoms into the static arena. */
resolve_look_at_init();
/* Pinned registers for the GPU init atom. */
register U4* io_base_addr rgcc(R_IO_BaseAddr) = u4_r(IO_BASE_ADDR);
register DoubleBuffer* screen_buf rgcc(R_ScreenBuf) = & smem.screen_buf;
@@ -530,15 +336,7 @@ int main(void)
tb_emit(& tb, gp_screen_init);
}
}
while (1) {
gknown S4* active_buf_id = & smem.active_buf_id;
gknown U4* ordering_buf = r_(smem.ordering_tbl)[active_buf_id[0]];
gknown PrimitiveArena* pa = & smem.primitives;
update(pa, ordering_buf);
render();
gp_display_frame(& smem.screen_buf, active_buf_id, ordering_buf, pa);
};
hot_reload_entry();
return 0;
}
GCC_OPTIMIZATION_ENABLE
+8 -8
View File
@@ -21,6 +21,12 @@ enum {
ScreenRes_CenterY = (ScreenRes_Y >> 1),
};
enum {
fp_one = (1 << 12),
};
#define v3s4_fp_one() v3s4(fp_one, fp_one, fp_one)
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
typedef Array_(OrderingTable_Buffer, 2);
@@ -61,7 +67,7 @@ I_ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
typedef Struct_(Ent_Cube) {
V3_S4 accel;
V3_S4 vel;
V3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
V3_S4 pos;
V3_S4 scale;
V3_S2 rot;
A8_V3_S2 verts;
@@ -88,15 +94,9 @@ I_ void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
};
typedef Struct_(Ent_Floor) {
V3_S4 accel;
V3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
V3_S4 pos;
V3_S4 scale;
V3_S2 rot;
A4_V3_S2 verts;
A2_V3_S2 faces;
};
typedef Struct_(Camera) {
P3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
V3_S2 rot;
MT3_S2S4 look_at;
};
+10 -10
View File
@@ -24,8 +24,8 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c);
#pragma region MACs (Mips Atom components)
FI_ Slice_MipsCode ac_put_disp_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ab, {
FI_ Slice_MipsCode ac_put_disp_env(U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_put_disp_env, {
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
@@ -35,8 +35,8 @@ MipsAtomComp_Proc_(ab, {
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
})
FI_ Slice_MipsCode ac_put_draw_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ab, {
FI_ Slice_MipsCode ac_put_draw_env(U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_put_draw_env, {
/*
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
* References:
@@ -116,7 +116,7 @@ internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,1)),
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + OA_(DoubleBuffer,draw,0)), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
mac_store_v2s2(R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + OA_(DoubleBuffer,draw,0)), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
mac_store_v2s2( R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + OA_(DoubleBuffer,draw,0)), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + OA_(DoubleBuffer,draw,1)),
@@ -129,7 +129,7 @@ internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,1)),
/* draw[0].texture_page = 10 (gp0_tpage_default). C11 SetDefDrawEnv at C11_only.elf:0x8001273C writes the same 0x0A. . */
add_ui(R_T0, R_0, gp0_tpage_default),
add_ui(R_T0, R_0, gp0_tpage_default),
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,0)),
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,1)),
@@ -144,7 +144,7 @@ internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,1)),
/* draw[0].initial_bg_color = (r=7, g=7, b=7). */
add_ui(R_T0, R_0, 7),
add_ui(R_T0, R_0, 7),
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,0)),
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,1)),
@@ -228,7 +228,7 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
/* BD-slot: write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
* harmless because the OT entry that points to this prim is created later, only on the body path. */
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
@@ -286,14 +286,14 @@ MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
, atom_writes(R_PrimCursor, R_FaceCursor)
) {
mac_load_tri_indices(R_FaceCursor, R_T0, R_T1, R_T2),
mac_load_tri_indices( R_FaceCursor, R_T0, R_T1, R_T2),
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
gte_cmdw_nclip,
/* Culling (Branch forward if Backface) */
gte_mv_from_data_r(R_T0, C2_MAC0),
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
/* Format Primitive */
mac_gte_store_f3(R_PrimCursor),
+2 -2
View File
@@ -24,8 +24,8 @@
* Emits 9 instructions (status/buttons/axes/attempt stores plus the
* two-instruction zero-extended buttons load).
*/
FI_ Slice_MipsCode ac_pad_sio_write_pad_state(MipsAtomBuilder_R ab, U4 status_val, U4 state_ptr_reg, U4 scratch_reg)
MipsAtomComp_Proc_(ac_pad_sio_write_pad_state, ab, {
FI_ Slice_MipsCode ac_pad_sio_write_pad_state(U4 status_val, U4 state_ptr_reg, U4 scratch_reg)
MipsAtomComp_Proc_(ac_pad_sio_write_pad_state, {
add_ui(scratch_reg, R_0, status_val),
store_word(scratch_reg, state_ptr_reg, O_(PadState,status)),
/* FIX 2026-08-02: buttons = 0x0000FFFF = "no buttons pressed" in
-10625
View File
File diff suppressed because one or more lines are too long
+271 -52
View File
@@ -1,3 +1,13 @@
# --- Parameter Surface (Task 8) -----------------------------------------
# -Reload : After a successful build, invoke reload.ps1 as a child pwsh and propagate its exit code.
# -HelperZipOnly : Skip the build entirely; regenerate the helper zip and exit. Honors -HelperZipOutput for out-of-tree paths.
# -HelperZipOutput: When -HelperZipOnly is set, writes the archive to this path instead of the scripts/pcsx_debug_helper.zip.
param(
[switch]$Reload,
[switch]$HelperZipOnly,
[string]$HelperZipOutput = ''
)
$path_root = split-path -Path $PSScriptRoot -Parent
$path_build = join-path $path_root 'build'
$path_code = join-path $path_root 'code'
@@ -8,6 +18,98 @@ if ((test-path $path_build) -eq $false) {
new-item -itemtype directory -path $path_build
}
# --- HelperZipOnly short-circuit ----------------------------------------
# Must run before any compile/link work.
# Inlines the same logic as Make-HelperZip below to avoid an extra pwsh process spawn (~200 ms).
#The helper zip is small and the BCL call is in-process; cold ~14 ms, warm ~10 ms (assembly load + tiny zip write).
if ($HelperZipOnly) {
$zipDest = if ([string]::IsNullOrEmpty($HelperZipOutput)) {
join-path $path_scripts 'pcsx_debug_helper.zip'
}
else {
$HelperZipOutput
}
$HelperDir = join-path $path_scripts 'pcsx_debug_helper'
$elf32Src = join-path $path_scripts 'elf32.lua'
$elf32Dest = join-path $HelperDir 'elf32.lua'
if (-not (test-path -LiteralPath $HelperDir)) {
write-error "helper dir not found: $HelperDir"
exit 1
}
if (-not (test-path -LiteralPath $elf32Src)) {
write-error "elf32.lua not found at $elf32Src"
exit 1
}
write-host "[build] HelperZipOnly mode -> $zipDest"
# --- Timestamp gate (Fix 1) -------------------------------------------
# PCSX-Redux holds pcsx_debug_helper.zip open via -archive at startup.
# The zip is consumed once at startup; the reload endpoint reads it
# from package.loaded on subsequent calls. Writing it on every build
# is dead work that fights the file lock. Skip the rewrite when the
# three sources (autoexec.lua, reload.lua, elf32.lua) are all older
# than the existing zip.
$sources = @(
(join-path $HelperDir 'autoexec.lua'),
(join-path $HelperDir 'reload.lua'),
$elf32Src
)
$zipMtime = $null
if (test-path -LiteralPath $zipDest) {
$zipMtime = (Get-Item -LiteralPath $zipDest).LastWriteTime
}
$needsRewrite = $false
if ($null -eq $zipMtime) {
$needsRewrite = $true
}
else {
foreach ($s in $sources) {
if (-not (test-path -LiteralPath $s)) { continue }
if ((Get-Item -LiteralPath $s).LastWriteTime -gt $zipMtime) {
$needsRewrite = $true
break
}
}
}
if (-not $needsRewrite) {
$sz = (Get-Item -LiteralPath $zipDest).Length
Write-Host "[build] helper zip up to date: $zipDest ($sz bytes); skipping"
return
}
Copy-Item -LiteralPath $elf32Src -Destination $elf32Dest -Force
try {
# Force the inode release so CreateFromDirectory can write fresh.
# ZipFile.CreateFromDirectory throws if the destination exists.
# If PCSX-Redux holds the file open, Remove-Item raises — fall
# back to writing pcsx_debug_helper.zip.new alongside. The next
# PCSX-Redux restart will read the canonical path; the .new file
# is a hint for the optional launch-script patch in fix 3.
if (test-path -LiteralPath $zipDest) {
try {
# -ErrorAction Stop is required so the catch below fires.
# Remove-Item raises a non-terminating error by default
# (ErrorActionPreference=Continue), which bypasses catch.
Remove-Item -LiteralPath $zipDest -Force -ErrorAction Stop
}
catch {
$zipDest = [System.IO.Path]::ChangeExtension($zipDest, '.zip.new')
Write-Warning "[build] canonical helper zip is locked; writing to $zipDest instead"
}
}
Add-Type -AssemblyName System.IO.Compression.FileSystem
[System.IO.Compression.ZipFile]::CreateFromDirectory(
$HelperDir, $zipDest,
[System.IO.Compression.CompressionLevel]::Optimal, $false) | Out-Null
$sz = (Get-Item -LiteralPath $zipDest).Length
Write-Host "[build] wrote $sz bytes to $zipDest"
}
finally {
if (test-path -LiteralPath $elf32Dest) { Remove-Item -LiteralPath $elf32Dest -Force }
}
return
}
# --- Toolchain Definition ---
# Assumes 'mipsel-none-elf' toolchain is in your system's PATH.
$Prefix = "mipsel-none-elf"
@@ -180,9 +282,12 @@ function link-modules { param([string[]]$link_modules, [string] $elf, [string[]
$link_args += ($f_link_pass_through_prefix + $f_link_mapfile + $map)
$link_args += ($f_link_pass_through_prefix + $f_link_start_group)
# 16 removed entries (c2, card, cd, comb, ds, gs, gun, hmd, math, mcrd, mcx, press, sio, snd, spu, tap)
# had LOAD lines in the map but ZERO .o files pulled in — they were unused.
# 5 kept libraries (api, c, etc, gpu, gte) are required by the C-side calls in hello_joypad.c (reset_graph, draw_sync, vsync, etc.).
# raw_sio_pad_poll_20260802 — Task 5.1c surgical library-list trim.
# The 16 removed entries (c2, card, cd, comb, ds, gs, gun, hmd, math,
# mcrd, mcx, press, sio, snd, spu, tap) had LOAD lines in the map but
# ZERO .o files pulled in — they were unused. The 5 kept libraries
# (api, c, etc, gpu, gte) are required by the C-side calls in
# hello_joypad.c (reset_graph, draw_sync, vsync, etc.).
$libraries = @(
"api",
"c",
@@ -211,32 +316,68 @@ function link-modules { param([string[]]$link_modules, [string] $elf, [string[]
function make-binary { param([string]$elf, [string]$exe)
Write-Host "--- Creating Binary ---" -ForegroundColor Cyan
write-host "Converting $elf to PS-EXE -> '$exe'"
$objcopy_args = ($f_objcopy_format + "binary"), $elf, $exe
$objcopy_args = ($f_objcopy_format + "binary"), $elf, $exe
& $Objcopy $objcopy_args
if ($LASTEXITCODE -ne 0) { Write-Error "Objcopy failed. Aborting."; exit 1 }
}
function ps1-meta { param(
[string]$unity_root,
[string] $unity_root,
[string[]]$sources,
[Parameter(Mandatory=$true)][string]$metadata,
[string]$out_root = (join-path $path_build 'gen'),
[string[]]$passes = @('--pre-link'),
[string] $out_root = (join-path $path_build 'gen'),
[string[]]$passes = @('--pre-link'),
[string[]]$extra_args = @()
)
# `--unity-root` and `--source` are mutually exclusive. Exactly one of `$unity_root` / `$sources` must be supplied; the other must be absent.
# `--unity-root` and `--source` are
# mutually exclusive. Exactly one of `$unity_root` / `$sources` must
# be supplied; the other must be absent.
if ($null -ne $unity_root -and $unity_root -ne '')
{
if ($null -ne $sources -and $sources.Count -gt 0) {
write-error 'ps1-meta: -unity_root and -sources are mutually exclusive'
exit 2
}
}
}
elseif ($null -eq $sources -or $sources.Count -eq 0) {
write-error 'ps1-meta: either -unity_root <file> or -sources <file...> is required'
exit 2
}
# --- Defensive attribute clear on tracked gen files ------------------------
# Git tracks code/<dir>/gen/*.h files and Windows keeps the Archive bit set
# on them. Combined with transient editor locks or co-running processes,
# this can make io.open(path, "wb") fail with Access Denied / Sharing
# Violation even though Get-ChildItem shows IsReadOnly = False. Clearing
# the Read-only + Archive bits locally is safe; git re-asserts them on
# the next operation but the metaprogram write always wins.
#
# Derived from the caller's parameters: $metadata lives in $path_duffle
# (so its parent is the duffle dir), and $unity_root / $sources[0] lives
# in $path_module (so its parent is the module dir).
$pathToDuffle = split-path -Path $metadata -Parent
$pathToModule = $null
if ($null -ne $unity_root -and $unity_root -ne '') {
$pathToModule = split-path -Path $unity_root -Parent
}
elseif ($null -ne $sources -and $sources.Count -gt 0) {
$pathToModule = split-path -Path $sources[0] -Parent
}
$genFiles = @(
join-path $pathToDuffle 'gen\macs.h'
join-path $pathToDuffle 'gen\offsets.h'
)
if ($null -ne $pathToModule) {
$genFiles += join-path $pathToModule 'gen\macs.h'
$genFiles += join-path $pathToModule 'gen\offsets.h'
}
foreach ($f in $genFiles) {
if (test-path -LiteralPath $f) {
attrib -R $f 2>&1 | Out-Null
attrib -A $f 2>&1 | Out-Null
}
}
$script = join-path $path_scripts 'ps1_meta.lua'
$input_summary = if ($null -ne $unity_root -and $unity_root -ne '') {
"unity=$unity_root"
@@ -265,14 +406,14 @@ function inject-dwarf { param(
[string]$path_gen
)
$base_name = [System.IO.Path]::GetFileNameWithoutExtension($elf)
$path_dwarf_line_bin = join-path $path_gen "$base_name.dwarf_line.bin"
$path_dwarf_aranges_bin = join-path $path_gen "$base_name.dwarf_aranges.bin"
$path_dwarf_rnglists_bin = join-path $path_gen "$base_name.dwarf_rnglists.bin"
$path_dwarf_info_bin = join-path $path_gen "$base_name.dwarf_info.bin"
$path_dwarf_abbrev_bin = join-path $path_gen "$base_name.dwarf_abbrev.bin"
$path_dwarf_str_bin = join-path $path_gen "$base_name.dwarf_str.bin"
$path_dwarf_loc_bin = join-path $path_gen "$base_name.dwarf_loc.bin"
$path_dwarf_loclists_bin = join-path $path_gen "$base_name.dwarf_loclists.bin"
$path_dwarf_line_bin = join-path $path_gen "$base_name.dwarf_line.bin"
$path_dwarf_aranges_bin = join-path $path_gen "$base_name.dwarf_aranges.bin"
$path_dwarf_rnglists_bin = join-path $path_gen "$base_name.dwarf_rnglists.bin"
$path_dwarf_info_bin = join-path $path_gen "$base_name.dwarf_info.bin"
$path_dwarf_abbrev_bin = join-path $path_gen "$base_name.dwarf_abbrev.bin"
$path_dwarf_str_bin = join-path $path_gen "$base_name.dwarf_str.bin"
$path_dwarf_loc_bin = join-path $path_gen "$base_name.dwarf_loc.bin"
$path_dwarf_loclists_bin = join-path $path_gen "$base_name.dwarf_loclists.bin"
$path_inject_elf = join-path $path_build "$base_name.dwarf-injected.elf"
if (-not (Test-Path $path_dwarf_line_bin)) { return }
@@ -517,7 +658,7 @@ function build-hello_camera {
$path_build_gen = join-path $path_build 'gen'
$src_c = join-path $path_module 'hello_camera.c'
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--pre-link')
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen
$assemble_args = @()
$assemble_args += $f_debug
@@ -532,7 +673,6 @@ function build-hello_camera {
$compile_args = @()
$compile_args += $f_debug
$compile_args += ($f_define + 'BUILD_DEBUG')
$compile_args += $f_optimize_none
# $compile_args += $f_optimize_intrinsics
# $compile_args += $f_optimize_size
@@ -553,50 +693,129 @@ function build-hello_camera {
link-modules $link_modules $elf $link_args
make-binary $elf $exe
# Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start).
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
inject-dwarf $elf $path_build_gen
}
build-hello_camera
# NO idea if this works yet...
function Send-ToEmulator { param( [string]$exePath )
$uri = "http://localhost:8080/api/v1/load-exec"
# ── Helper-zip + reload helpers (Task 8) ──
# Defined right after the final build-hello_camera function so they're in scope for the post-build calls below.
# The Make-HelperZip function is also reused by the -HelperZipOnly short-circuit at the top of this script.
# Both call the in-process BCL CreateFromDirectory rather than spawning a child pwsh to avoid the ~200 ms process-spawn overhead.
function Make-HelperZip {
param([string]$OutputPath = '')
# Absolute path is safest for the emulator web server
$absolutePath = [System.IO.Path]::GetFullPath($exePath)
$dest = if ([string]::IsNullOrEmpty($OutputPath)) {
join-path $path_scripts 'pcsx_debug_helper.zip'
}
else {
$OutputPath
}
# Create JSON payload pointing to your compiled .ps-exe
$body = @{ filename = $absolutePath } | ConvertTo-Json
$HelperDir = join-path $path_scripts 'pcsx_debug_helper'
$elf32Src = join-path $path_scripts 'elf32.lua'
$elf32Dest = join-path $HelperDir 'elf32.lua'
if (-not (test-path -LiteralPath $HelperDir)) {
write-warning "[build] helper dir not found: $HelperDir; skipping helper zip"
return
}
if (-not (test-path -LiteralPath $elf32Src)) {
write-warning "[build] elf32.lua not found at $elf32Src; skipping helper zip"
return
}
Write-Host "Pushing hot-reload to PCSX-Redux..." -ForegroundColor Magenta
try {
$response = Invoke-RestMethod -Uri $uri -Method Post -Body $body -ContentType "application/json"
Write-Host "Hot-reload successful!" -ForegroundColor Green
} catch {
Write-Warning "Could not connect to PCSX-Redux web server. Ensure the emulator is running and Web Server is enabled."
}
# --- Timestamp gate (Fix 1) -------------------------------------------
# PCSX-Redux holds pcsx_debug_helper.zip open via -archive at startup.
# The zip is consumed once at startup; the reload endpoint reads it
# from package.loaded on subsequent calls. Writing it on every build
# is dead work that fights the file lock. Skip the rewrite when the
# three sources (autoexec.lua, reload.lua, elf32.lua) are all older
# than the existing zip.
$sources = @(
(join-path $HelperDir 'autoexec.lua'),
(join-path $HelperDir 'reload.lua'),
$elf32Src
)
$zipMtime = $null
if (test-path -LiteralPath $dest) {
$zipMtime = (Get-Item -LiteralPath $dest).LastWriteTime
}
$needsRewrite = $false
if ($null -eq $zipMtime) {
$needsRewrite = $true
}
else {
foreach ($s in $sources) {
if (-not (test-path -LiteralPath $s)) { continue }
if ((Get-Item -LiteralPath $s).LastWriteTime -gt $zipMtime) {
$needsRewrite = $true
break
}
}
}
if (-not $needsRewrite) {
$sz = (Get-Item -LiteralPath $dest).Length
Write-Host "[build] helper zip up to date: $dest ($sz bytes); skipping"
return
}
write-host "[build] regenerating helper zip -> $dest"
Copy-Item -LiteralPath $elf32Src -Destination $elf32Dest -Force
try {
# Force the inode release so CreateFromDirectory can write fresh.
# ZipFile.CreateFromDirectory throws if the destination exists.
# If PCSX-Redux holds the file open, Remove-Item raises — fall
# back to writing pcsx_debug_helper.zip.new alongside. The next
# PCSX-Redux restart will read the canonical path; the .new file
# is a hint for the optional launch-script patch in fix 3.
if (test-path -LiteralPath $dest) {
try {
# -ErrorAction Stop is required so the catch below fires.
# Remove-Item raises a non-terminating error by default
# (ErrorActionPreference=Continue), which bypasses catch.
Remove-Item -LiteralPath $dest -Force -ErrorAction Stop
}
catch {
$dest = [System.IO.Path]::ChangeExtension($dest, '.zip.new')
Write-Warning "[build] canonical helper zip is locked; writing to $dest instead"
}
}
Add-Type -AssemblyName System.IO.Compression.FileSystem
[System.IO.Compression.ZipFile]::CreateFromDirectory(
$HelperDir, $dest,
[System.IO.Compression.CompressionLevel]::Optimal, $false) | Out-Null
$sz = (Get-Item -LiteralPath $dest).Length
Write-Host "[build] wrote $sz bytes to $dest"
}
finally {
if (test-path -LiteralPath $elf32Dest) { Remove-Item -LiteralPath $elf32Dest -Force }
}
}
# # Automatically hot-reloads it into the running emulator
# Send-ToEmulator (join-path $path_build 'hello_gte.ps-exe')
# Invokes reload.ps1 as a child pwsh instead of POSTing to the nonexistent /api/v1/load-exec endpoint.
# Exit code is propagated so the build fails loud if the reload fails.
function Send-ToEmulator {
param([string]$ElfPath = (join-path $path_build 'hello_camera.elf'))
# --- Hot Reload via PCSX-Redux Web Server ---
# $exe_path = join-path $path_build 'hello_gte.ps-exe'
# $absolute_path = [System.IO.Path]::GetFullPath($exe_path)
$reloadScript = join-path $path_scripts 'reload.ps1'
if (-not (test-path -LiteralPath $reloadScript)) {
write-error "[build] reload.ps1 not found at $reloadScript"
exit 1
}
# PCSX-Redux expects the file location in the URL query string?
# We URL-encode the path to ensure backslashes and spaces don't break the HTTP request?
# $encoded_path = [uri]::EscapeDataString($absolute_path)
# $uri = "http://localhost:8080/api/v1/load-exec?path=$encoded_path"
write-host "[build] hot-reloading $ElfPath via reload.ps1" -ForegroundColor Magenta
& pwsh -NoProfile -File $reloadScript -Mode elf -Target hello_camera -ElfPath $ElfPath
if ($LASTEXITCODE -ne 0) {
write-error "[build] reload.ps1 failed (exit $LASTEXITCODE)"
exit $LASTEXITCODE
}
}
# Write-Host "Pushing hot-reload to PCSX-Redux..." -ForegroundColor Magenta
# try {
# # Send the request with the query string included
# Invoke-RestMethod -Uri $uri -Method Post
# Write-Host "Hot-reload successful!" -ForegroundColor Green
# } catch {
# Write-Host "Failed to hot-reload." -ForegroundColor Red
# # This will print the *actual* HTTP error instead of our generic warning
# Write-Host $_.Exception.Message -ForegroundColor Yellow
# }
# Post-build: Regenerate the helper zip (canonical output) and, if -Reload was passed, kick a hot-reload against the just-built ELF.
# Any future targets compiled by this script should add their own Make-HelperZip call after their build step; today's only target is hello_camera.
Make-HelperZip
if ($Reload) {
Send-ToEmulator
}
+15 -279
View File
@@ -217,7 +217,7 @@ local function parse_path_root(input)
if not server_end or server_end == server_start then
error("UNC path requires //server/share: " .. input, 3)
end
local server = input:sub(server_start, server_end - 1)
local server = input:sub(server_start, server_end - 1)
local share_start = server_end + 1
while input:sub(share_start, share_start) == "/" do
share_start = share_start + 1
@@ -515,7 +515,8 @@ local function splice_c_lines(source)
local splice_len = nil
if byte == BYTE_BACKSLASH and source:byte(pos + 1) == BYTE_NEWLINE then
splice_len = 2
elseif byte == BYTE_BACKSLASH and source:byte(pos + 1) == BYTE_CR and source:byte(pos + 2) == BYTE_NEWLINE then
elseif byte == BYTE_BACKSLASH and source:byte(pos + 1) == BYTE_CR
and source:byte(pos + 2) == BYTE_NEWLINE then
splice_len = 3
end
@@ -1052,8 +1053,6 @@ M.GTE_COMMAND_ALIASES = {
-- gte_avg_sort_z3 / gte_avg_sort_z4 are the duffle-side aliases for AVSZ3/4.
["gte_avg_sort_z3"] = "gte_cmdw_avsz3",
["gte_avg_sort_z4"] = "gte_cmdw_avsz4",
["gte_cmdw_sqr"] = "gte_cmdw_sqr",
["gte_cmdw_gpf"] = "gte_cmdw_gpf",
}
-- GTE command input-set table.
@@ -1137,14 +1136,6 @@ M.GTE_COMMAND_INPUTS = {
"C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3",
"gte_cr_ZSF4",
},
-- SQR: reads IR1..IR3 (per PSX-SPX gte.md SQR section; libgte disassembly 0x800160b0).
["gte_cmdw_sqr"] = {
"C2_IR1", "C2_IR2", "C2_IR3",
},
-- GPF: reads IR0 + IR1..IR3 (per PSX-SPX gte.md GPF section; libgte disassembly 0x8001613c).
["gte_cmdw_gpf"] = {
"C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3",
},
}
-- GTE command output-set + semantic role table.
@@ -1217,22 +1208,6 @@ M.GTE_COMMAND_OUTPUTS = {
{ register = "C2_IR2", role = "latest_color" },
{ register = "C2_IR3", role = "latest_color" },
},
["gte_cmdw_sqr"] = {
{ register = "C2_MAC1", role = "mac_result" },
{ register = "C2_MAC2", role = "mac_result" },
{ register = "C2_MAC3", role = "mac_result" },
{ register = "C2_IR1", role = "latest_color" },
{ register = "C2_IR2", role = "latest_color" },
{ register = "C2_IR3", role = "latest_color" },
},
["gte_cmdw_gpf"] = {
{ register = "C2_MAC1", role = "mac_result" },
{ register = "C2_MAC2", role = "mac_result" },
{ register = "C2_MAC3", role = "mac_result" },
{ register = "C2_IR1", role = "latest_color" },
{ register = "C2_IR2", role = "latest_color" },
{ register = "C2_IR3", role = "latest_color" },
},
}
-- GTE command/post-command latch-window table.
@@ -1295,37 +1270,6 @@ M.GTE_COMMAND_LATCH_WINDOWS = {
{ register = "C2_IR2", required = 4 },
{ register = "C2_IR3", required = 4 },
},
["gte_cmdw_sqr"] = {
{ register = "C2_MAC1", required = 4 },
{ register = "C2_MAC2", required = 4 },
{ register = "C2_MAC3", required = 4 },
{ register = "C2_IR1", required = 4 },
{ register = "C2_IR2", required = 4 },
{ register = "C2_IR3", required = 4 },
},
["gte_cmdw_gpf"] = {
{ register = "C2_MAC1", required = 4 },
{ register = "C2_MAC2", required = 4 },
{ register = "C2_MAC3", required = 4 },
{ register = "C2_IR1", required = 4 },
{ register = "C2_IR2", required = 4 },
{ register = "C2_IR3", required = 4 },
},
}
--- GTE control-register alias groups.
--- Aliases within a group write to the same C2 control-register slot on real silicon
--- (the silicon double-maps some C2 slots across multiple PSX SDK / libgte conventions).
--- Aliases across groups write to distinct C2 slots.
---
--- Cross-alias writes inside one atom body, or across the wave-context boundary,
--- silently clobber each other. The `check_gte_cr_alias_writes` check warns about
--- each pair per source. See `docs/gte_reference.md` §"Control-register alias table"
--- for the silicon rationale and the libgte outer-product convention.
M.GTE_CR_ALIAS_GROUPS = {
{ 24, { "gte_cr_RBK", "gte_cr_OFX" } }, -- background R vs screen offset X
{ 25, { "gte_cr_GBK", "gte_cr_OFY" } }, -- background G vs screen offset Y
{ 26, { "gte_cr_BBK", "gte_cr_H" } }, -- background B vs projection plane distance H
}
-- Operand-class table for the COP2->GPR load-delay check.
@@ -1341,7 +1285,6 @@ M.GTE_CR_ALIAS_GROUPS = {
M.OPERAND_READ_POSITIONS = {
-- CPU ALU with one or two GPR operands. Reads every GPR operand.
["add_ui"] = {1, 2},
["li_s"] = {1, 2}, -- rt (write), imm16 (immediate)
["add_ui_self"] = {1},
["add_si"] = {1, 2},
["add_u"] = {1, 2, 3},
@@ -1411,8 +1354,6 @@ M.OPERAND_READ_POSITIONS = {
["gte_mv_to_ctrl_r"] = {},
["gte_lw"] = {},
["gte_sw"] = {},
["shift_lleft_var"] = {1, 2, 3}, -- rd, rt, rs (variable shift amount)
["shift_aright_var"] = {1, 2, 3},
}
-- GP0 packet sizes (total words including the 1-word tag) per GP0 cmd byte.
@@ -1494,10 +1435,8 @@ M.INSTRUCTION_LATENCY = {
["xor_i"] = 1, ["xor_u"] = 1,
["nor_u"] = 1,
["shift_lleft"] = 1, ["shift_lleft_self"] = 1,
["shift_lleft_var"] = 1, -- sllv: 1 cycle
["shift_lright"] = 1,
["shift_aright"] = 1,
["shift_aright_var"] = 1, -- srav: 1 cycle
["mask_upper"] = 1,
["mov_from_high"] = 2, -- mfhi: 2 cycles
["mov_from_low"] = 2, -- mflo: 2 cycles
@@ -1515,7 +1454,6 @@ M.INSTRUCTION_LATENCY = {
["load_half_u"] = 1, ["load_half"] = 1,
["load_byte_u"] = 1, ["load_byte"] = 1,
["load_upper_i"] = 1,
["li_s"] = 1, -- aliased to add_ui(rt, R_0, imm); 1 cycle
-- 2-word loads (lui + ori) used for >16-bit immediates
["load_imm"] = 2,
["load_imm_1w"] = 1,
@@ -1559,8 +1497,6 @@ M.INSTRUCTION_LATENCY = {
["gte_cmdw_op"] = 6, -- OP: 6 cycles (PSX-SPX)
["gte_cmdw_outer_product"] = 6, -- alias for OP
["gte_cmdw_wedge"] = 6, -- alias for OP
["gte_cmdw_sqr"] = 5, -- SQR(sf): 5 cycles (PSX-SPX); +2 nops for pre-fill if sf=0/1
["gte_cmdw_gpf"] = 5, -- GPF(sf,lm): 5 cycles (PSX-SPX); +2 nops for pre-fill if needed
-- Long-form aliases (same cycle cost as their short form)
["gte_cmdw_rotate_translate_perspective_single"] = 15, -- alias for rtps
["gte_cmdw_rotate_translate_perspective_triple"] = 23, -- alias for rtpt
@@ -1589,8 +1525,6 @@ M.INSTRUCTION_LATENCY = {
["atom_bind"] = 0,
["atom_reads"] = 0,
["atom_writes"] = 0,
["BdSlot_"] = 0,
["LdSlot_"] = 0,
}
-- Default cycle cost for unknown macros.
@@ -1843,7 +1777,6 @@ M.CU2_TRANSITION_POLICY = {
M.INSTRUCTION_GPR_EFFECTS = {
-- CPU ALU with one or two GPR operands. Reads every GPR operand position.
add_ui = { reads = {1, 2}, writes = {1} },
li_s = { reads = {1, 2}, writes = {1} }, -- RMW: rt is both read + written
add_ui_self = { reads = {1}, writes = {1} },
add_si = { reads = {1, 2}, writes = {1} },
add_u = { reads = {2, 3}, writes = {1} },
@@ -1960,54 +1893,6 @@ M.INSTRUCTION_GPR_EFFECTS = {
atom_writes = { reads = {}, writes = {} },
-- mac_yield transfers control to the next atom; zero GPR effects.
mac_yield = { reads = {}, writes = {} },
shift_lleft_var = { reads = {2, 3}, writes = {1} },
shift_aright_var = { reads = {2, 3}, writes = {1} },
}
-------------------------------------------------------------------------------
-- IMMEDIATE_FIELD_WIDTHS — maps instruction names to their immediate-argument
-- positions (1-based) and field widths (in bits). Consumed by the
-- `immediate_field_width` static-analysis check. Parallel to
-- INSTRUCTION_GPR_EFFECTS.
--
-- `signed = true` means the field is sign-extended (the value must fit in
-- the signed range). `signed = false` (default) means zero-extended.
-------------------------------------------------------------------------------
M.IMMEDIATE_FIELD_WIDTHS = {
-- CPU I-type immediates: 16-bit signed (addiu/addi/slti sign-extend)
add_ui = { { arg = 3, width = 16, signed = true } },
add_si = { { arg = 3, width = 16, signed = true } },
add_ui_self = { { arg = 2, width = 16, signed = true } },
slt_si = { { arg = 3, width = 16, signed = true } },
slt_ui = { { arg = 3, width = 16, signed = true } },
-- CPU I-type immediates: 16-bit unsigned (andi/ori/xori zero-extend)
and_i = { { arg = 3, width = 16 } },
or_i = { { arg = 3, width = 16 } },
or_i_self = { { arg = 2, width = 16 } },
xor_i = { { arg = 3, width = 16 } },
load_upper_i = { { arg = 2, width = 16 } },
-- Load/store offsets: 16-bit signed
load_word = { { arg = 3, width = 16, signed = true } },
load_half = { { arg = 3, width = 16, signed = true } },
load_half_u = { { arg = 3, width = 16, signed = true } },
load_byte = { { arg = 3, width = 16, signed = true } },
load_byte_u = { { arg = 3, width = 16, signed = true } },
store_word = { { arg = 3, width = 16, signed = true } },
store_half = { { arg = 3, width = 16, signed = true } },
store_byte = { { arg = 3, width = 16, signed = true } },
-- Shift amount: 5-bit unsigned
shift_lleft = { { arg = 3, width = 5 } },
shift_lleft_self = { { arg = 2, width = 5 } },
shift_lright = { { arg = 3, width = 5 } },
shift_aright = { { arg = 3, width = 5 } },
shift_aright_var = { { arg = 3, width = 5 } },
-- Branch offsets: 16-bit signed
branch_equal = { { arg = 3, width = 16, signed = true } },
branch_ne = { { arg = 3, width = 16, signed = true } },
branch_le_zero = { { arg = 2, width = 16, signed = true } },
branch_lt_zero = { { arg = 2, width = 16, signed = true } },
branch_ge_zero = { { arg = 2, width = 16, signed = true } },
branch_gt_zero = { { arg = 2, width = 16, signed = true } },
}
-- Bounded GPR-value rules consumed by the same forward event walk as `INSTRUCTION_GPR_EFFECTS`.
@@ -2018,19 +1903,18 @@ M.IMMEDIATE_FIELD_WIDTHS = {
-- * passes/static_analysis.lua::apply_gpr_effects
-- No second `bounded_value_pass` is permitted.
M.GPR_VALUE_RULES = {
load_upper_i = { op = "load_upper_i", dest = 1, immediate = 2, },
add_ui = { op = "add_ui", dest = 1, source = 2, immediate = 3, },
li_s = { op = "add_ui", dest = 1, source = 2, immediate = 3 }, -- R_0 + sign-ext(imm) folds into a constant
or_i = { op = "or_i", dest = 1, source = 2, immediate = 3, },
and_i = { op = "and_i", dest = 1, source = 2, immediate = 3, },
xor_i = { op = "xor_i", dest = 1, source = 2, immediate = 3, },
add_ui_self = { op = "add_ui", dest = 1, source = 1, immediate = 2, },
or_i_self = { op = "or_i", dest = 1, source = 1, immediate = 2, },
load_upper_i = { op = "load_upper_i", dest = 1, immediate = 2, },
add_ui = { op = "add_ui", dest = 1, source = 2, immediate = 3, },
or_i = { op = "or_i", dest = 1, source = 2, immediate = 3, },
and_i = { op = "and_i", dest = 1, source = 2, immediate = 3, },
xor_i = { op = "xor_i", dest = 1, source = 2, immediate = 3, },
add_ui_self = { op = "add_ui", dest = 1, source = 1, immediate = 2, },
or_i_self = { op = "or_i", dest = 1, source = 1, immediate = 2, },
-- Present register-form self variants. They are included here so a
-- known value is not needlessly lost when these encoders are used.
add_u_self = { op = "add_u", dest = 1, sources = {1, 2}, },
or_u_self = { op = "or", dest = 1, sources = {1, 2}, },
shift_lleft_self = { op = "shift_lleft", dest = 1, source = 1, immediate = 2, },
add_u_self = { op = "add_u", dest = 1, sources = {1, 2}, },
or_u_self = { op = "or", dest = 1, sources = {1, 2}, },
shift_lleft_self = { op = "shift_lleft", dest = 1, source = 1, immediate = 2, },
}
-- Control-transfer (branch/jump/call) delay-slot policy table.
@@ -2157,8 +2041,7 @@ local E_MAC_PREFIX_LEN = 4
--- * Unknown `mac_X` (not in `component_index`): fall back to `word_counts[ident]` if present; otherwise emit one opaque event so the cycle budget accounts for the word.
--- * Marker Tokens (`atom_label(...)` / `atom_offset(...)`): Zero events (they are pure metaprogram hints).
---
--- Cycle protection: a per-expansion `visiting` set tracks components currently on the expansion stack;
--- a re-entry produces a deterministic `{kind = "cycle", ...}` error and aborts that branch (does NOT hang, does NOT recurse).
--- Cycle protection: a per-expansion `visiting` set tracks components currently on the expansion stack; a re-entry produces a deterministic `{kind = "cycle", ...}` error and aborts that branch (does NOT hang, does NOT recurse).
---
--- Pure: reads `body_entry` / `component_index` / `word_counts`. Memoization is the caller's responsibility.
--- Callers wanting `word_events` / `word_event_errors` precomputed for many atoms should memoize them per atom.
@@ -2729,151 +2612,4 @@ function M.project_emission(body_text, component_index, word_counts, components)
})
end
-------------------------------------------------------------------------------
-- find_function_decl_for — backward walk for MipsAtomComp_Proc_ name extraction.
--
-- After the `sym` arg was dropped from MipsAtomComp_Proc_, the component name
-- is derived from the preceding `FI_ Slice_MipsCode ac_X(args)` function
-- declaration. This function walks backward from `before_pos` to find it.
--
-- Returns (raw_name, args_inner) or (nil, nil).
-- raw_name — e.g. "ac_load_word_imm"
-- args_inner — e.g. "AtomBuilder_R ab, Reg dst, U4 imm"
--
-- The walk finds the LAST "Slice_MipsCode" before before_pos, then skips
-- whitespace + qualifiers (FI_, atom_dbg_skip, comments) until it finds an
-- ident followed by "(". That ident is the function name; the parens contents
-- are the args.
-------------------------------------------------------------------------------
function M.find_function_decl_for(source, before_pos, slice_mips_code_len)
local search_pos = 1
local last_match = nil
while true do
local found = source:find("Slice_MipsCode", search_pos, true)
if not found or found >= before_pos then break end
last_match = found
search_pos = found + slice_mips_code_len
end
if not last_match then return nil, nil end
local pos = last_match + slice_mips_code_len
while pos < before_pos do
-- skip whitespace
while pos <= #source do
local c = source:sub(pos, pos)
if c == " " or c == "\t" or c == "\n" or c == "\r" then
pos = pos + 1
else
break
end
end
if pos > #source then break end
-- skip line comments
if source:sub(pos, pos + 1) == "//" then
while pos <= #source and source:sub(pos, pos) ~= "\n" do pos = pos + 1 end
pos = pos + 1
goto continue
end
-- skip block comments
if source:sub(pos, pos + 1) == "/*" then
local close = source:find("*/", pos + 2, true)
if not close then break end
pos = close + 2
goto continue
end
-- try to read an ident
local ident, ident_end = M.read_ident(source, pos)
if not ident then break end
-- check if the next non-ws char after ident is "("
local next_pos = M.skip_ws_and_cmt(source, ident_end)
if source:sub(next_pos, next_pos) == "(" then
local inner = M.read_parens(source, next_pos)
if inner then
return ident, inner
end
end
-- ident not followed by "(" — it's a qualifier (FI_, atom_dbg_skip, etc); skip it
pos = ident_end
::continue::
end
return nil, nil
end
-------------------------------------------------------------------------------
-- find_atom_proc_decl_for — backward walk for MipsAtom_Proc_ name extraction.
--
-- After the `sym` arg was dropped from MipsAtom_Proc_, the atom name is
-- derived from the preceding `MipsAtom* X_proc(args)` function declaration.
-- This function walks backward from `before_pos` to find it.
--
-- Returns (raw_name, args_inner) or (nil, nil).
-- raw_name — e.g. "normalize_v3s4" (the _proc suffix is stripped)
-- args_inner — e.g. "AtomArena_R aa, U4 r_scratch, ..."
--
-- The walk finds the LAST "MipsAtom*" before before_pos, then skips
-- whitespace + qualifiers (internal, I_, FI_, comments) until it finds an
-- ident followed by "(". That ident is the function name (with _proc suffix);
-- the suffix is stripped to get raw_name. The parens contents are the args.
-------------------------------------------------------------------------------
function M.find_atom_proc_decl_for(source, before_pos, mips_atom_ptr_len)
local search_pos = 1
local last_match = nil
while true do
-- plain=true: "*" is literal, no escaping needed
local found = source:find("MipsAtom*", search_pos, true)
if not found or found >= before_pos then break end
last_match = found
search_pos = found + mips_atom_ptr_len
end
if not last_match then return nil, nil end
local pos = last_match + mips_atom_ptr_len
while pos < before_pos do
-- skip whitespace
while pos <= #source do
local c = source:sub(pos, pos)
if c == " " or c == "\t" or c == "\n" or c == "\r" then
pos = pos + 1
else
break
end
end
if pos > #source then break end
-- skip line comments
if source:sub(pos, pos + 1) == "//" then
while pos <= #source and source:sub(pos, pos) ~= "\n" do pos = pos + 1 end
pos = pos + 1
goto continue
end
-- skip block comments
if source:sub(pos, pos + 1) == "/*" then
local close = source:find("*/", pos + 2, true)
if not close then break end
pos = close + 2
goto continue
end
-- try to read an ident
local ident, ident_end = M.read_ident(source, pos)
if not ident then break end
-- check if the next non-ws char after ident is "("
local next_pos = M.skip_ws_and_cmt(source, ident_end)
if source:sub(next_pos, next_pos) == "(" then
local inner = M.read_parens(source, next_pos)
if inner then
-- strip the _proc suffix to get the atom name
local proc_suffix = "_proc"
if #ident > #proc_suffix and ident:sub(-#proc_suffix) == proc_suffix then
return ident:sub(1, #ident - #proc_suffix), inner
end
-- no _proc suffix — return as-is
return ident, inner
end
end
-- ident not followed by "(" — it's a qualifier; skip it
pos = ident_end
::continue::
end
return nil, nil
end
return M
return M
+3 -2
View File
@@ -47,14 +47,15 @@ local function find_repo_root()
return root
end
--- Set `package.path` (for `require("duffle")` + `require("passes.X")`) and `package.cpath` (for `lpeg.dll`).
--- Set `package.path` (for `require("duffle")` + `require("passes.X")`) and
--- `package.cpath` (for `lpeg.dll`).
---
--- This script does NOT touch the OS environment: no `os.setenv`, no `os.putenv`, no `$PATH` mods.
--- It just sets `package.path` and `package.cpath` (the standard Lua way to register module search dirs).
--- lpeg is built by `update_deps.ps1` to `toolchain/lpeg/`,
--- which we wire into `package.cpath` here (so `require("lpeg")` from `duffle.lua` resolves without any global state).
function M.setup()
local repo_root = find_repo_root()
local repo_root = find_repo_root()
if not repo_root then
-- Unreachable in practice: find_repo_root() derives the repo root from this script's
-- own source path via debug.getinfo(1, "S").source (no subprocess, no git CLI, <1ms).
-418
View File
@@ -1,418 +0,0 @@
-- elf32.lua — Pure-Lua ELF32 format helpers with no lfs / no lpeg dependency.
-- The reload helper's `parse_manifest` (scripts/pcsx_debug_helper/reload.lua)
-- and the metaprogram's `read_elf_sections` + `read_nm` (scripts/elf_dwarf.lua)
-- both parsed ELF32 headers from wire bytes.
--
-- This module contains the format constants and the byte-level walker.
--- The metaprogram side keeps `read_u32_le` / `read_u16_le` as local forwarders; the helper side calls `E.*` directly.
--
-- **Adapter contract (explicit pass style):**
-- The helper VM's `Support.File` exposes byte-read methods that require `self` (fileffi.lua:225-227),
-- so callers wrap once in a 1-line adapter that strips `self`.
-- The parsers here operate on the unwrapped form.
-- Reads are flat function calls — `E.read_u8(adapter, off)`, `E.read_u32(adapter, off)`, `E.size(adapter)`.
-- read_u8(adapter, off) -> integer | nil
-- read_u16(adapter, off) -> integer | nil
-- read_u32(adapter, off) -> integer | nil
-- size(adapter) -> integer
--
-- **Convention:** every offset in the constants tables is a zero-based wire offset.
-- The `+ 1` conversion happens only at the `string.byte` boundary inside the readers.
--
-- spec: System V ABI gABI v1.2 §"ELF Header" (Table 1) + §"Section Header Table"
-- spec: System V ABI gABI v1.2 §"Symbol Table" (Elf32_Sym layout)
local M = {}
-- ════════════════════════════════════════════════════════════════════════════
-- Little-endian readers (bit-weighted accumulator, math.floor only)
-- ════════════════════════════════════════════════════════════════════════════
--- Read a 4-byte little-endian unsigned integer from `adapter` at zero-based wire offset `off`.
---
--- Bit weights are written as `0x100`, `0x10000`, `0x1000000` (i.e. 2^8, 2^16, 2^24) so the LE byte positions are visually explicit:
--- byte 0 contributes its value directly;
--- byte 1 is shifted left by 8; byte 2 by 16; byte 3 by 24.
---
--- math.floor (not LuaJIT's `>>`) keeps the body portable across LuaJIT 2.0/2.1 and plain Lua 5.x. `string.byte` receives `+ 1` at the boundary.
---
--- **Call form:** explicit-pass. The reader receives `adapter` as the first positional argument and the offset as the second; no `self` is passed.
--- Test fixtures declare `function(offset) ... end` and the parsers call them via dot syntax `adapter.read_u8_at(off)`.
--- The colon form `adapter:read_u8_at(off)` would prepend the adapter table as `offset` and break the contract.
--- @param adapter table
--- @param off integer -- zero-based wire offset
--- @return integer|nil
function M.read_u32(adapter, off)
return adapter.read_u8_at(off)
+ adapter.read_u8_at(off + 0x01) * 0x00000100
+ adapter.read_u8_at(off + 0x02) * 0x00010000
+ adapter.read_u8_at(off + 0x03) * 0x01000000
end
--- Read a 2-byte little-endian unsigned integer from `adapter` at zero-based wire offset `off`.
--- @param adapter table
--- @param off integer -- zero-based wire offset
--- @return integer|nil
function M.read_u16(adapter, off)
return adapter.read_u8_at(off)
+ adapter.read_u8_at(off + 0x01) * 0x00000100
end
--- Read a 1-byte unsigned integer from `adapter` at zero-based wire offset `off`.
--- @param adapter table
--- @param off integer -- zero-based wire offset
--- @return integer|nil
function M.read_u8(adapter, off)
return adapter.read_u8_at(off)
end
--- Total adapter byte length.
--- @param adapter table
--- @return integer
function M.size(adapter)
return adapter.read_size()
end
--- Forwarders kept for backward compat with scripts/elf_dwarf.lua.
--- The metaprogram side keeps `read_u32_le` / `read_u16_le`;
--- both layers now use the same byte-level helpers under the hood.
function M.read_u32_le(buf, off)
local byte_off = off + 1
return buf:byte(byte_off)
+ buf:byte(byte_off + 0x01) * 0x00000100
+ buf:byte(byte_off + 0x02) * 0x00010000
+ buf:byte(byte_off + 0x03) * 0x01000000
end
--- Read a 2-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
--- @param buf string
--- @param off integer -- zero-based wire offset
--- @return integer
function M.read_u16_le(buf, off)
local byte_off = off + 1
return buf:byte(byte_off) + buf:byte(byte_off + 0x01) * 0x00000100
end
-- ════════════════════════════════════════════════════════════════════════════
-- Format constants
-- ════════════════════════════════════════════════════════════════════════════
-- ELF format constants (System V ABI gABI v1.2).
M.ELFCLASS32 = 1 -- spec: gABI v1.2 §"ELF Header" — EI_CLASS byte
M.ELFDATA2LSB = 1 -- spec: gABI v1.2 §"ELF Header" — EI_DATA byte
M.EM_MIPS = 8 -- spec: gABI v1.2 §"Machine Information" — MIPS architecture
-- Section type constants (System V ABI gABI v1.2 §"Section Header Table").
M.SHT_SYMTAB = 2 -- spec: gABI v1.2 §"Section Types" — symbol table
M.SHT_STRTAB = 3 -- spec: gABI v1.2 §"Section Types" — string table
M.SHT_NOBITS = 8 -- spec: gABI v1.2 §"Section Types" — no space in file
-- Section flag constants (System V ABI gABI v1.2 §"Section Header Table").
M.SHF_WRITE = 0x1 -- spec: gABI v1.2 §"Section Attributes" — writable
M.SHF_ALLOC = 0x2 -- spec: gABI v1.2 §"Section Attributes" — occupies memory
M.SHF_EXECINSTR = 0x4 -- spec: gABI v1.2 §"Section Attributes" — executable
-- ---------------------------------------------------------------------------
-- ELF32 header layout (System V ABI gABI v1.2 §"ELF Header" Table 1)
-- ---------------------------------------------------------------------------
-- All offsets are zero-based wire offsets. The header is 52 bytes total (header_bytes = 0x34 = 52).
M.ELF32_HEADER = {
magic_offset = 0x00, -- 4 bytes; expected "\127ELF"
magic = "\127ELF",
class_offset = 0x04, -- 1 byte; 1 = ELF32, 2 = ELF64
endian_offset = 0x05, -- 1 byte; 1 = little-endian, 2 = big-endian
header_bytes = 0x34, -- ELF32 header is 52 bytes total
e_entry_offset = 0x18, -- 4-byte LE; entry-point virtual address
e_shoff_offset = 0x20, -- 4-byte LE; section-header table file offset
e_shentsize_offset = 0x2E, -- 2-byte LE; section-header entry size in bytes
e_shnum_offset = 0x30, -- 2-byte LE; number of section headers
e_shstrndx_offset = 0x32, -- 2-byte LE; index of section-name string table
}
-- ---------------------------------------------------------------------------
-- ELF32 section-header layout (System V ABI gABI v1.2 §"Section Header Table")
-- ---------------------------------------------------------------------------
-- Each entry is 40 bytes (sh_entsize_bytes = 0x28 = 40);
-- zero-based, field offsets relative to the start of the entry.
M.ELF32_SECTION = {
sh_name_offset = 0x00, -- 4-byte LE; offset into .shstrtab
sh_type_offset = 0x04, -- 4-byte LE; section type (SHT_*)
sh_flags_offset = 0x08, -- 4-byte LE; section flags (SHF_*)
sh_addr_offset = 0x0C, -- 4-byte LE; virtual address at execution
sh_offset_offset = 0x10, -- 4-byte LE; section's file offset
sh_size_offset = 0x14, -- 4-byte LE; section's size in bytes
sh_link_offset = 0x18, -- 4-byte LE; link to a related section
sh_entsize_bytes = 0x28, -- spec: gABI v1.2 §"Section Header Table" — 40 bytes per entry
}
-- ---------------------------------------------------------------------------
-- ELF32 symbol-table entry layout (System V ABI gABI v1.2 §"Symbol Table")
-- ---------------------------------------------------------------------------
-- Each entry is 16 bytes (sym_entry_bytes = 0x10 = 16);
-- zero-based, field offsets relative to the start of the entry.
M.ELF32_SYM = {
st_name = 0x00, -- 4-byte LE; offset into the linked string table
st_value = 0x04, -- 4-byte LE; symbol value (address / absolute)
st_size = 0x08, -- 4-byte LE; symbol size in bytes
st_info = 0x0C, -- 1 byte; binding (high nibble) + type (low nibble)
sym_entry_bytes = 0x10, -- spec: gABI v1.2 §"Symbol Table" — 16 bytes per entry
}
-- DWARF32 initial-length terminator (DWARF4 §7.4) — kept here so the metaprogram's elf_dwarf.lua can drop its own copy of the same constant.
M.dw_dwarf32_terminator = 0xFFFFFFFF
-- ════════════════════════════════════════════════════════════════════════════
-- Adapter validation
-- ════════════════════════════════════════════════════════════════════════════
--- Validate that `adapter` exposes the byte-read surface.
--- Returns true on success, false + a stable error code on failure.
--- The helper side calls this before parse_manifest to reject callers before any byte is read.
--- @param adapter any
--- @return boolean, string|nil
function M.validate_adapter(adapter)
if type(adapter) ~= "table" then return false, "bad_file_adapter" end
if type(adapter.read_u8_at) ~= "function" then return false, "bad_file_adapter" end
if type(adapter.read_u16_at) ~= "function" then return false, "bad_file_adapter" end
if type(adapter.read_u32_at) ~= "function" then return false, "bad_file_adapter" end
if type(adapter.read_size) ~= "function" then return false, "bad_file_adapter" end
return true, nil
end
-- ════════════════════════════════════════════════════════════════════════════
-- String-table reader
-- ════════════════════════════════════════════════════════════════════════════
--- Extract a NUL-terminated C string from `strtab` at zero-based offset `off`.
--- Returns nil if `off` is out of range or the string is not NUL-terminated.
--- @param strtab string
--- @param off integer
--- @return string|nil
function M.get_str(strtab, off)
if off < 0 or off >= #strtab then return nil end
local end_pos = strtab:find("\0", off + 1, true)
if not end_pos then return nil end
return strtab:sub(off + 1, end_pos - 1)
end
-- ════════════════════════════════════════════════════════════════════════════
-- Header / section / symbol walkers
-- ════════════════════════════════════════════════════════════════════════════
--- Read the ELF32 header through `adapter` and validate the magic, class, and data encoding.
--- Returns a table on success:
--- { e_entry, e_shoff, e_shentsize, e_shnum, e_shstrndx, error = nil }
--- On failure returns nil + a stable error code:
--- bad_magic, unsupported_elf_class, unsupported_elf_data, truncated_header
--- The header's machine field is NOT validated here — callers (e.g. the helper's prime path) decide whether to require EM_MIPS before symbol reads.
--- @param adapter table
--- @return table|nil, string|nil
function M.parse_elf32_headers(adapter)
local ok, err = M.validate_adapter(adapter)
if not ok then return nil, err end
-- 4-byte magic: 0x7F 'E' 'L' 'F'.
-- The byte readers take the adapter explicitly.
-- The production `Support.File` adapter is wrapped by the caller to drop its implicit `self` so the parser shape is flat pass-style.
local b1 = M.read_u8(adapter, 0)
local b2 = M.read_u8(adapter, 1)
local b3 = M.read_u8(adapter, 2)
local b4 = M.read_u8(adapter, 3)
if not (b1 and b2 and b3 and b4)
or not (b1 == 0x7f and b2 == 0x45 and b3 == 0x4c and b4 == 0x46) then
return nil, "bad_magic"
end
local class = M.read_u8(adapter, M.ELF32_HEADER.class_offset)
if class ~= M.ELFCLASS32 then
return nil, "unsupported_elf_class"
end
local data = M.read_u8(adapter, M.ELF32_HEADER.endian_offset)
if data ~= M.ELFDATA2LSB then
return nil, "unsupported_elf_data"
end
local e_entry = M.read_u32(adapter, M.ELF32_HEADER.e_entry_offset)
local e_shoff = M.read_u32(adapter, M.ELF32_HEADER.e_shoff_offset)
local e_shentsize = M.read_u16(adapter, M.ELF32_HEADER.e_shentsize_offset)
local e_shnum = M.read_u16(adapter, M.ELF32_HEADER.e_shnum_offset)
local e_shstrndx = M.read_u16(adapter, M.ELF32_HEADER.e_shstrndx_offset)
if not (e_entry and e_shoff and e_shentsize and e_shnum and e_shstrndx) then
return nil, "truncated_header"
end
return {
e_entry = e_entry,
e_shoff = e_shoff,
e_shentsize = e_shentsize,
e_shnum = e_shnum,
e_shstrndx = e_shstrndx,
error = nil,
}
end
--- Read one section-header entry from `adapter` at `sh_off`.
--- Returns a table with the wire fields plus a (yet-unresolved) `name` field.
--- @param adapter table
--- @param sh_off integer
--- @return table|nil, string|nil -- entry, error
local function read_section_entry(adapter, sh_off)
local entry = {
sh_name = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_name_offset),
sh_type = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_type_offset),
sh_flags = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_flags_offset),
sh_addr = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_addr_offset),
sh_offset = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_offset_offset),
sh_size = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_size_offset),
sh_link = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_link_offset),
name = "",
}
if not (entry.sh_name and entry.sh_type and entry.sh_flags and entry.sh_addr
and entry.sh_offset and entry.sh_size and entry.sh_link) then
return nil, "truncated_section_headers"
end
return entry, nil
end
--- Walk every section header in `hdr` and return a 1-based array of entries
--- (the section at logical index 0 is at array position 1, etc.).
--- Each entry has the wire fields plus a resolved `name` derived from `.shstrtab`.
--- Returns nil + a stable error code on failure: truncated_section_headers, missing_shstrtab, truncated_strtab
--- @param adapter table
--- @param hdr table -- the table returned by parse_elf32_headers
--- @return table|nil, string|nil
function M.walk_sections(adapter, hdr)
if not hdr or hdr.error then return nil, hdr and hdr.error or "truncated_section_headers" end
local file_size = M.size(adapter)
if hdr.e_shoff + hdr.e_shnum * hdr.e_shentsize > file_size then
return nil, "truncated_section_headers"
end
-- Read every section header first; we need .shstrtab to resolve names.
local sections = {}
for i = 0, hdr.e_shnum - 1 do
local sh_off = hdr.e_shoff + i * hdr.e_shentsize
local entry, err = read_section_entry(adapter, sh_off)
if not entry then return nil, err end
sections[i + 1] = entry
end
if hdr.e_shstrndx >= hdr.e_shnum then
return nil, "missing_shstrtab"
end
local shstrtab = sections[hdr.e_shstrndx + 1]
if not shstrtab or shstrtab.sh_type ~= M.SHT_STRTAB then
return nil, "missing_shstrtab"
end
if shstrtab.sh_offset + shstrtab.sh_size > file_size then
return nil, "truncated_section_headers"
end
local shstrtab_bytes = M.read_section_bytes(adapter, shstrtab)
if not shstrtab_bytes then return nil, "truncated_section_headers" end
for _, s in ipairs(sections) do
s.name = M.get_str(shstrtab_bytes, s.sh_name) or ""
end
return sections, nil
end
--- Read the bytes of one section. Returns a string, or nil if the adapter returns nil for any byte (out-of-bounds).
--- The caller is responsible fors sizing the buffer (the section's sh_offset + sh_size must fit in adapter.size).
--- @param adapter table
--- @param section table -- one entry from walk_sections
--- @return string|nil
function M.read_section_bytes(adapter, section)
local size = section.sh_size
if size == 0 then return "" end
local out = {}
for i = 0, size - 1 do
local b = M.read_u8(adapter, section.sh_offset + i)
if b == nil then return nil end
out[#out + 1] = string.char(b)
end
return table.concat(out)
end
--- Convenience: walk sections, then look up the named section, then read its bytes.
--- Returns nil + a stable error code if the section is absent or out-of-bounds.
--- @param adapter table
--- @param sections table -- 1-based array from walk_sections
--- @param name string
--- @return string|nil, string|nil
function M.read_named_section(adapter, sections, name)
if not sections then return nil, "missing_section" end
for _, s in ipairs(sections) do
if s.name == name then
local bytes = M.read_section_bytes(adapter, s)
if not bytes then return nil, "truncated_section_data" end
return bytes, nil
end
end
return nil, "missing_section"
end
--- Walk every SHT_SYMTAB section in `sections` and accumulate symbols by name.
--- Each stored entry is `{ value = st_value, size = st_size, info = st_info, shndx = st_shndx }`.
--- Both STB_LOCAL and STB_GLOBAL symbols are included; the live ELF stores `smem` as a local symbol.
--- Returns nil + a stable error code on failure: missing_symtab_strtab, truncated_section_headers
--- @param adapter table
--- @param sections table
--- @return table|nil, string|nil
function M.collect_symbols(adapter, sections)
if not sections then return nil, "missing_sections" end
local symbols = {}
local file_size = M.size(adapter)
for _, s in ipairs(sections) do
if s.sh_type == M.SHT_SYMTAB then
local strtab = sections[s.sh_link + 1]
if not strtab or strtab.sh_type ~= M.SHT_STRTAB then
return nil, "missing_symtab_strtab"
end
if strtab.sh_offset + strtab.sh_size > file_size then
return nil, "truncated_section_headers"
end
local strtab_bytes = M.read_section_bytes(adapter, strtab)
if not strtab_bytes then return nil, "truncated_section_headers" end
if s.sh_offset + s.sh_size > file_size then
return nil, "truncated_section_headers"
end
local symtab_bytes = M.read_section_bytes(adapter, s)
if not symtab_bytes then return nil, "truncated_section_headers" end
local n = #symtab_bytes / M.ELF32_SYM.sym_entry_bytes
for j = 0, n - 1 do
local e = s.sh_offset + j * M.ELF32_SYM.sym_entry_bytes
local st_name = M.read_u32(adapter, e + M.ELF32_SYM.st_name)
if st_name then
local st_value = M.read_u32(adapter, e + M.ELF32_SYM.st_value)
local st_size = M.read_u32(adapter, e + M.ELF32_SYM.st_size)
local st_info = M.read_u8(adapter, e + M.ELF32_SYM.st_info)
-- st_shndx is at offset 14 (2 bytes) — derived from the layout
-- the metaprogram reads too. Inline the read to keep the
-- adapter as the only I/O surface.
local b1 = M.read_u8(adapter, e + 14)
local b2 = M.read_u8(adapter, e + 15)
if not (b1 and b2) then
return nil, "truncated_section_headers"
end
local st_shndx = b1 + b2 * 0x100
local name = M.get_str(strtab_bytes, st_name) or ""
if name ~= "" then
symbols[name] = {
value = st_value,
size = st_size,
info = st_info,
shndx = st_shndx,
}
end
end
end
end
end
return symbols, nil
end
return M
+140 -153
View File
@@ -11,11 +11,6 @@
-- lfs is wired into package.cpath by `duffle_paths.lua` (vendored under `toolchain/lfs/lfs.dll`).
local lfs = require("lfs")
-- scripts/elf32.lua contains format-constant tables + the byte-level walker.
-- The this file re-exports `read_u32_le` / `read_u16_le` (and the DWARF32 terminator).
-- TODO(Ed): Remove re-export.
local E = require("elf32")
local M = {}
-- ════════════════════════════════════════════════════════════════════════════
@@ -107,13 +102,27 @@ M.MIPS_BYTES_PER_WORD = 0x04
--- **Wire-offset contract:** format offsets, fixed-width reader offsets, LEB/parser cursors, and section-relative values are zero-based wire offsets.
--- Only Lua string APIs receive a `+ 1` conversion at their boundary (`byte`, `sub`, and `find`).
--- ELF/DWARF field offsets are expressed in hex so they map directly to the zero-based byte positions in the binary file.
---
--- The ELF32 header / section / sym layout tables are within scripts/elf32.lua.
--- The metaprogram re-exports the DWARF32 initial-length terminator.
--- spec: DWARF4 spec §7.4 — 32-bit DWARF initial-length terminator
M.dw_dwarf32_terminator = E.dw_dwarf32_terminator
-- TODO(Ed): Remove re-export.
--- spec: System V ABI gABI v1.2 §"ELF Header" (Table 1) + §"Section Header Table"
M.ELF32 = {
magic_offset = 0x00, -- 4-byte magic "\127ELF" at file offset 0x00
magic = "\127ELF",
class_offset = 0x04, -- 1-byte; 1 = ELF32, 2 = ELF64
class_elf32 = 1,
endian_offset = 0x05, -- 1-byte; 1 = little-endian, 2 = big-endian
endian_little = 1,
header_bytes = 0x34, -- spec: gABI v1.2 §"ELF Header" — ELF32 header is 52 bytes total
e_shoff_offset = 0x20, -- 4-byte LE; section-header table file offset
e_shentsize_offset = 0x2E, -- 2-byte LE; section-header entry size in bytes
e_shnum_offset = 0x30, -- 2-byte LE; number of section headers
e_shstrndx_offset = 0x32, -- 2-byte LE; index of section-name string table
sh_size_bytes = 0x28, -- spec: gABI v1.2 §"Section Header Table" — each entry is 40 bytes
sh_name_offset = 0x00, -- 4-byte LE; offset into .shstrtab
sh_type_offset = 0x04, -- 4-byte LE; section type (SHT_*)
sh_offset_offset = 0x10, -- 4-byte LE; section's file offset
sh_size_offset = 0x14, -- 4-byte LE; section's size in bytes
dw_dwarf32_terminator = 0xFFFFFFFF, -- spec: DWARF4 spec §7.4 — 32-bit DWARF initial-length terminator
}
-- ----------------------------------------------------------------------------
-- DWARF4 .debug_aranges (per DWARF5 spec §7.4 — Address Range Table)
@@ -232,24 +241,27 @@ M.DWARF5_DEBUG_LINE = {
--- (which has partial `string.unpack` coverage).
--- **Convention:** `off` is a zero-based wire offset; `+ 1` is applied only at the `string.byte` boundary.
---
--- Thin forwarder: the canonical implementation lives in scripts/elf32.lua.
--- The "second caller lifts" pattern keeps the metaprogram side fluent
--- (`M.read_u32_le(buf, off)`) while the body is deduped.
--- **Byte weights** are written as `0x100`, `0x10000`, `0x1000000` (i.e. 2^8, 2^16, 2^24) so the LE byte positions are visually explicit:
--- byte 0 contributes its value directly; byte 1 is shifted left by 8 (= 0x100); byte 2 by 16 (= 0x10000); byte 3 by 24 (= 0x1000000).
--- @param buf string
--- @param off integer -- zero-based wire offset
--- @return integer
function M.read_u32_le(buf, off)
return E.read_u32_le(buf, off)
local byte_off = off + 1
return buf:byte(byte_off)
+ buf:byte(byte_off + 0x01) * 0x00000100
+ buf:byte(byte_off + 0x02) * 0x00010000
+ buf:byte(byte_off + 0x03) * 0x01000000
end
--- Read a 2-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
--- (`off` is zero-based; `+ 1` is applied only at the `string.byte` boundary.)
--- Thin forwarder — see `M.read_u32_le` for the rationale.
--- @param buf string
--- @param off integer -- zero-based wire offset
--- @return integer
function M.read_u16_le(buf, off)
return E.read_u16_le(buf, off)
local byte_off = off + 1
return buf:byte(byte_off) + buf:byte(byte_off + 0x01) * 0x00000100
end
-- Pure-Lua 5.3 LEB128 readers (no `bit` library). `2^shift` arithmetic matches the existing parser.
@@ -430,20 +442,20 @@ function M.read_ref_sig8(buf, pos)
return M.read_u32_le(buf, pos), M.read_u32_le(buf, pos + 4), pos + 8
end
--- DWARF5 §7.5.6 (Type Entries).
--- Walk all units in `info` and return the 0-based offset of the first unit whose `DW_AT_type_signature`
--- (8-byte value at the end of the unit header) equals `target_sig`.
--- The signature is interpreted as two 32-bit halves (low/high) per the read_ref_sig8 contract;
--- we match both halves (i.e. the 8-byte value as a whole). Returns nil if no matching unit exists.
---
--- Unit header layout (from pos 0):
--- unit_length(4) + version(2) + unit_type(1) + address_size(1) + debug_abbrev_offset(4)
--- followed by type_unit_specific fields: type_signature(8) + type_offset(4)
--- The type_signature is at byte offset 8 of the body (right after debug_abbrev_offset).
--- @param info string -- the .debug_info section bytes
--- @param target_sig_lo integer -- low 4 bytes (LE) of the desired signature
--- @param target_sig_hi integer -- high 4 bytes (LE) of the desired signature
--- @return integer|nil, integer|nil -- unit offset, type_offset within the unit
-- DWARF5 §7.5.6 (Type Entries).
-- Walk all units in `info` and return the 0-based offset of the first unit
-- whose `DW_AT_type_signature` (8-byte value at the end of the unit header) equals `target_sig`.
-- The signature is interpreted as two 32-bit halves (low/high) per the read_ref_sig8 contract;
-- we match both halves (i.e. the 8-byte value as a whole). Returns nil if no matching unit exists.
--
-- Unit header layout (from pos 0):
-- unit_length(4) + version(2) + unit_type(1) + address_size(1) + debug_abbrev_offset(4)
-- followed by type_unit_specific fields: type_signature(8) + type_offset(4)
-- The type_signature is at byte offset 8 of the body (right after debug_abbrev_offset).
-- @param info string -- the .debug_info section bytes
-- @param target_sig_lo integer -- low 4 bytes (LE) of the desired signature
-- @param target_sig_hi integer -- high 4 bytes (LE) of the desired signature
-- @return integer|nil, integer|nil -- unit offset, type_offset within the unit
function M.find_type_unit_by_signature(info, target_sig_lo, target_sig_hi)
local pos = 0
local section_len = #info
@@ -527,7 +539,7 @@ end
--- (we walk all `e_shnum` headers regardless of how many names are requested, to find the .shstrtab first).
--- For frequent callers, pass the union of all needed sections in one call.
-- Can add `.debug_info` + `.debug_loc` + `.debug_str_offsets` to the list without writing a 2nd ELF walker.
--- @param elf_path Path
--- @param elf_path Path
--- @param section_names string[] -- list of section names to read
--- @return table<string, string>
function M.read_elf_sections(elf_path, section_names)
@@ -552,58 +564,69 @@ function M.read_elf_sections(elf_path, section_names)
return result
end
local file_size
do
f:seek("end", 0)
file_size = f:seek("cur", 0)
end
local adapter = {
read_u8_at = function(offset)
f:seek("set", offset)
local b = f:read(1)
if not b then return nil end
return b:byte()
end,
read_u16_at = function(offset)
f:seek("set", offset)
local b1 = f:read(1)
local b2 = f:read(1)
if not b1 or not b2 then return nil end
return b1:byte() + b2:byte() * 0x100
end,
read_u32_at = function(offset)
f:seek("set", offset)
local b1 = f:read(1)
local b2 = f:read(1)
local b3 = f:read(1)
local b4 = f:read(1)
if not b1 or not b2 or not b3 or not b4 then return nil end
return b1:byte() + b2:byte() * 0x100
+ b3:byte() * 0x10000 + b4:byte() * 0x1000000
end,
read_size = function() return file_size end,
}
-- Delegate the header parse + section walk to E.*.
local hdr, hdr_err = E.parse_elf32_headers(adapter)
if not hdr then
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] header parse failed: %s\n", tostring(hdr_err)))
-- Read the ELF32 header.
local header = f:read(M.ELF32.header_bytes)
if not header or #header < M.ELF32.header_bytes then
io.stderr:write("[elf_dwarf.read_elf_sections] ELF too small for ELF32 header\n")
f:close()
return result
end
local sections, walk_err = E.walk_sections(adapter, hdr)
if not sections then
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] section walk failed: %s\n", tostring(walk_err)))
-- Sanity-check magic + class + endianness.
if header:sub(M.ELF32.magic_offset + 1, M.ELF32.magic_offset + 0x04) ~= M.ELF32.magic then
io.stderr:write("[elf_dwarf.read_elf_sections] not an ELF file\n")
f:close()
return result
end
if header:byte(M.ELF32.class_offset + 1) ~= M.ELF32.class_elf32 then
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] not ELF32 (class=%d)\n", header:byte(M.ELF32.class_offset + 1)))
f:close()
return result
end
if header:byte(M.ELF32.endian_offset + 1) ~= M.ELF32.endian_little then
io.stderr:write("[elf_dwarf.read_elf_sections] not little-endian; unsupported\n")
f:close()
return result
end
-- Resolve the requested sections.
for _, s in ipairs(sections) do
if wanted[s.name] then
local bytes = E.read_section_bytes(adapter, s)
if bytes then result[s.name] = bytes end
-- Parse section-header table location + dimensions from the header.
local e_shoff = M.read_u32_le(header, M.ELF32.e_shoff_offset)
local e_shentsize = M.read_u16_le(header, M.ELF32.e_shentsize_offset)
local e_shnum = M.read_u16_le(header, M.ELF32.e_shnum_offset)
local e_shstrndx = M.read_u16_le(header, M.ELF32.e_shstrndx_offset)
-- Read the section-header string table (.shstrtab) so we can resolve section names from their `sh_name` offsets.
f:seek("set", e_shoff + e_shstrndx * e_shentsize)
local strtab_hdr = f:read(e_shentsize)
if not strtab_hdr or #strtab_hdr < e_shentsize then
io.stderr:write("[elf_dwarf.read_elf_sections] could not read .shstrtab header\n")
f:close()
return result
end
local strtab_offset = M.read_u32_le(strtab_hdr, M.ELF32.sh_offset_offset)
local strtab_size = M.read_u32_le(strtab_hdr, M.ELF32.sh_size_offset)
f:seek("set", strtab_offset)
local strtab = f:read(strtab_size) or ""
-- Walk all section headers; collect (offset, size) for the wanted names.
local function read_section_bytes(sh_offset, sh_size)
f:seek("set", sh_offset)
return f:read(sh_size) or ""
end
for sh_idx = 0, e_shnum - 1 do
f:seek("set", e_shoff + sh_idx * e_shentsize)
local sh = f:read(e_shentsize)
if not sh or #sh < e_shentsize then break end
local sh_name = M.read_u32_le(sh, M.ELF32.sh_name_offset)
local sh_offset = M.read_u32_le(sh, M.ELF32.sh_offset_offset)
local sh_size = M.read_u32_le(sh, M.ELF32.sh_size_offset)
-- Extract the name (null-terminated C string in strtab).
local name_end = strtab:find("\0", sh_name + 1, true) or (sh_name + 1)
local name = strtab:sub(sh_name + 1, name_end - 1)
if wanted[name] then
result[name] = read_section_bytes(sh_offset, sh_size)
end
end
@@ -620,87 +643,48 @@ end
--- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded.
--- - The `code_` prefix is stripped (MipsAtom_ macros emit bare atom names, no `code_` prefix).
--- - `st_size > 0` filter excludes undefined/imported symbols.
---
--- @param elf_path Path
--- @return table<string, {integer, integer}>
function M.read_nm(elf_path)
local addrs = {}
-- Existence check first; an empty or missing ELF returns an empty map.
if lfs.attributes(elf_path, "mode") ~= "file" then
-- Read .symtab + .strtab via the existing ELF walker (no subprocess).
local sections = M.read_elf_sections(elf_path, {".symtab", ".strtab"})
local symtab = sections[".symtab"]
local strtab = sections[".strtab"]
if not symtab or not strtab or #symtab == 0 or #strtab == 0 then
-- No symbol table (e.g. stripped ELF). Return empty.
return addrs
end
local f = io.open(elf_path, "rb")
if not f then
return addrs
end
-- Build the file adapter for E.*.
local file_size
do
f:seek("end", 0)
file_size = f:seek("cur", 0)
end
local adapter = {
read_u8_at = function(offset)
f:seek("set", offset)
local b = f:read(1)
if not b then return nil end
return b:byte()
end,
read_u16_at = function(offset)
f:seek("set", offset)
local b1 = f:read(1)
local b2 = f:read(1)
if not b1 or not b2 then return nil end
return b1:byte() + b2:byte() * 0x100
end,
read_u32_at = function(offset)
f:seek("set", offset)
local b1 = f:read(1)
local b2 = f:read(1)
local b3 = f:read(1)
local b4 = f:read(1)
if not b1 or not b2 or not b3 or not b4 then return nil end
return b1:byte() + b2:byte() * 0x100
+ b3:byte() * 0x10000 + b4:byte() * 0x1000000
end,
read_size = function() return file_size end,
}
-- Delegate the header + section walk to E.*.
local hdr, hdr_err = E.parse_elf32_headers(adapter)
if not hdr then
io.stderr:write(string.format("[elf_dwarf.read_nm] header parse failed: %s\n", tostring(hdr_err)))
f:close()
return addrs
end
local sections, walk_err = E.walk_sections(adapter, hdr)
if not sections then
io.stderr:write(string.format("[elf_dwarf.read_nm] section walk failed: %s\n", tostring(walk_err)))
f:close()
return addrs
end
-- E.collect_symbols returns every defined symbol (no binding filter).
-- The metaprogram then applies its STB_LOCAL / STB_GLOBAL + size>0 filter, matching `nm`'s default (external symbols only).
local symbols, sym_err = E.collect_symbols(adapter, sections)
if not symbols then
io.stderr:write(string.format("[elf_dwarf.read_nm] symbol collection failed: %s\n", tostring(sym_err)))
f:close()
return addrs
end
f:close()
for name, entry in pairs(symbols) do
-- High nibble of st_info = binding (STB_LOCAL=0, STB_GLOBAL=1, STB_WEAK=2).
-- math.floor(/16) is portable across LuaJIT 2.0/2.1 and plain Lua 5.x.
local binding = math.floor(entry.info / 16)
if (binding == 0 or binding == 1) and entry.size > 0 then
addrs[name] = { entry.value, entry.size }
-- Iterate the 16-byte ELF32 symtab entries.
-- Each entry (zero-based): st_name at 0, st_value at 4, st_size at 8, st_info at 12, st_other at 13, st_shndx at 14.
local SYM_ENTRY_BYTES = 0x10
local SYM_ST_NAME = 0x00
local SYM_ST_VALUE = 0x04
local SYM_ST_SIZE = 0x08
local SYM_ST_INFO = 0x0C
local n_syms = #symtab / SYM_ENTRY_BYTES
for i = 0, n_syms - 1 do
local entry_off = i * SYM_ENTRY_BYTES
local st_info = symtab:byte(entry_off + SYM_ST_INFO + 1)
-- High nibble = binding (STB_LOCAL=0, STB_GLOBAL=1, STB_WEAK=2).
-- Use math.floor(/16) instead of bit.rshift for LuaJIT 2.1 compat (LuaJIT's `>>` is 5.3+, but math.floor(x/16) works on all versions).
local binding = math.floor(st_info / 16)
if binding == 0 or binding == 1 then -- STB_LOCAL or STB_GLOBAL
local st_size = M.read_u32_le(symtab, entry_off + SYM_ST_SIZE)
if st_size > 0 then
local st_name_off = M.read_u32_le(symtab, entry_off + SYM_ST_NAME)
-- Extract the name from .strtab (null-terminated C string).
local name_end = strtab:find("\0", st_name_off + 1, true) or (st_name_off + 1)
local name = strtab:sub(st_name_off + 1, name_end - 1)
-- Filter: keep all symbol-table symbols (atoms emit their name as the bare `<name>` — MipsAtom_ macros strip the `code_` prefix).
-- The atoms_source_map pass already filters out non-atom symbols via the source-map.txt cross-ref.
if name and #name > 0 then
local st_value = M.read_u32_le(symtab, entry_off + SYM_ST_VALUE)
addrs[name] = { st_value, st_size }
end
end
end
end
@@ -838,11 +822,12 @@ end
--- * The `.debug_line` section may contain MULTIPLE line-program units
--- File indices are 1-based, **per unit**; we concatenate all units and the index ranges from 1..N₁ in unit 1, N₁+1..N₁+N₂ in unit 2, etc.
--- Per-unit indices (the way gcc emits them, and the way `DW_LNS_set_file` references them in the line program)
--- are returned via the `basename_to_index` map only when the unit boundary happens to align with the metaprogram's per-atom `inv.call_file`
--- are returned via the `basename_to_index` map only when the unit boundary happens to align with the metaprogram's per-atom
--- `inv.call_file` (true today for hello_joypad — the C unit is the LAST unit, and atom-side file indices fit 1-based).
--- * Per spec, the `.debug_line_str` section (DWARF5 §7.5.6) holds the strings referenced by `DW_FORM_line_strp`.
--- The legacy DWARF3 format embeds strings directly with null terminators. This helper handles BOTH.
--- * File entries may have multiple forms (gcc -gdwarf-5 with `DW_LNCT_directory_index` emits 2 forms: path + dir_index).
--- The helper supports:
--- * File entries may have multiple forms (gcc -gdwarf-5 with `DW_LNCT_directory_index`
--- emits 2 forms: path + dir_index). The helper supports:
--- - DW_FORM_line_strp (DWARF5; offset into .debug_line_str)
--- - DW_FORM_string (DWARF4-compat; inline null-terminated in .debug_line)
--- - DW_FORM_udata (ULEB128)
@@ -852,7 +837,9 @@ end
---
--- Behavior on failure: writes to stderr and returns nil.
--- Helpers consumed by `passes/dwarf_injection.lua::init_file_index_lookup(elf_path)` calls this once at pass start to populate the module-level `basename_to_index` map;
--- downstream `resolve_provenance_file_index(path)` consumers consult the map directly.
--- downstream `resolve_provenance_file_index(path)` consumers
--- (which replaced the former hardcoded `ATOM_SOURCE_FILE_INDEX` + `PROVENANCE_BASENAME_TO_FILE_INDEX` table per `conductor/tracks/dwarf_file_index_lookup_20260731/`)
--- consult the map directly.
---
--- @param elf_path string -- absolute path to the post-link ELF (typically the gcc-emitted `.elf` BEFORE dwarf_injector's splice; both shapes work since the splice preserves `.debug_line`)
--- @return table|nil, table|nil, table|nil
+96 -16
View File
@@ -1,41 +1,77 @@
# scripts/launch_pcsx_debug.ps1
#
# One-shot launcher for debug sessions: starts pcsx-redux with the .ps-exe
# loaded, the gdb stub enabled, AND the pcsx_debug_helper Lua plugin loaded
# so external CLI tools (gdb's `shell` command, etc.)
# can read GTE state via http://localhost:8080/api/v1/lua/gte
# (the gdb stub doesn't expose COP2 at all).
#
# usage:
# .\scripts\launch_pcsx_debug.ps1
# .\scripts\launch_pcsx_debug.ps1 -ExePath build\hello_gte.ps-exe
# .\scripts\launch_pcsx_debug.ps1 -HelperZip scripts\pcsx_debug_helper.zip
# loaded, the gdb stub enabled, the web server enabled, AND the
# pcsx_debug_helper Lua plugin loaded so external CLI tools can drive
# reloads via http://localhost:8080/api/v1/lua/reload.
#
# After launch:
# - gdb: target remote localhost:3333
# - web: curl http://localhost:8080/api/v1/lua/gte
# - web: POST http://localhost:8080/api/v1/lua/reload?mode=prime&...
#
# usage:
# .\scripts\launch_pcsx_debug.ps1
# .\scripts\launch_pcsx_debug.ps1 -ExePath build\hello_camera.ps-exe
# .\scripts\launch_pcsx_debug.ps1 -Cpu dynarec
# .\scripts\launch_pcsx_debug.ps1 -ElfPath build\hello_camera.elf
#
# Companion: scripts/debug_psyq.ps1 (bare launch — no .ps-exe, no helper).
[CmdletBinding()]
param(
[string]$PcsxPath = (Join-Path $PSScriptRoot '..\toolchain\pcsx-redux\vsprojects\x64\Release\pcsx-redux.exe'),
[string]$ExePath = (Join-Path $PSScriptRoot '..\build\hello_gte.ps-exe'),
[string]$ExePath = (Join-Path $PSScriptRoot '..\build\hello_camera.ps-exe'),
[string]$ElfPath = '',
[string]$HelperZip = (Join-Path $PSScriptRoot 'pcsx_debug_helper.zip'),
[int] $GdbPort = 3333,
[int] $WebPort = 8080
[int] $WebPort = 8080,
[ValidateSet('interpreter', 'dynarec')][string]$Cpu = 'interpreter'
)
$ErrorActionPreference = 'Stop'
# ── Derive -ElfPath when absent ──
# Convention: the .elf sits beside the .ps-exe with the same stem.
if ([string]::IsNullOrEmpty($ElfPath)) {
$exeFull = [System.IO.Path]::GetFullPath($ExePath)
$stem = [System.IO.Path]::GetFileNameWithoutExtension($exeFull)
$exeDir = [System.IO.Path]::GetDirectoryName($exeFull)
$ElfPath = Join-Path $exeDir "$stem.elf"
}
# ── Pre-checks ──
foreach ($p in @($PcsxPath, $ExePath, $HelperZip)) {
if (-not (Test-Path $p)) {
foreach ($p in @($PcsxPath, $ExePath, $ElfPath, $HelperZip)) {
if (-not (Test-Path -LiteralPath $p)) {
Write-Error "Missing: $p"
exit 1
}
}
# ── Reject a stale helper zip (Task 8) ──
# The helper zip must be newer than every .lua source that contributes
# to it. A stale zip means the running plugin does not match the on-disk
# source, which makes the reload contract meaningless.
$helperDir = Join-Path $PSScriptRoot 'pcsx_debug_helper'
$elf32Src = Join-Path $PSScriptRoot 'elf32.lua'
$sourceLuas = @(
(Join-Path $helperDir 'autoexec.lua'),
(Join-Path $helperDir 'reload.lua'),
$elf32Src
) | Where-Object { Test-Path -LiteralPath $_ }
$zipTime = (Get-Item -LiteralPath $HelperZip).LastWriteTime
$stale = $false
foreach ($src in $sourceLuas) {
$srcTime = (Get-Item -LiteralPath $src).LastWriteTime
if ($srcTime -gt $zipTime) {
Write-Error "helper zip is older than source: $src (zip=$($zipTime.ToString('o')) src=$($srcTime.ToString('o')); rerun build_psyq.ps1 to regenerate."
$stale = $true
}
}
if ($stale) {
exit 1
}
# Kill any existing pcsx-redux so the archive file isn't locked.
Get-Process pcsx-redux -ErrorAction SilentlyContinue | Stop-Process -Force
Start-Sleep -Seconds 2
@@ -44,17 +80,23 @@ Start-Sleep -Seconds 2
$absExe = [System.IO.Path]::GetFullPath($ExePath)
$absZip = [System.IO.Path]::GetFullPath($HelperZip)
$cpuFlag = if ($Cpu -eq 'dynarec') { '-dynarec' } else { '-interpreter' }
$args = @(
'-gdb', '-run'
'-loadexe', "`"$absExe`""
'-archive', "`"$absZip`""
'-webserver'
$cpuFlag
)
Write-Host "Launching pcsx-redux..." -ForegroundColor Cyan
Write-Host " ps-exe : $absExe"
Write-Host " elf : $ElfPath"
Write-Host " helper zip: $absZip"
Write-Host " gdb : localhost:$GdbPort"
Write-Host " web : localhost:$WebPort/api/v1/lua/gte"
Write-Host " web : localhost:$WebPort/api/v1/lua/reload"
Write-Host " cpu : $Cpu ($cpuFlag)"
Write-Host ""
Start-Process -FilePath $PcsxPath -ArgumentList $args | Out-Null
@@ -89,6 +131,44 @@ try {
Write-Host "Check the pcsx-redux Lua Console for debug cli messages." -ForegroundColor Yellow
}
# ── Prime the reload handler (Task 8) ──
# The reload handler keeps an internal ACTIVE manifest of the running
# ELF; reload requests fail with reload_not_primed until prime succeeds.
# We retry until the response carries ok=true or the launch deadline
# expires — the helper may not have finished registering handlers in the
# first web-poll cycle after the gte handler comes up.
$absElf = [System.IO.Path]::GetFullPath($ElfPath)
$encodedPath = [uri]::EscapeDataString($absElf)
$primeUri = "http://localhost:${WebPort}/api/v1/lua/reload?mode=prime&target=hello_camera&path=${encodedPath}"
Write-Host "Priming reload handler: $primeUri" -ForegroundColor Cyan
$primeDeadline = (Get-Date).AddSeconds(15)
$primeOk = $false
while ((Get-Date) -lt $primeDeadline) {
try {
$resp = Invoke-WebRequest -Method Post -Uri $primeUri -UseBasicParsing -TimeoutSec 5
$body = if ($resp.Content -is [byte[]]) {
[System.Text.Encoding]::UTF8.GetString([byte[]]$resp.Content)
} else {
[string]$resp.Content
}
$obj = $body | ConvertFrom-Json
if ($obj.ok) {
Write-Host "Prime OK: $(($obj | ConvertTo-Json -Compress))" -ForegroundColor Green
$primeOk = $true
break
} else {
Write-Host "Prime not yet ready: error=$($obj.error)" -ForegroundColor Yellow
}
} catch {
Write-Host "Prime request failed: $($_.Exception.Message)" -ForegroundColor Yellow
}
Start-Sleep -Milliseconds 500
}
if (-not $primeOk) {
Write-Warning "Prime did not return ok=true before the launch deadline. Reload requests will fail until the user primes manually."
}
Write-Host ""
Write-Host "pcsx-redux running. PIDs:" -ForegroundColor Cyan
Get-Process pcsx-redux | Select-Object Id, ProcessName | Format-Table
Get-Process pcsx-redux | Select-Object Id, ProcessName | Format-Table
+82
View File
@@ -0,0 +1,82 @@
# make_helper_zip.ps1
#
# Regenerate scripts/pcsx_debug_helper.zip from scripts/pcsx_debug_helper/.
# The archive contains exactly three entries at archive root:
#
# autoexec.lua
# elf32.lua (copied in from scripts/elf32.lua before packaging)
# reload.lua
#
# Determinism: CreateFromDirectory on the same set of files produces
# identical bytes. Verified by running the same command twice and
# asserting SHA-256 equality (see plan.md Task 6 Step 4).
#
# Performance: the implementation uses System.IO.Compression.ZipFile
# (BCL, in-process). Benchmarked: ~2 ms cold, ~2 ms warm on this
# workstation. Compress-Archive is rejected because its first call
# takes ~200 ms (assembly load) and subsequent calls take ~16 ms
# (process spawn per invocation). The 50 ms budget documented in
# plan.md Task 8 Step 3 excludes the compiler/assembler toolchain.
#
# Usage:
# pwsh -NoProfile -File scripts\make_helper_zip.ps1
#
# Optional -OutputPath switches the destination. Default is
# scripts/pcsx_debug_helper.zip next to the helper dir.
#
# Companion: scripts/pcsx_debug_helper/{autoexec,elf32,reload}.lua
# tests/reload_helper_zip_regen.ps1 (planned Task 8 verifier)
[CmdletBinding()]
param(
[string]$HelperDir = (Join-Path $PSScriptRoot 'pcsx_debug_helper'),
[string]$SourcesDir = $PSScriptRoot,
[string]$OutputPath = (Join-Path $PSScriptRoot 'pcsx_debug_helper.zip')
)
$ErrorActionPreference = 'Stop'
if (-not (Test-Path -LiteralPath $HelperDir)) {
throw "helper dir not found: $HelperDir"
}
# Stage elf32.lua into the helper dir so the in-process ZipFile walker
# picks it up alongside the helper-local files. elf32.lua is the shared
# ELF32 byte reader; the production reload.lua loads it through
# Support.extra.dofile("elf32.lua") at runtime.
$elf32Src = Join-Path $SourcesDir 'elf32.lua'
$elf32Dest = Join-Path $HelperDir 'elf32.lua'
if (-not (Test-Path -LiteralPath $elf32Src)) {
throw "elf32.lua not found at $elf32Src"
}
Copy-Item -LiteralPath $elf32Src -Destination $elf32Dest -Force
try {
# Remove any existing archive so CreateFromDirectory can write fresh.
# ZipFile.CreateFromDirectory throws if the destination exists.
if (Test-Path -LiteralPath $OutputPath) {
Remove-Item -LiteralPath $OutputPath -Force
}
# In-process zip; ~2 ms cold, ~2 ms warm. BCL compression matches
# Compress-Archive at CompressionLevel Optimal for these small files.
# Assembly is loaded once per pwsh.exe; the first run pays ~14 ms,
# subsequent runs pay ~0.2 ms.
Add-Type -AssemblyName System.IO.Compression.FileSystem
[System.IO.Compression.ZipFile]::CreateFromDirectory(
$HelperDir, $OutputPath,
[System.IO.Compression.CompressionLevel]::Optimal,
$false) | Out-Null
$sha = (Get-FileHash -LiteralPath $OutputPath -Algorithm SHA256).Hash
Write-Output ("[make_helper_zip] wrote {0} bytes, sha256={1}" -f `
(Get-Item -LiteralPath $OutputPath).Length, $sha)
Write-Output "[make_helper_zip] entries: autoexec.lua, elf32.lua, reload.lua"
}
finally {
# Remove the staged elf32.lua so the helper directory only contains
# the files the user expects to see there.
if (Test-Path -LiteralPath $elf32Dest) {
Remove-Item -LiteralPath $elf32Dest -Force
}
}
-355
View File
@@ -1,355 +0,0 @@
--- passes/auto_reg.lua — Per-phase automatic GPR allocator + gen/auto_reg.h emitter.
---
--- Reads the per-source + corpus-level `atom_auto_regs` + `phase_auto_regs` registries populated by `passes/scan_source.lua`.
--- Runs a deterministic first-fit allocator in the `R_T0..R_T7 + R_V0..R_V1` pool (10 physical GPRs).
--- Emits one `#define R_<Sym>_Code R_Tn_Code` per marker into per-directory `gen/auto_reg.h`.
---
--- User-pinned GPRs : The corpus's `register_alias_registry` is consulted to exclude GPRs the user has pinned via
--- `atom_reg` + `_Code` defs (e.g. carriers like `R_ResolveScratch = R_T4 atom_reg`).
--- These GPRs are unavailable to EVERY atom's source pool.
--- Carriers are preserved across atoms by context discipline and must never be reallocated.
--- Per-atom body parsing also catches alias references (R_<Alias>) and hardcoded R_Tn references,
--- so the user can write either `R_T4` or `R_ResolveScratch` in an atom body and the pass will
--- exclude R_T4 from that atom's pool.
---
--- Conflict detection: If the user hardcodes `R_Tn` in an atom body that shares a phase with an auto-reg that picked `R_Tn`,
--- emit `phase_register_clash` as an info finding (no build stop).
--- Should be unreachable after the user-pinning + body-parsing fix above; kept as a defensive safety net.
---
--- Pool exhaustion: If a phase declares more `R_<Sym>` mappings than the 10-register pool can hold,
--- emit `phase_register_pool_exhausted` as a build-stopping error.
--- @class AutoRegResult
--- @field outputs table[] -- {kind=, path=} entries
--- @field errors table[] -- {line=, msg=} entries (build-stops)
--- @field warnings table[] -- {line=, msg=} entries (build-continues)
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
--- ════════════════════════════════════════════════════════════════════════════
--- THE GPR ALLOCATION POOL — what is allocatable, and (more importantly) WHY
--- ════════════════════════════════════════════════════════════════════════════
---
--- The auto-reg pass picks physical GPRs for `atom_auto_reg(...)` / `phase_auto_reg(...)` markers.
--- It allocates from a FIXED 10-register pool.
--- This comment block makes the inclusion AND exclusion criteria obvious so a reader doesn't have
--- to grep lottes_tape.h + mips.h to understand the design.
---
--- ── WHAT'S IN THE POOL (10 GPRs, all caller-trash per the O32 ABI) ────────
--- R_T0..R_T7 (GPR codes 8..15), R_V0..R_V1 (GPR codes 2..3)
--- The workhorse of every atom body. The uesr should be aware of atom allocation across atoms they chain.
--- If they have a collision it means either they didn't saturate the register file optimally for a phase,
--- or the may have made the workload to large for the run.
---
--- ── WHAT'S NOT IN THE POOL — and WHY (the "obvious exclusions") ────────────
--- R_T9 (GPR code 25) — R_TapePtr, the tape instruction stream pointer.
--- Owned by the tape runtime (in tape_run / tape_run_a02_s07).
--- `rgcc(R_TapePtr)` register-variable ties the C compiler's view to $t9 across the whole tape_run.
--- The auto-reg pass MUST NOT clobber this; doing so would desync the C-side tape pointer from the
--- hardware pointer and crash on the next tape_run.
---
--- R_T8 (GPR code 24) — R_AtomJmp, the atom-jump register used by the 4-word yield handshake.
--- Every `mac_yield()` / `mac_yield_tail` does `load_word R_AtomJmp, R_TapePtr, 0` then
--- `jump_reg R_AtomJmp`. The auto-reg pass MUST NOT clobber this either, or the atom dispatcher breaks.
--- Owned by the tape runtime, same family as R_TapePtr.
---
--- R_AT (GPR code 1) — Assembler temporary. Reserved by the MIPS O32 ABI for pseudoinstruction expansion
--- (lottes_tape.h:86, mips.h:93). The ISA's psuedo instructions use it as a scratch temporary.
---
--- R_A0..A3 (codes 4..7) — Function arguments. Used in tape_run_a02_s07, see below.
--- R_S0..S7 (codes 16..23) — Callee-saved. Preserved across C-ABI calls by convention.
--- The `tape_run_a02_s07` variant clobbers them deliberately, but the default `tape_run` does NOT.
--- Kept out of POOL to preserve the conservative default.
--- Add them in a separate "big clobber" pool if/when needed.
---
--- R_K0/K1 (codes 26..27) — Kernel / interrupt handler reserves. Never touched by user code; OS-internal.
--- R_GP/SP/FP/RA (codes 28..31) — Stack frame + return-address. Owned by the C compiler; never allocatable.
--- R_0 (code 0) — Hardwired zero. Cannot be written.
---
local POOL = {
"R_T0", "R_T1", "R_T2", "R_T3",
"R_T4", "R_T5", "R_T6", "R_T7",
"R_V0", "R_V1",
}
-- Map from integer MIPS GPR code (the `code` field on AliasEntry) to the physical GPR ident in POOL.
-- The standard MIPS O32 ABI register numbering matches mips.h's R_*_Code #defines (mips.h).
-- Only the POOL entries matter for auto_reg — non-pool aliases
-- (R_AT=1, R_A0..A3=4..7, R_T8=24, R_T9=25, R_K0/K1=26..27, R_GP/SP/FP/RA=28..31)
-- are deliberately omitted — see the comment block above for the WHY of each exclusion.
local INT_CODE_TO_POOL_GPR = {
[2] = "R_V0", [3] = "R_V1",
[8] = "R_T0", [9] = "R_T1", [10] = "R_T2", [11] = "R_T3",
[12] = "R_T4", [13] = "R_T5", [14] = "R_T6", [15] = "R_T7",
}
-- Stable sort for deterministic allocation order.
local function stable_sort_keys(tbl)
local keys = {}
for k in pairs(tbl) do keys[#keys + 1] = k end
table.sort(keys)
return keys
end
-- Allocate one phase's auto-reg mappings.
-- Returns (allocated_map, errors). On pool exhaustion, errors is populated and the function halts.
local function allocate_phase(phase_label, decls)
-- Deep-copy POOL into a fresh sequence table. The original `table.unpack and table.unpack(POOL) or { unpack(POOL) }`
-- idiom wraps the unpacked values in a single inner table under LuaJIT 5.1 (`table.unpack` is nil; the `or` returns one value),
-- which corrupts the pool into `{ {R_T0, R_T1, ...} }` — making `table.remove(pool, 1)` return the inner table on iteration.
local pool = {}
for i = 1, #POOL do pool[i] = POOL[i] end
local result = {}
local errors = {}
for _, sym in ipairs(stable_sort_keys(decls)) do
local next_gpr = table.remove(pool, 1)
if not next_gpr then
errors[#errors + 1] = {
line = 0,
msg = string.format("phase_register_pool_exhausted: "
.. "phase '%s' requested symbol '%s' but the pool has no remaining registers "
.. "(max 10 per phase: R_T0..R_T7 + R_V0..R_V1). Split the phase or use hardcoded GPRs."
, phase_label, sym),
}
return result, errors
end
result[sym] = next_gpr
end
return result, errors
end
-- Build two projections from corpus.register_alias_registry:
-- user_pinned -- { [physical_gpr_ident] = true } -- GPRs unavailable to auto_reg globally (wave-context carriers, file-scope pinned aliases)
-- alias_to_gpr -- { [alias_ident] = physical_gpr_ident } -- for body parsing
-- Both projections are derived from the same set of entries: every AliasEntry in register_alias_registry has `has_atom_reg = true`
-- (only those entries are added to the registry; see passes/scan_source.lua parse_enum_entry).
-- Each entry's `code` is the integer MIPS GPR number (0..31); INT_CODE_TO_POOL_GPR translates it back to the physical GPR ident.
-- Aliases whose `code` points to a non-POOL GPR (e.g. R_S0, R_T8, R_K1) are ignored —
-- they don't affect the auto_reg pool, and they're already excluded from POOL above.
local function build_user_pins(corpus)
local user_pinned = {}
local alias_to_gpr = {}
if not corpus.register_alias_registry then return user_pinned, alias_to_gpr end
for alias_name, alias_entry in pairs(corpus.register_alias_registry) do
if alias_entry.has_atom_reg and alias_entry.code then
local gpr = INT_CODE_TO_POOL_GPR[alias_entry.code]
if gpr then
user_pinned[gpr] = true
alias_to_gpr[alias_name] = gpr
end
end
end
return user_pinned, alias_to_gpr
end
-- Find every physical GPR referenced in the atom body, via EITHER:
-- (a) A hardcoded physical GPR ident (R_T\d+|R_V\d+|R_A\d+|R_S\d+) — the existing regex;
-- (b) An alias ident (R_<Alias>) resolved via alias_to_gpr back to its physical GPR ident.
-- Returns { [physical_gpr_ident] = count }. Clash-detection and source-pool-exclusion logic
-- only needs the presence of each GPR (boolean test), but keeping count preserves the
-- original find_hardcoded_rn shape so callers can switch without churn.
-- The alias pattern is sorted lexicographically to keep the regex deterministic.
local function find_used_gprs(body_text, alias_to_gpr)
local found = {}
-- (a) Hardcoded physical GPRs (R_T0..R_T7, R_V0..R_V1, R_A0..R_A3, R_S0..R_S7).
for gpr in body_text:gmatch("(R_T%d+|R_V%d+|R_A%d+|R_S%d+)") do
found[gpr] = (found[gpr] or 0) + 1
end
-- (b) Alias references (R_<Alias>) resolved to physical GPRs via the registry.
-- Sorted by name so the regex is byte-stable across runs.
if alias_to_gpr and next(alias_to_gpr) then
local aliases = {}
for alias_name in pairs(alias_to_gpr) do
aliases[#aliases + 1] = alias_name
end
table.sort(aliases)
local pattern = "(" .. table.concat(aliases, "|") .. ")"
for alias_name in body_text:gmatch(pattern) do
local gpr = alias_to_gpr[alias_name]
if gpr and not found[gpr] then
found[gpr] = 1
end
end
end
return found
end
-- Emit one gen/auto_reg.h header per directory.
local function emit_auto_reg_h(out_dir, dir, sources, mappings)
if not mappings or next(mappings) == nil then return end
local out_path = out_dir .. "/" .. "auto_reg.h"
duffle.ensure_dir(out_dir)
local lines = {
"#ifdef INTELLISENSE_DIRECTIVES",
"#pragma once",
"#endif",
"// Auto-generated by ps1_meta.lua (passes/auto_reg.lua) — DO NOT EDIT",
"// Directory: " .. dir:gsub("/", "\\"),
}
for _, src in ipairs(sources) do
lines[#lines + 1] = "// source: " .. src.path
end
lines[#lines + 1] = "// Per-phase register allocations resolved by the lua pass."
lines[#lines + 1] = "// R_<Sym>_Code = <chosen GPR's _Code constant> for every marker in this directory."
lines[#lines + 1] = ""
for _, sym in ipairs(stable_sort_keys(mappings)) do
local gpr = mappings[sym]
local gpr_code = gpr .. "_Code"
lines[#lines + 1] = "#define " .. sym .. "_Code " .. gpr_code
end
lines[#lines + 1] = ""
duffle.write_file_lf(out_path, table.concat(lines, "\n") .. "\n")
print(" -> " .. out_path)
return out_path
end
-- ════════════════════════════════════════════════════════════════════════════
-- Pass entry
-- ════════════════════════════════════════════════════════════════════════════
local M = {}
--- @param ctx PassCtx
--- @return AutoRegResult
function M.run(ctx)
local outputs = {}
local errors = {}
local warnings = {}
local corpus = ctx.shared and ctx.shared.corpus
if type(corpus) ~= "table" then
error("auto_reg.run requires ctx.shared.corpus", 0)
end
-- 0. Build the user-pinned GPR exclusion set + alias-to-GPR resolution map.
-- Wave-context carriers (e.g. `R_ResolveScratch = R_T4 atom_reg` in hello_camera.atom.c)
-- MUST NOT be allocated to any auto-reg marker — they're preserved across atoms by the wave-context discipline.
-- The corpus's register_alias_registry is the source of truth for these opt-in pins.
-- Body references to those aliases (via alias_to_gpr) are also excluded on a per-atom basis in step 2 below.
local user_pinned, alias_to_gpr = build_user_pins(corpus)
-- 1. Allocate phase pools first (phase declarations take precedence over per-atom declarations).
local phase_allocations = {}
for phase_label, decls in pairs(corpus.phase_auto_regs or {}) do
local mapping, errs = allocate_phase(phase_label, decls)
for sym, gpr in pairs(mapping) do
phase_allocations[phase_label] = phase_allocations[phase_label] or {}
phase_allocations[phase_label][sym] = gpr
end
for _, e in ipairs(errs) do
errors[#errors + 1] = e
end
end
-- 2. Allocate per-atom auto-regs. If the atom scope matches a phase, reuse the phase pool.
-- Otherwise, allocate a private pool for the atom.
-- The phase membership is in `corpus.atom_phases[phase_label].atoms` (an array of atom names declared via `atom_phase(<phase>)`
-- in the atom's `atom_info` line). Build a reverse map `atom_name -> phase_label` so the lookup is O(1) per atom scope.
local atom_name_to_phase = {}
for phase_label, entry in pairs(corpus.atom_phases or {}) do
for _, atom_name in ipairs(entry.atoms or {}) do
atom_name_to_phase[atom_name] = phase_label
end
end
local atom_allocations = {}
for atom_scope, decls in pairs(corpus.atom_auto_regs or {}) do
local phase_label = atom_name_to_phase[atom_scope]
-- Build the atom's source pool: start with the full POOL, subtract:
-- (a) every GPR already committed (phase allocations + prior atom allocations)
-- (b) every USER-PINNED GPR (wave-context carriers + file-scope pinned aliases)
-- (c) every GPR referenced in the atom's body — either hardcoded R_X or alias R_Xxx
-- (the latter resolved via alias_to_gpr; this catches cases where the user wrote R_ResolveScratch instead of R_T4 directly)
-- Atoms whose scope matches a phase share the global pool with the phase allocations;
-- the original `source_pool = phase_allocations[phase_label]` form used the phase
-- allocation MAP as a pool, but that map has no array part, so `table.remove(source_pool, 1)`
-- returned nil and every atom-with-phase marker errored with `phase_register_pool_exhausted`.
local used = {}
for _, m in pairs(phase_allocations) do for _, gpr in pairs(m) do used[gpr] = true end end
for _, m in pairs(atom_allocations) do for _, gpr in pairs(m) do used[gpr] = true end end
-- (c) Body references — scan the atom body for hardcoded + alias-resolved GPRs.
-- Folded into `used` so the source_pool exclusion is a single check.
local atom = corpus.atoms_by_name and corpus.atoms_by_name[atom_scope]
if atom and atom.body then
local body_used = find_used_gprs(atom.body, alias_to_gpr)
for gpr in pairs(body_used) do used[gpr] = true end
end
local source_pool = {}
for _, gpr in ipairs(POOL) do
-- Exclude (a) prior commitments, (b) USER-PINNED GPRs (wave-context carriers
-- declared via atom_reg + _Code defs, preserved across atoms globally).
if not used[gpr] and not user_pinned[gpr] then
source_pool[#source_pool + 1] = gpr
end
end
local result = {}
for _, sym in ipairs(stable_sort_keys(decls)) do
local next_gpr = table.remove(source_pool, 1)
if not next_gpr then
errors[#errors + 1] = {
line = 0,
msg = string.format("phase_register_pool_exhausted: atom '%s' requested symbol '%s' "
.. "but no free registers remain in its scope pool."
, atom_scope, sym),
}
else
result[sym] = next_gpr
end
end
atom_allocations[atom_scope] = result
end
-- 3. Conflict-with-hardcoded detection (defensive — should be unreachable now).
-- The source_pool exclusion in step 2 (b) + (c) already accounts for both user-pinned GPRs
-- and body-referenced GPRs (hardcoded R_Tn OR alias R_<Alias>).
-- An auto-reg allocation that matched an existing body reference would be impossible by construction.
-- This warning is kept as a defensive safety net for cases the body scanner might miss
-- (e.g. macros that expand to register references the scanner cannot resolve).
-- For each resolved (scope, sym) -> R_Tn mapping, scan the atom body source for used GPRs.
for atom_scope, decls in pairs(atom_allocations) do
local atom = corpus.atoms_by_name and corpus.atoms_by_name[atom_scope]
if atom and atom.body then
local used_in_body = find_used_gprs(atom.body, alias_to_gpr)
for sym, allocated_gpr in pairs(decls) do
if used_in_body[allocated_gpr] and used_in_body[allocated_gpr] > 0 then
warnings[#warnings + 1] = {
line = atom.line or 0,
msg = string.format("phase_register_clash: atom '%s' has hardcoded '%s' in its body AND an auto-reg marker '%s' "
.. "that was allocated to '%s' (same phase). Resolve by removing the hardcoded reference or renaming the auto-reg."
, atom_scope, allocated_gpr, sym, allocated_gpr),
}
end
end
end
end
-- 4. Emit per-directory gen/auto_reg.h.
-- For each source directory that has atom_auto_regs or phase_auto_regs entries, emit one header.
local sources_by_dir = corpus.sources_by_dir or {}
for dir, sources in pairs(sources_by_dir) do
local per_dir_mappings = {}
for _, src in ipairs(sources) do
-- Collect every (sym -> gpr) entry that originated from a source in this directory.
-- `src.scan.atom_auto_regs` is keyed by ATOM SCOPE NAME; `pairs(t)` iterates KEYS so `scope_name` here is the scope ident (e.g. "cube_g4_face").
-- The previous `for _, scan_atom_auto` form silently assigned the VALUE (a `{sym = sym}` table) to the variable,
-- which made `atom_allocations[scan_atom_auto]` a table-indexed lookup that never resolved.
for scope_name in pairs(src.scan and src.scan.atom_auto_regs or {}) do
for sym, gpr in pairs(atom_allocations[scope_name] or {}) do
per_dir_mappings[sym] = gpr
end
end
for scope_name in pairs(src.scan and src.scan.phase_auto_regs or {}) do
for sym, gpr in pairs(phase_allocations[scope_name] or {}) do
per_dir_mappings[sym] = gpr
end
end
end
local out_dir = dir .. "/gen"
local out_path = emit_auto_reg_h(out_dir, dir, sources, per_dir_mappings)
if out_path then outputs[#outputs + 1] = { auto_reg_h = out_path } end
end
return { outputs = outputs, errors = errors, warnings = warnings }
end
return M
+57 -56
View File
@@ -3,12 +3,9 @@
--- Ownership: `corpus.word_counts`, `corpus.components`, and `corpus.component_body_index`.
--- Scanner owns `declaration_comment` and `debug_skip` on each declaration record; this pass projects both forward.
---
--- Reads the pre-scanned SourceScan payload from `duffle.scan_source` for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations (kind="comp_bare" / "comp_proc"),
--- Reads the pre-scanned SourceScan payload from `duffle.scan_source` for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations,
--- then resolves the function-args string from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration via a backward walk.
---
--- `MipsAtom_Proc_(X, ab, { body })` declarations (kind="atom_proc") are ATOMS, not components, and are deliberately excluded —
--- atoms get emitted via `tb_emit(tb, code_<name>)` linker symbols, not inlined as `mac_*` macros.
---
--- Emits one `gen/macs.h` per *immediate source directory* with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
--- All sources inside the same directory contribute to the same file (per-directory aggregation).
--- The directory itself is the namespace, so the filename does not repeat the module name.
@@ -79,7 +76,7 @@ local MACS_FILENAME = "macs.h"
--- @field args string|nil -- Function-args string (function form only)
--- @field line integer -- Source line of the declaration
--- @field comment string|nil -- Scanner-owned `declaration_comment`; the components pass reads it from the scanner record
--- @field kind string -- "comp_bare" | "comp_proc" (atom_proc is NOT a component — see `project_components`)
--- @field kind string -- "comp_bare" | "comp_proc"
--- @field debug_skip boolean -- Mirror of `a.debug_skip` (scanner-owned); true iff a bare `atom_dbg_skip` marker immediately preceded the declaration
-- ════════════════════════════════════════════════════════════════════════════
@@ -96,21 +93,48 @@ local M = {}
-- so this file reads it forward rather than re-walking the source.
-- ════════════════════════════════════════════════════════════════════════════
--- Find the args of the function declaration that immediately precedes a `MipsAtomComp_Proc_` invocation.
--- Find the args of the function declaration that immediately precedes a `MipsAtomComp_Proc_` invocation of the given name.
--- Returns the args string (e.g., `"U4 off, U4 code, U1 r, U1 g, U1 b"`) or nil if no function declaration is found.
---
--- After the `sym` arg was dropped from MipsAtomComp_Proc_, the component name
--- and the args both come from the preceding `FI_ Slice_MipsCode ac_X(args)`
--- declaration. The shared `duffle.find_function_decl_for` helper does the
--- backward walk; this function returns just the args.
--- Convention: function form is
--- `FI_ Slice_MipsCode ac_X(args) MipsAtomComp_Proc_(ac_X, { body })`
--- We find the LAST occurrence of `"ac_X("` before `before_pos` and extract the args from inside the parens.
--- We then verify the preceding context ends with `Slice_MipsCode`
--- (the function-decl keyword with possible qualifiers between).
---
--- @param source string
--- @param name string (retained for signature stability; unused — the walk derives the name)
--- @param name string
--- @param before_pos integer
--- @return string|nil
local function find_function_args_for(source, name, before_pos)
local _, args_inner = duffle.find_function_decl_for(source, before_pos, #MIPS_ATOM)
return args_inner
-- Find the LAST occurrence of `name + "("` in `source[1..before_pos]`.
local name_open = name .. "("
local last_idx = nil
local scan_pos = 1
while true do
-- Pass `before_pos + 1` so string.find only returns positions < before_pos + 1
-- (string.find's 4th arg `plain` is true; we use the 3rd arg `init` for the upper bound).
local found = source:find(name_open, scan_pos, true)
if not found or found >= before_pos then break end
last_idx = found
scan_pos = found + #name_open
end
if not last_idx then return nil end
-- Verify the preceding context ends with "MipsAtom" (with possible qualifiers between).
local before = source:sub(1, last_idx - 1)
local trimmed = duffle.trim(before)
if trimmed:sub(-#MIPS_ATOM) ~= MIPS_ATOM then
-- Preceding context is not a function declaration.
return nil
end
local open_paren = last_idx + #name -- position of "("
-- scan: MipsAtom ac_X(
local inner = duffle.read_parens(source, open_paren)
-- scan: MipsAtom ac_X(<args>)
if not inner then return nil end
return inner
end
-- ════════════════════════════════════════════════════════════════════════════
@@ -176,16 +200,7 @@ end
local function project_components(source, scan)
local out = {}
for _, a in ipairs(scan.atoms) do
-- Only `MipsAtomComp_(ac_X)` (kind="comp_bare") and `MipsAtomComp_Proc_(ac_X, ...)` (kind="comp_proc")
-- are COMPONENTS — they get inlined via `mac_<name>` aliases inside atom bodies.
-- `MipsAtom_Proc_` (kind="atom_proc") is an ATOM (ends with `mac_yield()`); it gets emitted via
-- `tb_emit(tb, code_<name>)` (linker symbol), NOT inlined as a macro. Including `atom_proc` here
-- would incorrectly emit `mac_<name>` aliases for atoms, polluting `gen/macs.h`.
-- See `docs/duffle_dsl_primer.md` §"mac_* aliases" for the contract.
if a.kind == "comp_bare" or a.kind == "comp_proc" then
-- Function-args lookup is meaningful for `MipsAtomComp_Proc_` components
-- (the macro sits inside `FI_ Slice_MipsCode ac_X(...)`); the alias expansion
-- discards the `ab` (atom-builder) arg the same way both forms do.
local args = find_function_args_for(source, a.raw_name, a.ident_pos)
-- Comment ownership: scan_source.lua stamps `declaration_comment` on the record by walking backward past any associated bare marker.
-- The pass reads `declaration_comment` directly.
@@ -284,9 +299,7 @@ local function word_count_rec(name, comp_by_name, wc, cache)
local trimmed = t.tok
if trimmed ~= "" then
local lookup = strip_mac_prefix(duffle.read_ident(trimmed, 1))
if lookup == "atom_label" or lookup == "atom_offset" then
-- Pure metaprogram anchors; emit zero words.
elseif lookup and comp_by_name[lookup] then
if lookup and comp_by_name[lookup] then
-- It's a `mac_X(...)` call. Recurse.
n = n + word_count_rec(lookup, comp_by_name, wc, cache)
elseif lookup and wc and wc[lookup] then
@@ -377,7 +390,8 @@ local function cycle_cost_rec(name, comp_by_name, latency, cache)
end
--- (internal) Recursive GP0 prim-buffer contribution. Count `store_word` / `store_half` / `store_byte`
--- calls in the component body that target `R_PrimCursor` (these are the RAM-side prim-buffer words the macro contributes), recursing through nested `mac_*` calls.
--- calls in the component body that target `R_PrimCursor` (these are the
--- RAM-side prim-buffer words the macro contributes), recursing through nested `mac_*` calls.
--- Only `R_PrimCursor`-targeting stores count. Stores targeting other registers (e.g. `R_OtBase`, heap pointers) are not prim-buffer contributions.
--- @param name string
--- @param comp_by_name table<string, Component>
@@ -459,25 +473,12 @@ local function split_comment_lines(s)
end
--- Determine the macro signature: function-args list (function form) or variadic-ignored (bare form).
--- For `MipsAtomComp_Proc_` components, the leading `ab` (atom-builder) arg is dropped:
--- the generated `mac_<name>` macros are inline-expansion aliases for baked atoms; their bodies don't reference `ab`
--- (the builder is only consumed by the procedural `atombuilder_unroll` line that `MipsAtomComp_Proc_` appends after the body).
--- Inline callers therefore don't need to thread a builder context.
--- @param args_str string|nil
--- @return string
local function signature_from_args(args_str)
local arg_names = extract_arg_names(args_str)
if arg_names and #arg_names > 0 then
-- Drop the leading `ab` (atom-builder) first arg if present.
-- Convention: `MipsAtomComp_Proc_` components always declare `ab` as the first function-arg
-- (type `MipsAtomBuilder_R`), mirroring the macro signature in `lottes_tape.h`.
if arg_names[1] == "ab" then
table.remove(arg_names, 1)
end
if #arg_names > 0 then
return table.concat(arg_names, ", ")
end
return "..." -- `ab` was the only arg; fall through to variadic
return table.concat(arg_names, ", ")
end
return "..."
end
@@ -519,7 +520,7 @@ local function build_component_lines(c, counts)
-- Marker comment: emitted once for every skipped component.
-- The marker is scanner-owned (declared by `atom_dbg_skip` immediately before the declaration in the source);
-- This pass projects `c.debug_skip` and emits the marker as a generated comment.
-- the components pass projects `c.debug_skip` and emits the marker as a generated comment.
if c.debug_skip then
lines[#lines + 1] = "/* atom_dbg_skip */"
end
@@ -553,8 +554,8 @@ end
--- Build the boilerplate header lines (the `#ifdef INTELLISENSE_DIRECTIVES` block,
--- the `// Auto-generated` comment, the `// Source:` line, and the self-contained `WORD_COUNT` macro definition).
--- @param dir string -- Absolute source directory
--- @param sources SourceFile[] -- Sources contributing to this directory (for the header comment)
--- @param dir string -- the absolute source directory
--- @param sources SourceFile[] -- sources contributing to this directory (for the header comment)
--- @return string[]
local function header_boilerplate(dir, sources)
local source_lines = { "// Directory: " .. duffle.to_absolute_path(dir) .. "/" }
@@ -585,9 +586,9 @@ end
--- Compute the per-directory output path for `.macs.h`.
--- e.g. any source in `code/duffle/` produces `code/duffle/gen/macs.h` regardless of source filename.
--- The directory name is the namespace; the filename does not repeat it.
--- @param dir string -- Absolute source directory
--- @return string -- Output directory
--- @return string -- Full output path
--- @param dir string -- the absolute source directory
--- @return string -- the output directory
--- @return string -- the full output path
local function compute_macs_h_path(dir)
local out_dir = dir .. "/" .. GEN_SUBDIR
local out_path = out_dir .. "/" .. MACS_FILENAME
@@ -597,11 +598,11 @@ end
--- Emit a per-directory `.macs.h` header with the aggregated `mac_X` macros + `WORD_COUNT` entries.
--- Writes in BINARY mode so LF line endings are preserved (the git blob is LF; Windows text-mode would emit CRLF and break the byte-identical diff).
--- @param ctx PassCtx
--- @param dir string -- Absolute source directory
--- @param sources SourceFile[] -- Sources contributing to this directory (for the header comment)
--- @param components Component[] -- Aggregated components from all sources in this directory
--- @param counts table<string, integer> -- Precomputed word counts (from count_all_components)
--- @return string|nil -- Path to the written file (nil if no components)
--- @param dir string -- the absolute source directory
--- @param sources SourceFile[] -- sources contributing to this directory (for the header comment)
--- @param components Component[] -- aggregated components from all sources in this directory
--- @param counts table<string, integer> -- precomputed word counts (from count_all_components)
--- @return string|nil -- path to the written file (nil if no components)
local function emit_component_macros_h(ctx, dir, sources, components, counts)
if #components == 0 then return nil end
local out_dir, out_path = compute_macs_h_path(dir)
@@ -640,11 +641,11 @@ local function update_canonical_word_counts(corpus, components, counts)
end
--- @class ComponentDef
--- @field name string -- Bare name (without ac_/mac_ prefix)
--- @field line integer -- Definition source line (line of `MipsAtomComp_(ac_X)` / `MipsAtomComp_Proc_(ac_X, ...)`)
--- @field path string -- Absolute source path of the definition
--- @field kind string -- "comp_bare" | "comp_proc" (atom_proc is NOT a component)
--- @field debug_skip boolean -- Mirror of the scanner-owned `a.debug_skip`; consumers read this directly
--- @field name string -- bare name (without ac_/mac_ prefix)
--- @field line integer -- definition source line (line of `MipsAtomComp_(ac_X)` / `MipsAtomComp_Proc_(ac_X, ...)`)
--- @field path string -- absolute source path of the definition
--- @field kind string -- "comp_bare" | "comp_proc"
--- @field debug_skip boolean -- mirror of the scanner-owned `a.debug_skip`; consumers read this directly
--- (internal) Populate `corpus.components` with this source's components-by-name map.
--- First declaration wins; later declarations of the same bare name are dropped and recorded as a collision via `corpus.collisions` (kind = "component").
+19 -20
View File
@@ -703,9 +703,9 @@ end
--- `{comp_name, call_file, call_line, comp_file, comp_line, start_pos, end_pos, body_lines, debug_skip}`. `body_lines[k]`
--- is the k-th word's source line within the component body.
---
--- @param corpus table -- From `ctx.shared.corpus`
--- @param corpus table -- the corpus from `ctx.shared.corpus`
--- @param addrs table -- ELF symbols keyed by atom name from `elf_dwarf.read_nm`
--- @return table[] -- List of {name, addr, size_bytes, words, entries, invocations, debug_skip?}
--- @return table[] -- list of {name, addr, size_bytes, words, entries, invocations, debug_skip?}
local function build_atom_table(corpus, addrs)
-- Cross-ref: keep only atoms present in BOTH the nm symbol table AND `corpus.atoms_by_name`. Output is sorted by ascending addr.
local atoms_by_name = corpus.atoms_by_name or {}
@@ -834,10 +834,10 @@ end
--- This is intentional: silently falling back to a hardcoded GPR would mask the missing opt-in.
---
--- Pre-tokenized: `body_tokens` is the scan-source pass's pre-split list of top-level statements (each entry is a single `load_*` call or other statement).
--- @param body_tokens table[] -- The atom's pre-tokenized body statements (from atom.body_tokens)
--- @param binds_name string -- Expected Binds_X name (skip pairs with mismatching binds)
--- @param registries table -- Merged registries from collect_per_source_registries
--- @return table[] -- List of {reg = <MIPS index>, field = <field name>}
--- @param body_tokens table[] -- the atom's pre-tokenized body statements (from atom.body_tokens)
--- @param binds_name string -- expected Binds_X name (skip pairs with mismatching binds)
--- @param registries table -- merged registries from collect_per_source_registries
--- @return table[] -- list of {reg = <MIPS index>, field = <field name>}
local function parse_body_load_pairs(body_tokens, binds_name, registries)
local pairs = {}
local reg_index_by_name = (registries and registries.register_alias_registry) or {}
@@ -880,9 +880,9 @@ end
--- The piece chain uses (DW_OP_regN, DW_OP_piece, ULEB128(field_size)).
---
--- Binds fields come from `scan.binds`; the per-source `scan.binds[i].fields` already carries the typed-field record after the scan-source generalization.
--- @param corpus table -- From `ctx.shared.corpus`
--- @param atom_table table[] -- Cross-ref'd atom table from build_atom_table
--- @param registries table -- Merged registries from collect_per_source_registries
--- @param corpus table -- the corpus from `ctx.shared.corpus`
--- @param atom_table table[] -- the cross-ref'd atom table from build_atom_table
--- @param registries table -- merged registries from collect_per_source_registries
--- @return table, table -- (rbind_atoms, rbind_structs)
local function parse_rbind_atoms(corpus, atom_table, registries)
registries = registries or {}
@@ -944,7 +944,7 @@ local function parse_rbind_atoms(corpus, atom_table, registries)
binds = ai.binds,
fields = struct.fields, -- {name, offset} from scan.binds
bytes = struct.bytes,
regs = pairs, -- Ordered list of {reg, field}
regs = pairs, -- ordered list of {reg, field}
info_line = ai.info_line,
}
table.insert(struct.atom_names, atom_name)
@@ -992,7 +992,7 @@ local function build_dwarf_line_section(existing, atom_table)
while unit_pos < #existing do
if unit_pos + 4 > #existing then return existing end
local unit_length = elf_dwarf.read_u32_le(existing, unit_pos)
if unit_length == elf_dwarf.dw_dwarf32_terminator then return existing end
if unit_length == elf_dwarf.ELF32.dw_dwarf32_terminator then return existing end
local unit_end_excl = unit_pos + 4 + unit_length
if unit_end_excl > #existing then return existing end
last_pos, last_length, last_end = unit_pos, unit_length, unit_end_excl
@@ -1036,13 +1036,13 @@ local function build_dwarf_aranges_section(existing, atom_table)
-- We bump the unit's length field accordingly.
--
-- Unit structure (DWARF4 §7.21):
-- unit_length (4)
-- version (2)
-- unit_length (4)
-- version (2)
-- debug_info_offset (4) -- CU DIE offset in .debug_info
-- address_size (1)
-- segment_size (1)
-- entries... (4-byte addr + 4-byte length)
-- terminator (8 bytes: addr=0, length=0)
-- address_size (1)
-- segment_size (1)
-- entries... (4-byte addr + 4-byte length)
-- terminator (8 bytes: addr=0, length=0)
-- Walk all units and emit each one (preserving existing structure).
-- For the LAST unit, replace the terminator with my entries + new term.
@@ -1053,7 +1053,7 @@ local function build_dwarf_aranges_section(existing, atom_table)
while i < #existing do
-- Read this unit's length.
local ul = elf_dwarf.read_u32_le(existing, i)
if ul == elf_dwarf.dw_dwarf32_terminator then
if ul == elf_dwarf.ELF32.dw_dwarf32_terminator then
-- DWARF64 marker - not supported.
io.stderr:write("[dwarf_injection] WARN: .debug_aranges contains a DWARF64 marker (0xFFFFFFFF); the 64-bit extension is not supported by this metaprogram; passing through unchanged\n")
return existing
@@ -1780,8 +1780,7 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
emit(uleb128(ABBREV_TYPED_VIEW_POINTER)) -- DW_TAG_pointer_type (abbrev 110; NOT 9; void chain target)
emit(elf_dwarf.write_u32_le(ref4_of(void_chain_offset))) -- 4-byte ref4: points at the void base_type's tag byte
-- type_chain_offsets["void|1"] is what step (f) of the per-RR_<R_Name> chain looks up.
type_chain_offsets["void|1"] = void_chain_offset -- both the base_type offset and the pointer_type are emitted consecutively; the OUTERMOST is the pointer_type.
-- The variable's DW_AT_type must reference the pointer_type, not the base_type. Patch below.
type_chain_offsets["void|1"] = void_chain_offset -- both the base_type offset and the pointer_type are emitted consecutively; the OUTERMOST is the pointer_type. The variable's DW_AT_type must reference the pointer_type, not the base_type. Patch below.
-- Capture the pointer_type's offset (the last-thing-emitted DIE start) and overwrite the lookup.
-- The pointer_type was emitted as: uleb(9) (1 byte) + 4-byte ref4 = 5 bytes. Its tag byte is at void_chain_offset + 8 (the base_type's 8 bytes: 1 tag + 5 name + 1 byte_size + 1 encoding).
local ptr_void_offset = void_chain_offset + 8
+3 -3
View File
@@ -188,11 +188,11 @@ function M.run(ctx)
if type(corpus.source_order) ~= "table" then error("emission_model: ctx.shared.corpus.source_order is required", 0) end
-- Project once, collect errors + warnings for one atom.
-- Kind must be one of: atom | atom_proc | raw_atom | comp_bare | comp_proc.
-- Kind must be one of: atom | raw_atom | comp_bare | comp_proc.
local function process_atom(atom, src)
if not (atom and atom.body) then return end
local kind = atom.kind
if kind ~= "atom" and kind ~= "atom_proc" and kind ~= "raw_atom" and kind ~= "comp_bare" and kind ~= "comp_proc" then
if kind ~= "atom" and kind ~= "raw_atom" and kind ~= "comp_bare" and kind ~= "comp_proc" then
return
end
local proj = project_atom(atom, src, corpus)
@@ -215,7 +215,7 @@ function M.run(ctx)
end
-- Walk `corpus.source_order`; within each source, visit atoms followed by raw_atoms.
-- Recognized kinds (atom | atom_proc | raw_atom | comp_bare | comp_proc) each receive the atom.paths projection via duffle.project_emission.
-- Recognized kinds (atom | raw_atom | comp_bare | comp_proc) each receive the atom.paths projection via duffle.project_emission.
-- Components are macros inlined into atom bodies; focused tests and isolated component analyses consume atom.paths directly.
for _, src in ipairs(corpus.source_order) do
local scan = src.scan or {}
-8
View File
@@ -4,17 +4,9 @@
--- for `MipsAtom_(name)` and `MipsCode code_<name>` declarations, computes the word offset
--- from each `atom_offset(F, T)` marker to its target `atom_label(T)` declaration, and emits
--- `gen/offsets.h` with one `#define _atom_offset_F_T = N` per branch.
---
--- Per-directory aggregation: every source in the same directory contributes to the same `gen/offsets.h`.
--- The directory itself is the namespace; the filename does not repeat the module name.
---
--- (Task 12.16 note: atom-namespaced enum names — e.g., `atom_offset__normalize_v3s4__srav_path__aligned_done` —
--- were considered to prevent cross-atom label collisions, but the C-side `atom_offset(F, T)` macro in
--- `code/duffle/dsl.atom.h` doesn't know the current atom_name at expansion time, so any namespacing
--- on the metaprogram side breaks the C build. Reverted. The C-side would need a per-atom
--- `CURRENT_ATOM` #define (set by `MipsAtom_`/`MipsAtom_Proc_` macros) plus an updated `atom_offset`
--- macro that uses it. That's a coordinated refactor — deferred to a future track.)
---
--- The offset is `target_word - branch_word - 1` (the standard MIPS branch-immediate encoding: branch_offset = relative_pc_in_words - 1).
-- ════════════════════════════════════════════════════════════════════════════
+14 -228
View File
@@ -3,7 +3,6 @@
--- Single source-walk pass that produces the fat `SourceScan` payload consumed by all downstream passes. Walks each corpus source record once,
--- extracting every construct type the metaprograms need:
--- MipsAtom_ (kind = "atom", with optional atom_info inner)
--- MipsAtom_Proc_ (kind = "atom_proc", body inside last {})
--- MipsAtomComp_ (kind = "comp_bare")
--- MipsAtomComp_Proc_ (kind = "comp_proc", body inside last {})
--- atom_dbg_skip — bare whole-atom/component debug-step marker; following declaration disambiguates
@@ -35,7 +34,7 @@ local parse_enum_int_literal
-- ════════════════════════════════════════════════════════════════════════════
--- @class SourceScan
--- @field atoms AtomEntry[] -- MipsAtom_ + MipsAtom_Proc_ + MipsAtomComp_ + MipsAtomComp_Proc_
--- @field atoms AtomEntry[] -- MipsAtom_ + MipsAtomComp_ + MipsAtomComp_Proc_
--- @field raw_atoms AtomEntry[] -- MipsCode code_<name> { body } (offsets pass only)
--- @field binds BindsEntry[] -- typedef Struct_(Binds_X) { fields } (fields pre-parsed)
--- @field atom_infos AtomInfoEntry[] -- MipsAtom_(name) atom_info(...) (sub-calls pre-parsed)
@@ -56,7 +55,7 @@ local parse_enum_int_literal
--- @field args string|nil -- Trimmed args inside the `(...)` (nil when has_parens is false)
--- @field pending boolean -- true while awaiting the following declaration
--- @field superseded_by_marker_line integer|nil -- set when a newer marker bumped this one out of the pending slot
--- @field target_kind string|nil -- "atom" | "atom_proc" | "comp_bare" | "comp_proc" | "unrelated" once observed (nil if no declaration ever followed)
--- @field target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed (nil if no declaration ever followed)
--- @field proc_prelude boolean|nil -- true after the marker crossed an `FI_` prelude and awaits `MipsAtomComp_Proc_`
--- @class RegTypeDefault
@@ -112,7 +111,7 @@ local parse_enum_int_literal
--- @field name string -- Atom name (for components: without ac_ prefix)
--- @field body string -- Brace-delimited body (without the braces)
--- @field body_off integer -- Char offset of body[1] in source
--- @field kind string -- "atom" | "atom_proc" | "comp_bare" | "comp_proc" | "raw_atom"
--- @field kind string -- "atom" | "comp_bare" | "comp_proc" | "raw_atom"
--- @field raw_name string -- Un-stripped name (for components: with ac_ prefix)
--- @field ident_pos integer -- Position of the MipsAtom_/MipsAtomComp_ ident start
--- @field after_paren integer -- Position past the closing paren
@@ -137,16 +136,6 @@ local QUALIFIER_KEYWORDS = {
local AC_PREFIX = "ac_"
local AC_PREFIX_LEN = 3
-- The function-decl keyword that precedes a MipsAtomComp_Proc_ call.
-- Used by the backward walk in duffle.find_function_decl_for.
local SLICE_MIPS_CODE = "Slice_MipsCode"
local SLICE_MIPS_CODE_LEN = #SLICE_MIPS_CODE
-- The return type that precedes a MipsAtom_Proc_ function declaration.
-- Used by the backward walk in duffle.find_atom_proc_decl_for.
local MIPS_ATOM_PTR = "MipsAtom*"
local MIPS_ATOM_PTR_LEN = #MIPS_ATOM_PTR
--- Strip the "ac_" prefix from a component name.
--- Returns the input unchanged if it doesn't start with the prefix.
--- @param raw_name string
@@ -279,7 +268,7 @@ end
--- marker_kind == "atom_dbg_skip" AND is_bare == true
--- Any other spelling or shape (parenthesized form, legacy name) is recorded as a raw marker for annotation validation but never stamps `debug_skip`.
--- @param out SourceScan
--- @param target_kind string|nil -- "atom" | "atom_proc" | "comp_bare" | "comp_proc" | "unrelated" once observed
--- @param target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed
--- @return boolean|nil -- true iff the marker is the positive bare form
local function attach_debug_skip_marker(out, target_kind)
local markers = out.debug_skip_markers
@@ -810,11 +799,6 @@ local BYTE_x = 0x78 -- 'x'
local BYTE_X = 0x58 -- 'X'
local BYTE_OPEN_BRACE = 0x7B -- '{'
local BYTE_CLOSE_BRACE= 0x7D -- '}'
local BYTE_SLASH = 0x2F -- '/'
local BYTE_STAR = 0x2A -- '*'
local BYTE_SPACE = 0x20 -- ' '
local BYTE_TAB = 0x09 -- '\t'
local BYTE_CR = 0x0D -- '\r'
-- Maximum chain depth when resolving `R_*_Code` symbol RHS references.
-- Eight hops is enough for any production chain (R_TapePtr_Code -> R_T8_Code -> ...).
@@ -838,44 +822,6 @@ local function hex_digit_value(b)
return nil
end
-- Read one trailing C-comment that appears immediately after `pos` in `body`,
-- skipping horizontal whitespace and newlines first. Used by `parse_enum_entry` to
-- recover the `atom_auto_reg:` / `phase_auto_reg:` scope annotation embedded by
-- the `atom_auto_reg` / `phase_auto_reg` macros' RHS expansion
-- (`R_<Sym> = R_<Sym>_Code /* atom_auto_reg: <scope> */`).
-- Handles both block (`/* ... */`) and line (`// ...`) forms.
-- Returns the comment text (without delimiters), or nil if no comment is adjacent.
local function read_trailing_cmt_after(body, pos)
local body_len = #body
while pos <= body_len do
local b = body:byte(pos)
if b == BYTE_SPACE or b == BYTE_TAB or b == BYTE_NEWLINE or b == BYTE_CR then
pos = pos + 1
elseif b == BYTE_SLASH then
local b2 = body:byte(pos + 1)
if b2 == BYTE_STAR then
-- Block comment /* ... */
local i = pos + 2
while i < body_len do
if body:byte(i) == BYTE_STAR and body:byte(i + 1) == BYTE_SLASH then
return body:sub(pos + 2, i - 1)
end
i = i + 1
end
return nil -- unterminated; treat as no comment
elseif b2 == BYTE_SLASH then
-- Line comment // ... (strip the trailing newline)
local end_pos = duffle.find_byte(body, BYTE_NEWLINE, pos + 2) or (body_len + 1)
return body:sub(pos + 2, end_pos - 1)
end
return nil
else
return nil
end
end
return nil
end
--- Parse a decimal/negative-decimal/hex integer literal starting at byte position `start`.
--- Returns (value, end_pos) on success, or (nil, start) on failure / no match.
--- Accepts: 12, -1, 0, 0x10, 0X1F, -0x10.
@@ -1182,46 +1128,6 @@ local function parse_dbg_skip_marker(source, pos, ident_end, line_of, out)
return marker_end
end
--- Parse `atom_auto_reg(<atom>, R_<Sym>)` and `phase_auto_reg(<phase>, R_<Sym>)` markers.
---
--- The macros expand to `sym = sym##_Code` per their definition in dsl.atom.h.
--- After preprocessing, the marker renders as a full enum entry of the form `R_<Sym> = R_<Sym>_Code,`.
--- This parser detects the macro invocation site, extracts `(scope_name, sym)`, and stores it
--- in the per-source table (atom_auto_regs or phase_auto_regs) under the scope's name.
---
--- @param source string
--- @param pos integer
--- @param ident_end integer
--- @param line_of fun(pos: integer): integer
--- @param out SourceScan
--- @return integer
local function parse_auto_reg_marker(source, pos, ident_end, line_of, out)
local marker_kind = source:sub(pos, ident_end - 1) -- "atom_auto_reg" or "phase_auto_reg"
local scope_kind = marker_kind == "atom_auto_reg" and "atom" or "phase"
local inner, after_paren = read_parens_after(source, ident_end)
if not inner then return after_paren end
local args = duffle.split_top_level_commas(inner)
local scope_name = args[1] and duffle.trim(args[1]) or nil
local sym = args[2] and duffle.trim(args[2]) or nil
-- Filter: only accept `R_<Sym>` form (matches `^R_[%w_]+$`).
if scope_name and sym and sym:match("^R_[%w_]+$") then
if scope_kind == "atom" then
out.atom_auto_regs = out.atom_auto_regs or {}
out.atom_auto_regs[scope_name] = out.atom_auto_regs[scope_name] or {}
out.atom_auto_regs[scope_name][sym] = sym
else
out.phase_auto_regs = out.phase_auto_regs or {}
out.phase_auto_regs[scope_name] = out.phase_auto_regs[scope_name] or {}
out.phase_auto_regs[scope_name][sym] = sym
end
end
return after_paren
end
-- Parse `atom_dbg_reg_default(R_X, <type>...)`;
-- the second argument may be a `Type` or `Type*`/`Type**` chain. Records in `out.types[R_X]`.
local function parse_atom_dbg_reg_default(source, pos, ident_end, line_of, out)
@@ -1367,11 +1273,7 @@ local function parse_mips_atom_comp_proc(source, pos, ident_end, line_of, out)
local body, close_pos = duffle.read_braces(inner, last_brace_pos)
if close_pos > #inner + 1 then return after_paren end
-- The component name is derived from the preceding function declaration
-- (`FI_ Slice_MipsCode ac_X(...)`), not from the first macro arg (which
-- is now `ab`). The backward walk finds the function decl before open_paren.
local raw_name = duffle.find_function_decl_for(source, open_paren, SLICE_MIPS_CODE_LEN)
if not raw_name then raw_name = "?" end
local raw_name = inner:match("^%s*([%w_]+)") or "?"
local name = strip_ac_prefix(raw_name)
-- Position of body[1] in source = open_paren + 1 (start of inner) + last_brace_pos + 1 (past '{').
local body_off = open_paren + 2 + last_brace_pos
@@ -1380,52 +1282,6 @@ local function parse_mips_atom_comp_proc(source, pos, ident_end, line_of, out)
return after_paren
end
--- Parse: `MipsAtom_Proc_(<name>, <abuilder>, { <body> })` — body is inside the LAST `{` in args.
--- Per Task 12.10: full support for the runtime-proc atom form. Registers the atom
--- with kind `"atom_proc"` so offsets.lua / components.lua can emit
--- * `mac_<name>` aliases in `gen/macs.h` (the components pass)
--- * `atom_offset__X__Y` defs in `gen/offsets.h` (the offsets pass)
--- The atom name is the FIRST ident of the args (the second arg `ab` is the
--- atom-builder, not the name). Unlike `MipsAtomComp_Proc_`, there is no `ac_`
--- prefix on the symbol — `MipsAtom_Proc_` is the runtime-proc wrapper, so the
--- symbol IS the bare atom name (e.g. `normalize_v3s4`, not `ac_normalize_v3s4`).
--- @param source string
--- @param pos integer
--- @param ident_end integer
--- @param line_of fun(pos: integer): integer
--- @param out SourceScan
--- @return integer
local function parse_mips_atom_proc(source, pos, ident_end, line_of, out)
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
if not inner then return after_paren end
-- Find the LAST `{` in inner (the body brace, not any potential embedded braces in expressions).
local last_brace_pos = nil
for search_pos = #inner, 1, -1 do
if inner:sub(search_pos, search_pos) == "{" then last_brace_pos = search_pos; break end
end
if not last_brace_pos then return after_paren end
-- Use duffle.read_braces to find the matching close brace.
-- Uses `read_balanced` for delimiter-depth tracking.
-- If close_pos is past the end of inner, the brace didn't match (malformed input); skip.
local body, close_pos = duffle.read_braces(inner, last_brace_pos)
if close_pos > #inner + 1 then return after_paren end
-- The atom name is derived from the preceding function declaration
-- (`internal MipsAtom* X_proc(...)`), not from the first macro arg (which
-- is now `aa`). The backward walk finds the function decl before open_paren
-- and strips the `_proc` suffix.
local raw_name = duffle.find_atom_proc_decl_for(source, open_paren, MIPS_ATOM_PTR_LEN)
if not raw_name then raw_name = "?" end
local name = strip_ac_prefix(raw_name)
-- Position of body[1] in source = open_paren + 1 (start of inner) + last_brace_pos + 1 (past '{').
local body_off = open_paren + 2 + last_brace_pos
register_atom(out, "atom_proc", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
return after_paren
end
--- Parse: `MipsCode code_<name> { <body> }` (raw atom form — offsets pass only).
--- @param source string
--- @param pos integer
@@ -1746,16 +1602,6 @@ local function parse_enum_entry(source, body, body_offset, line_of, out, entry_n
local value, value_end = parse_enum_value(body, after_ws, out)
if value == nil then return value_start end
-- Capture the trailing C-comment (if any) before `skip_ws_and_cmt` discards it.
-- The `atom_auto_reg(<scope>, <sym>)` macro expands to `R_<Sym> = R_<Sym>_Code /* atom_auto_reg: <scope> */`,
-- so the scope name lives in the comment after the RHS value. Routes through `out.atom_entry_comments`
-- for downstream `parse_enum` to split into `out.atom_auto_regs` / `out.phase_auto_regs`.
local trailing_cmt = read_trailing_cmt_after(body, value_end)
if trailing_cmt then
out.atom_entry_comments = out.atom_entry_comments or {}
out.atom_entry_comments[entry_name] = trailing_cmt
end
local after_value = duffle.skip_ws_and_cmt(body, value_end)
local has_atom_reg, end_after_atom_reg = check_bare_atom_reg(body, after_value)
@@ -1811,24 +1657,15 @@ local function parse_enum_body(source, body, body_offset, line_of, out)
else
local entry_name, name_end = duffle.read_ident(body, pos)
if entry_name then
-- In-enum `atom_auto_reg(<scope>, R_<Sym>)` / `phase_auto_reg(<scope>, R_<Sym>)` markers:
-- the C preprocessor expands them to `R_<Sym> = R_<Sym>_Code /* atom_auto_reg: <scope> */`,
-- but the metaprogram reads source-as-written so we must dispatch the parser here too.
-- Mirrors the top-level `DECL_PARSERS` entry for `atom_auto_reg` / `phase_auto_reg`.
if entry_name == "atom_auto_reg" or entry_name == "phase_auto_reg" then
local new_pos = parse_auto_reg_marker(body, pos, name_end, line_of, out)
if new_pos > pos then pos = new_pos else pos = name_end end
local after_name = duffle.skip_ws_and_cmt(body, name_end)
if body:byte(after_name) == BYTE_EQUAL then
local new_pos = parse_enum_entry(
source, body, body_offset, line_of, out,
entry_name, pos, after_name + 1
)
if new_pos > pos then pos = new_pos else pos = after_name + 1 end
else
local after_name = duffle.skip_ws_and_cmt(body, name_end)
if body:byte(after_name) == BYTE_EQUAL then
local new_pos = parse_enum_entry(
source, body, body_offset, line_of, out,
entry_name, pos, after_name + 1
)
if new_pos > pos then pos = new_pos else pos = after_name + 1 end
else
pos = name_end
end
pos = name_end
end
else
pos = pos + 1
@@ -1858,25 +1695,6 @@ local function parse_enum(source, pos, ident_end, line_of, out)
if not body then return after_brace end
parse_enum_body(source, body, body_off, line_of, out)
-- Route `atom_auto_reg:` / `phase_auto_reg:` markers discovered in trailing C-comments
-- into the per-source `atom_auto_regs` / `phase_auto_regs` projections.
-- Pattern matches the RHS expansion `R_<Sym> = R_<Sym>_Code /* <kind>_auto_reg: <scope> */`
-- emitted by the `atom_auto_reg` / `phase_auto_reg` macros in dsl.atom.h.
for entry_name, cmt_text in pairs(out.atom_entry_comments or {}) do
local atom_scope = cmt_text:match("atom_auto_reg:%s*([%w_]+)")
if atom_scope then
out.atom_auto_regs = out.atom_auto_regs or {}
out.atom_auto_regs[atom_scope] = out.atom_auto_regs[atom_scope] or {}
out.atom_auto_regs[atom_scope][entry_name] = entry_name
end
local phase_scope = cmt_text:match("phase_auto_reg:%s*([%w_]+)")
if phase_scope then
out.phase_auto_regs = out.phase_auto_regs or {}
out.phase_auto_regs[phase_scope] = out.phase_auto_regs[phase_scope] or {}
out.phase_auto_regs[phase_scope][entry_name] = entry_name
end
end
return after_brace
end
@@ -1890,18 +1708,12 @@ end
local DECL_PARSERS = {
MipsAtom_ = parse_mips_atom,
MipsAtom_Proc_ = parse_mips_atom_proc,
MipsAtomComp_ = parse_mips_atom_comp,
MipsAtomComp_Proc_ = parse_mips_atom_comp_proc,
-- `atom_dbg_skip` is the only debug-skip parser entry. Every other
-- identifier follows the ordinary unrelated-token path; there is no alias.
atom_dbg_skip = parse_dbg_skip_marker,
atom_dbg_reg_default = parse_atom_dbg_reg_default,
-- `atom_auto_reg(atom, R_<Sym>)` and `phase_auto_reg(phase, R_<Sym>)` populate per-source
-- `out.atom_auto_regs` / `out.phase_auto_regs`; the cross-source merge lands in
-- `corpus.atom_auto_regs` / `corpus.phase_auto_regs` (first-wins).
atom_auto_reg = parse_auto_reg_marker,
phase_auto_reg = parse_auto_reg_marker,
MipsCode = parse_mips_code,
typedef = parse_typedef_binds,
_Pragma = parse_pragma_macro,
@@ -1936,14 +1748,6 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
debug_skip_markers = {},
types = {},
atom_views = {},
-- Per-source projection for `atom_auto_reg(<atom>, R_<Sym>)` markers.
-- Each entry is keyed by atom_name; the inner table maps `R_<Sym>` -> `R_<Sym>` (raw LHS sym).
-- Merged cross-source into `corpus.atom_auto_regs` (first-wins).
atom_auto_regs = {},
-- Per-source projection for `phase_auto_reg(<phase>, R_<Sym>)` markers.
-- Each entry is keyed by phase_label; the inner table maps `R_<Sym>` -> `R_<Sym>` (raw LHS sym).
-- Merged cross-source into `corpus.phase_auto_regs` (first-wins).
phase_auto_regs = {},
line_of = line_of,
-- Source-derived register-alias registry (atom_reg opt-in entries).
-- Keys are full R_* idents (never stripped); see parse_enum / parse_enum_body.
@@ -2183,8 +1987,6 @@ local function merge_corpus_registries(corpus)
corpus.atom_ctxs = corpus.atom_ctxs or {}
corpus.atom_phases = corpus.atom_phases or {}
corpus.atom_infos = corpus.atom_infos or {}
corpus.atom_auto_regs = corpus.atom_auto_regs or {}
corpus.phase_auto_regs = corpus.phase_auto_regs or {}
corpus.collisions = corpus.collisions or {}
-- Replace the existing corpus collections with empty tables so a re-run on the same corpus produces identical state (deterministic merge).
@@ -2228,7 +2030,7 @@ local function merge_corpus_registries(corpus)
corpus.collisions, "binds", bind_shape)
end
-- atoms_by_name: MipsAtom_(name) + MipsAtom_Proc_(name) + MipsAtomComp_(name) + MipsAtomComp_Proc_(name).
-- atoms_by_name: MipsAtom_(name) + MipsAtomComp_(name) + MipsAtomComp_Proc_(name).
-- Each atom carries `{line, name, body, body_off, kind, raw_name, ...}`.
-- Duplicate atom names across sources are first-wins + collision; see the atom_infos block below for the evidence list.
for _, atom_entry in ipairs(scan.atoms or {}) do
@@ -2263,22 +2065,6 @@ local function merge_corpus_registries(corpus)
corpus.collisions, "phase", phase_shape)
end
-- atom_auto_regs: keyed by atom scope name; each carries a `{R_<Sym> = R_<Sym>}` map.
-- Per-source entries are simple inner maps (no body / no shape comparison); first-wins suffices.
for atom_scope, syms in pairs(scan.atom_auto_regs or {}) do
if corpus.atom_auto_regs[atom_scope] == nil then
corpus.atom_auto_regs[atom_scope] = syms
end
end
-- phase_auto_regs: keyed by phase label; each carries a `{R_<Sym> = R_<Sym>}` map.
-- Per-source entries are simple inner maps (no body / no shape comparison); first-wins suffices.
for phase_label, syms in pairs(scan.phase_auto_regs or {}) do
if corpus.phase_auto_regs[phase_label] == nil then
corpus.phase_auto_regs[phase_label] = syms
end
end
-- atom_infos: ALWAYS append every record in source/declaration order.
-- Duplicates are preserved so the annotation pass can flag them via `check_unique_annotation`;
-- The merge is purely order-preserving.
+39 -375
View File
@@ -39,8 +39,8 @@
--- `── Info` section renders finding-level info between `── Warnings` and the per-atom cycle counts.
---
--- The structural handshake checks (`mac_yield_uniformity`, `hazard_nop_use`, `control_transfer_delay_slot_use`) skip atoms/components with `debug_skip == true`.
--- `atom_dbg_skip` marker designates runtime-helper declarations whose structure is fixed by the tape runtime (e.g. `tape_exit`, `ac_yield`).
--- Flagging them as "missing mac_yield" or "BD slot is redundant".
--- The `atom_dbg_skip` marker designates runtime-helper declarations whose structure is fixed by the tape runtime (e.g. `tape_exit`, `ac_yield`).
--- Flagging them as "missing mac_yield" or "BD slot is redundant" is signal noise, not a logic failure.
--- Other checks (transfer_hazards, gpu_portstore_shape, abi_handoff, enum_alias_membership, …) still apply to debug_skip declarations because real hazards / typos can still surface in them.
---
--- The orchestrator (`ps1_meta.lua`) wires this module in via the PASSES table:
@@ -256,21 +256,8 @@ local BRANCH_PATTERN = "^branch_[%w_]+%s*%("
-- The C preprocessor expands it BEFORE the metaprogram sees the source, but for source-level metadata consistency we still match it here and classify it as a branch_equal.
-- This keeps `consuming_encoder` canonical for any downstream tooling that consults the metadata field.
local JUMP_REL_PATTERN = "^jump_rel%s*%("
local UNCOND_JUMP_PATTERNS = {
"^%f[%w]jump%f[%W]",
"^%f[%w]call_addr%f[%W]",
}
local TERMINAL_JUMP_PATTERNS = {
"^%f[%w]jump_reg%f[%W]",
"^%f[%w]call_reg%f[%W]",
"^%f[%w]jump_link%f[%W]",
}
local function matches_any(tok, patterns)
for i = 1, #patterns do
if tok:match(patterns[i]) then return true end
end
return false
end
local UNCOND_JUMP_PATTERN = "^%f[%w](jump|call_addr)%f[%W]"
local TERMINAL_JUMP_PATTERN = "^%f[%w](jump_reg|call_reg|jump_link)%f[%W]"
local function classify_tokens(tokens)
local n = #tokens
@@ -314,13 +301,13 @@ local function classify_tokens(tokens)
-- Both encode a 16-bit signed relative word offset.
is_branch = true
branch_label = tok:match("atom_offset%s*%([^,]+,%s*([%w_]+)%s*%)") or false
elseif matches_any(tok, UNCOND_JUMP_PATTERNS) then
elseif tok:match(UNCOND_JUMP_PATTERN) then
-- Unconditional absolute jump / call: `jump(off)` / `call_addr(off)`.
-- One immediate offset field; can carry an `atom_offset(F, T)` marker (the offsets pass dispatches on `consuming_encoder` — see `passes/offsets.lua::compute_offsets`).
is_branch = true
is_unconditional_jump = true
branch_label = tok:match("atom_offset%s*%([^,]+,%s*([%w_]+)%s*%)") or false
elseif matches_any(tok, TERMINAL_JUMP_PATTERNS) then
elseif tok:match(TERMINAL_JUMP_PATTERN) then
-- Register-form jump / call: no offset field; `atom_offset` is invalid here (the offsets pass will error if one is supplied).
-- Transfers control OUT of the current atom — the CFG treats this as a path terminator.
is_terminal_jump = true
@@ -452,7 +439,7 @@ local function is_cop2_consumer_of(consumer_event, destination, producer_rel)
end
-- True iff `consumer_event` reads the GPR operand at any position the destination register occupies.
-- read_pos lookup consults `duffle.OPERAND_READ_POSITIONS` for the consumer's encoder and walks each `args[pos]` to find an operand-equal match.
-- The read-position lookup consults `duffle.OPERAND_READ_POSITIONS` for the consumer's encoder and walks each `args[pos]` to find an operand-equal match.
local function is_gpr_consumer_of(consumer_event, destination)
local consumer_token = consumer_event.encoder or consumer_event.ident
local read_pos = duffle.OPERAND_READ_POSITIONS or {}
@@ -580,31 +567,13 @@ local function evaluate_gpr_value_rule(rule, ev_args, gpr_values)
return shift_left_u4(immediate % 0x10000, 16)
end
-- Encoders that take `R_0` implicitly (e.g. `li_s(rt, imm)` which is `add_ui(rt, R_0, imm)`) have a non-GPR operand at the source position.
-- Fall back to R_0 = 0.
-- The implicit-R_0 macros also use a different immediate position (e.g. `li_s`'s `add_ui` rule has source = 2 / immediate = 3
-- but the macro takes 2 args); when the configured immediate position is out of bounds.
-- Fall back instead to scanning the macro's args for the first integer literal and use that as the immediate.
local source = 0
local source = nil
if rule.source then
if is_gpr_operand(ev_args[rule.source]) then
source = constant_for_operand(gpr_values, ev_args[rule.source])
if source == nil then return nil end
end
-- Non-GPR at source position = implicit R_0; source stays 0.
end
local immediate = nil
if rule.immediate and ev_args[rule.immediate] ~= nil then
immediate = parse_integer_literal(ev_args[rule.immediate])
if immediate == nil then return nil end
elseif rule.immediate then
-- Immediate position out of bounds: scan for the first integer literal in the args.
for _, arg in ipairs(ev_args) do
immediate = parse_integer_literal(arg)
if immediate ~= nil then break end
end
if immediate == nil then return nil end
source = constant_for_operand(gpr_values, ev_args[rule.source])
if source == nil then return nil end
end
local immediate = rule.immediate and parse_integer_literal(ev_args[rule.immediate]) or nil
if rule.immediate and immediate == nil then return nil end
if operation == "add_ui" then return wrap_u4( source + sign_extend_i16(immediate))
elseif operation == "or_i" then return bit_binary( source, immediate % 0x10000, "or")
elseif operation == "and_i" then return bit_binary( source, immediate % 0x10000, "and")
@@ -1464,21 +1433,17 @@ end
--- The register becomes non-volatile again at word N+2 (the load has retired), OR sooner if a non-load instruction overwrites the register
--- (the overwriter's write is the fresh producer; the load's value is shadowed and never observed by any reader).
---
--- Runtime-helper atoms / components (`debug_skip == true`) are exempt from some checks, but load-delay
--- safety applies to their emitted instructions as well.
--- Runtime-helper atoms / components (`debug_skip == true`) are exempt: their internal load-then-use sequences
--- are part of the fixed handshake (e.g. `ac_load_tri_indices` loads into R_T0..R_T2, but those are caller-supplied).
---
--- The walker reads `duffle.OPERAND_READ_POSITIONS[event.encoder]` to determine which args are read-source
--- (the destination of a load is in `writes`, not `reads` — see `duffle.INSTRUCTION_GPR_EFFECTS`).
--- The check is purely structural; it does not consult the GPR-value lattice
--- (no constant propagation needed for load-delay detection — the volatility window is unconditional).
--- The check is purely structural; it does not consult the GPR-value lattice (no constant propagation needed for load-delay detection — the volatility window is unconditional).
local function check_load_delay_slots(atom, pipe_ctx, findings)
-- The load-delay check applies to every atom and component body, including debug-skipped components (`ac_*` and `atom_dbg_skip MipsAtom_(...)`).
-- The `atom_dbg_skip` marker controls debugger stepping, not instruction safety.
-- `atom_proc` atoms have full bodies with loads that need delay slots, so the check applies to them too.
local p = atom.paths or {}
if atom.kind ~= "atom" and atom.kind ~= "atom_proc" then return end
local events = p.word_events or {}
if atom.kind ~= "atom" then return end
local events = atom.paths.word_events or {}
if #events == 0 then return end
if is_runtime_helper(atom) then return end
local gpr_effects = duffle.INSTRUCTION_GPR_EFFECTS or {}
local read_positions = duffle.OPERAND_READ_POSITIONS or {}
@@ -1580,8 +1545,6 @@ local function check_mac_yield_uniformity(atom, pipe_ctx, findings)
if is_runtime_helper(atom) then return end
-- Per-kind semantics:
-- MipsAtom_ (baked atom): exactly 1 mac_yield at the end of the body. Control transfer is the atom's job.
-- MipsAtom_Proc_ (runtime-proc atom): exactly 1 mac_yield at the end of the body. Same as baked atom;
-- the proc IS the atom; the runtime call to `atombuilder_unroll` doesn't introduce a parent atom.
-- MipsAtomComp_ (bare static-array component): ZERO mac_yield.
-- The component is invoked from inside an atom body; the parent atom does the yield.
-- MipsAtomComp_Proc_ (procedural component): ZERO mac_yield.
@@ -1605,7 +1568,7 @@ local function check_mac_yield_uniformity(atom, pipe_ctx, findings)
return atom.line + line_in_body[tokens[idx].rel]
end
if atom.kind == "atom" or atom.kind == "atom_proc" then
if atom.kind == "atom" then
-- Baked atom: exactly 1 yield at the end.
if count == 0 then
findings[#findings + 1] = {
@@ -1650,7 +1613,6 @@ local function check_mac_yield_uniformity(atom, pipe_ctx, findings)
-- The parent atom does the yield.
-- A yield inside a component would either be dead code (bare) or prematurely terminate the function (proc).
-- Both are bugs.
-- `atom_proc` atoms are NOT components; they're runtime-proc atoms that own their own yield (handled in the `if` branch above).
if count > 0 then
findings[#findings + 1] = {
atom = atom.name,
@@ -1682,7 +1644,7 @@ end
--- Per-atom. Runtime-helper atoms (`debug_skip`) are exempt.
--- Takes `(atom, pipe_ctx, findings)`; `pipe_ctx` is unused.
local function check_yield_load_tail_pairing(atom, _pipe_ctx, findings)
if atom.kind ~= "atom" and atom.kind ~= "atom_proc" then return end
if atom.kind ~= "atom" then return end
if is_runtime_helper(atom) then return end
local tokens = atom.paths.tokens
@@ -1694,39 +1656,21 @@ local function check_yield_load_tail_pairing(atom, _pipe_ctx, findings)
return atom.line + line_in_body[tokens[idx].rel]
end
-- ── Rule 1: every `mac_yield_load()` must be in a branch BD-slot, OR sit between two `atom_label`s (natural fall-through load pattern).
-- When the pattern is satisfied, the check stays silent; only violations emit findings.
-- ── Rule 1: every `mac_yield_load()` must be in a branch BD-slot.
for tok_idx = 1, n do
local c = tc[tok_idx]
if c.ident == "mac_yield_load" then
local prev_tc = (tok_idx >= 2) and tc[tok_idx - 1] or nil
-- Look for the next `atom_label()` token (skip `atom_offset` markers; check immediately-adjacent first).
local next_label_tc = (tok_idx + 1 <= n) and tc[tok_idx + 1] or nil
if next_label_tc and next_label_tc.ident ~= "atom_label" then
next_label_tc = nil
for j = tok_idx + 1, n do
local t = tc[j]
if t.ident == "atom_label" then
next_label_tc = t
break
end
end
end
local natural_fallthrough = prev_tc and prev_tc.is_atom_label and next_label_tc ~= nil
if not natural_fallthrough then
if tok_idx < 2 or not prev_tc.is_branch then
local prev_ident = prev_tc and (prev_tc.ident or "?") or "<none>"
local next_ident = next_label_tc and (next_label_tc.ident .. "(" .. (next_label_tc.label_name or "?") .. ")") or "<no following label>"
findings[#findings + 1] = {
atom = atom.name,
line = tok_idx >= 2 and line_for(tok_idx) or atom.line,
check = "yield_load_tail_pairing",
kind = "error",
msg = string.format(
"%s at line %d has `mac_yield_load()` at word %d but the previous token is `%s`, not a branch — and the next `atom_label()` token is `%s` — `mac_yield_load()` must fill a branch BD-slot or sit between two `atom_label`s for the natural fall-through load."
, atom.name, tok_idx >= 2 and line_for(tok_idx) or atom.line, tok_idx, prev_ident, next_ident),
}
end
if tok_idx < 2 or not tc[tok_idx - 1].is_branch then
local prev_ident = (tok_idx >= 2) and (tc[tok_idx - 1].ident or "?") or "<none>"
findings[#findings + 1] = {
atom = atom.name,
line = tok_idx >= 2 and line_for(tok_idx) or atom.line,
check = "yield_load_tail_pairing",
kind = "error",
msg = string.format(
"%s at line %d has `mac_yield_load()` at word %d but the previous token is `%s`, not a branch — `mac_yield_load()` must fill a branch BD-slot."
, atom.name, tok_idx >= 2 and line_for(tok_idx) or atom.line, tok_idx, prev_ident),
}
end
end
end
@@ -1901,9 +1845,9 @@ end
--- - Atoms containing a `mac_<name>(...)` call whose `name` is not registered in `pipe_ctx.components_by_name` emit a "new macro;
--- Not in corpus.components" advisory — the auto-derivation returned nil for that name.
---
--- Applies only to `kind = "atom"` or `kind = "atom_proc"` (full-atom bodies). Components don't emit full primitives.
--- Applies only to `kind = "atom"` (baked atoms). Components don't emit full primitives.
local function check_gpu_portstore_shape(atom, pipe_ctx, findings)
if atom.kind ~= "atom" and atom.kind ~= "atom_proc" then return end
if atom.kind ~= "atom" then return end
local tokens = atom.paths.tokens
local line_in_body = atom.paths.line_in_body
local tc = atom.paths.tok_class
@@ -2071,9 +2015,8 @@ local function analyze_atom_paths(atom, pipe_ctx)
succ[#succ + 1] = label_pos + 1
end
end
-- For literal-offset jumps (label == false), control transfers out unconditionally.
-- Treat as a terminator so the path is recorded (NOT as a silent fall-through to the next token, which is unreachable in this atom's execution).
return {}, tok_idx
-- For literal-offset jumps (label == false), the target is a non-tracked address; conservatively omit.
return succ, nil
end
-- Conditional branch: BD slot absorbed; two successors — fall-through (tok_idx+2) + taken (if known).
if tok_idx + 2 <= n then
@@ -2089,11 +2032,9 @@ local function analyze_atom_paths(atom, pipe_ctx)
-- Return (succ, nil), the second value is the terminator marker (nil = not a terminator).
return succ, nil
end
-- Normal token: just the next one.
-- The final ordinary word of the body has no successor and terminates the path;
-- record it as an implicit endpoint so the cycle budget for non-yield components is not silently zeroed.
-- Normal token: just the next one
if tok_idx + 1 <= n then return { tok_idx + 1 }, nil end
return {}, tok_idx
return {}, nil
end
-- DFS through all paths. Track the current cycle sum, a visited set scoped to the current path (to detect loops), and a count of paths.
@@ -2388,274 +2329,6 @@ end
-- ════════════════════════════════════════════════════════════════════════════
-- ════════════════════════════════════════════════════════════════════════════
-- GTE control-register alias + RT-diagonal + TR-naming helpers and checks
-- ════════════════════════════════════════════════════════════════════════════
--- Resolve a `gte_cr_<Alias>` ident to its alias-group entry, or nil if the alias
--- is in a distinct-slot group (or the alias name is not a known C2 control-register alias).
--- Reads `M.GTE_CR_ALIAS_GROUPS` from `duffle.lua`.
local function find_alias_pair_for(alias_name, duffle)
local groups = (duffle and duffle.GTE_CR_ALIAS_GROUPS) or {}
for _, group in ipairs(groups) do
for _, name in ipairs(group[2] or {}) do
if name == alias_name then return group end
end
end
return nil
end
-- True iff `c` (a TokClass entry) is a CPU→COP2 control-register transfer
-- (`gte_mv_to_ctrl_r` / `gte_mv_from_ctrl_r`).
local function is_ctrl_r_transfer(c)
if c == nil then return false end
return c.ident == "gte_mv_to_ctrl_r" or c.ident == "gte_mv_from_ctrl_r"
end
-- Resolve a token's source line. The per-token `line` is the body-relative
-- line; `atom.line` is the source line of the atom declaration; `line_in_body`
-- (atom.paths) maps a body-relative line to its source line. The arithmetic
-- `atom.line + line_in_body[tok.rel] - 1` matches the convention used by
-- check_abi_handoff and check_control_transfer_delay_slot_use elsewhere.
local function atom_body_token_source_line(atom, token, line_in_body)
if line_in_body == nil or token == nil or token.rel == nil then
return atom.line or 0
end
local body_line = line_in_body[token.rel]
if body_line == nil then return atom.line or 0 end
return (atom.line or 0) + body_line - 1
end
-- Check #N: gte_cr_alias_writes
-- Fires one warning per atom per alias-group when the atom body touches two
-- distinct aliases from the same group. Aliases within a group write to the
-- same C2 control-register slot on real silicon; cross-alias writes inside
-- one atom body silently clobber each other.
--
-- Severity: warning. Build continues. The libgte outer-product convention
-- uses only RT-row aliases (which are NOT in `M.GTE_CR_ALIAS_GROUPS`), so
-- the canonical convention does not trigger this check.
local function check_gte_cr_alias_writes(atom, pipe_ctx, findings)
local groups = pipe_ctx.gte_cr_alias_groups or {}
if not next(groups) then return end
local tokens = atom.paths and atom.paths.tokens or {}
local tc = atom.paths and atom.paths.tok_class or {}
local line_in_body = atom.paths and atom.paths.line_in_body
if not next(tokens) then return end
-- Build a per-group set of (alias, source_line) pairs touched in this atom body.
-- Walks every token; when the token is a ctrl-r transfer, the alias is at
-- position tok_idx + 2 (rt, alias, [imm-or-arg]). The pre-classified
-- `tc` table tells us whether the token is a ctrl-r transfer and what its
-- source line is.
local touched = {}
for tok_idx, token in ipairs(tokens) do
local c = tc[tok_idx]
if is_ctrl_r_transfer(c) and tokens[tok_idx + 2] then
local alias = tokens[tok_idx + 2].tok
local group = find_alias_pair_for(alias, pipe_ctx.duffle)
if group then
touched[group[1]] = touched[group[1]] or {}
touched[group[1]][#touched[group[1]] + 1] = {
alias = alias,
line = atom_body_token_source_line(atom, token, line_in_body),
}
end
end
end
-- Fire one warning per group touched with 2+ distinct aliases.
for slot, hits in pairs(touched) do
local seen = {}
local distinct = {}
for _, h in ipairs(hits) do
if not seen[h.alias] then
seen[h.alias] = true
distinct[#distinct + 1] = h
end
end
if #distinct >= 2 then
local aliases = {}
for _, d in ipairs(distinct) do aliases[#aliases + 1] = d.alias end
findings[#findings + 1] = {
atom = atom.name or "",
line = distinct[1].line,
check = "gte_cr_alias_writes",
kind = "warning",
msg = string.format(
"atom '%s' touches %d aliases that share C2[%d]: %s; verify the intent"
, atom.name or "", #distinct, slot, table.concat(aliases, ", ")),
}
end
end
end
-- Check #N+1: rtdiagonal_completeness
-- Fires one info per atom body when the bare `gte_cmdw_mvmva` macro is used.
-- The bare macro encodes only the cmd field; the canonical libgte-2-pass
-- shape uses `gte_cmdw_mvmva_c11_pass2_exact = 0x4A49E012` (gte.h:430).
--
-- Severity: info by default. Escalates to warning when
-- `GTE_RT_DIAGONAL_STRICT=1` env var is set (CI / production builds).
--
-- The bare macro IS the right call for the canonical libgte outer-product
-- convention, so this is an opt-out hint rather than a hard warning.
local function check_rtdiagonal_completeness(atom, _pipe_ctx, findings)
local tokens = atom.paths and atom.paths.tokens or {}
local tc = atom.paths and atom.paths.tok_class or {}
local line_in_body = atom.paths and atom.paths.line_in_body
if not next(tokens) then return end
local strict = os.getenv("GTE_RT_DIAGONAL_STRICT") == "1"
for tok_idx, token in ipairs(tokens) do
local c = tc[tok_idx]
if c and c.ident == "gte_cmdw_mvmva" then
findings[#findings + 1] = {
atom = atom.name or "",
line = atom_body_token_source_line(atom, token, line_in_body),
check = "rtdiagonal_completeness",
kind = strict and "warning" or "info",
msg = string.format(
"atom '%s' uses the bare gte_cmdw_mvmva macro; "
.. "the canonical libgte-2-pass shape is gte_cmdw_mvmva_c11_pass2_exact = 0x4A49E012 "
.. "(gte.h:430). The bare macro does not encode RT23/RT31/RT32/RT33; "
.. "for a full 3x3 matrix, use the dedicated literal or hand-build via enc_gte_*()."
, atom.name or ""),
}
end
end
end
-- Check #N+2: gte_cr_TR_naming
-- Fires one info per atom body when a `gte_cr_TR[XYZ]` alias is used.
-- Translation-vector registers are the only 3-letter-suffix C2 aliases
-- (`TRX/TRY/TRZ`); an agent who reads `TRX` might typo it as `RT_X` or
-- `RTX0` and either get a compile error (best case) or a build that
-- links but routes the `ctc2` write to the wrong C2 slot.
--
-- Severity: info. The convention is correct; this is a documentation-pointer check.
local function check_gte_cr_TR_naming(atom, _pipe_ctx, findings)
local tokens = atom.paths and atom.paths.tokens or {}
local tc = atom.paths and atom.paths.tok_class or {}
local line_in_body = atom.paths and atom.paths.line_in_body
if not next(tokens) then return end
local touched = false
local first_line = 0
for tok_idx, token in ipairs(tokens) do
local c = tc[tok_idx]
if c and c.ident and c.ident:match("^gte_cr_TR[XYZ]$") then
touched = true
if first_line == 0 then
first_line = atom_body_token_source_line(atom, token, line_in_body)
end
end
end
if touched then
findings[#findings + 1] = {
atom = atom.name or "",
line = first_line,
check = "gte_cr_TR_naming",
kind = "info",
msg = string.format(
"atom '%s' uses gte_cr_TR[XYZ]; translation-vector registers are the only "
.. "3-letter-suffix C2 aliases (TRX/TRY/TRZ). See docs/gte_reference.md §"
.. "\"The `gte_cmdw_mvmva_c11_pass2_exact` literal\" for the libgte outer-product "
.. "convention that uses these names."
, atom.name or ""),
}
end
end
-- check_immediate_field_width — flags integer literals passed to instruction
-- macros that exceed the immediate field width. Reads `IMMEDIATE_FIELD_WIDTHS`
-- from duffle.lua. Only fires on parseable integer literals; register names,
-- O_(...) offsets, atom_offset(...) markers, and enum tokens are skipped.
local function check_immediate_field_width(atom, pipe_ctx, findings)
local widths = duffle.IMMEDIATE_FIELD_WIDTHS or {}
local events = atom.paths and atom.paths.word_events or {}
local line_for_word_event = pipe_ctx.line_for_word_event
for _, ev in ipairs(events) do
local ev_ident = ev.encoder or ev.ident or "?"
local rules = widths[ev_ident]
if rules then
local ev_args = ev.args or {}
local ev_line = line_for_word_event and line_for_word_event(ev) or atom.line
for _, rule in ipairs(rules) do
local arg_str = ev_args[rule.arg]
if arg_str then
local value = parse_integer_literal(arg_str)
if value then
local width = rule.width
local is_signed = rule.signed == true
-- parse_integer_literal returns a U4-wrapped value in [0, 2^32).
-- For signed fields, re-interpret the high bit as the sign.
local signed_value = value
if is_signed and value >= 0x80000000 then
signed_value = value - 0x100000000
end
local lo, hi
if is_signed then
lo = -(bit.lshift(1, width - 1))
hi = bit.lshift(1, width - 1) - 1
else
lo = 0
hi = bit.lshift(1, width) - 1
end
-- For unsigned fields, a negative C literal (high bit set in U4)
-- is valid if the low `width` bits fit — IMM_MASK truncates it.
-- Flag as a warning (code smell), not an error.
local check_value = is_signed and signed_value or value
local field_max = bit.lshift(1, width) - 1
local low_bits_fit = (value % (bit.lshift(1, width))) == value or (is_signed and signed_value >= lo and signed_value <= hi)
if is_signed then
if signed_value < lo or signed_value > hi then
findings[#findings + 1] = {
check = "immediate_field_width",
kind = "error",
atom = atom.name,
line = ev_line,
msg = string.format(
"%s: immediate %d at arg %d overflows %d-bit %s field (valid %d..%d)",
ev_ident, signed_value, rule.arg, width,
"signed", lo, hi),
}
end
else
-- Unsigned field: check if the low `width` bits exceed the field.
-- A negative C literal (U4 >= 0x80000000) whose low bits fit is
-- valid but a code smell — warn, don't error.
local low_bits = value % (bit.lshift(1, width))
if value > field_max then
if value >= 0x80000000 and low_bits <= field_max then
findings[#findings + 1] = {
check = "immediate_field_width",
kind = "warning",
atom = atom.name,
line = ev_line,
msg = string.format(
"%s: negative immediate %d at arg %d on unsigned %d-bit field (truncated to %d by IMM_MASK)",
ev_ident, signed_value, rule.arg, width, low_bits),
}
else
findings[#findings + 1] = {
check = "immediate_field_width",
kind = "error",
atom = atom.name,
line = ev_line,
msg = string.format(
"%s: immediate %d at arg %d overflows %d-bit unsigned field (valid 0..%d)",
ev_ident, value, rule.arg, width, field_max),
}
end
end
end
end
end
end
end
end
end
-- CHECK_RULES — data-driven check dispatch (Muratori: data over control flow)
-- ════════════════════════════════════════════════════════════════════════════
@@ -2682,10 +2355,6 @@ local CHECK_RULES = {
{ name = "abi_handoff", per_atom = check_abi_handoff },
{ name = "gpu_portstore_shape", per_atom = check_gpu_portstore_shape },
{ name = "per_atom_cycle_budget", per_atom = check_per_atom_cycle_budget },
{ name = "gte_cr_alias_writes", per_atom = check_gte_cr_alias_writes },
{ name = "rtdiagonal_completeness", per_atom = check_rtdiagonal_completeness },
{ name = "gte_cr_TR_naming", per_atom = check_gte_cr_TR_naming },
{ name = "immediate_field_width", per_atom = check_immediate_field_width },
{ name = "enum_alias_membership", per_source = check_enum_alias_membership },
{ name = "atom_type_consistency", per_source = check_atom_type_consistency },
{ name = "binds_no_substruct_deref", per_source = check_binds_no_substruct_deref },
@@ -2723,17 +2392,12 @@ local function build_corpus_pipe_ctx(ctx)
atoms_by_name = corpus.atoms_by_name or {},
-- Per-component metadata (cycle_cost + gp0_contrib) auto-derived from the original
-- `MipsAtomComp_` body by `passes/components.lua::compute_components_metadata`.
-- Keyed by bare name (e.g. `format_f3_color`, `gte_store_f3`); the `mac_` prefix at call sites is stripped before lookup.
-- Keyed by bare name (e.g. `format_f3_color`, `gte_store_f3`); the `mac_` prefix at call sites is stripped before lookup.
components_by_name = corpus.components or {},
-- Corpus-wide ordered list of atom_info records (source-order + duplicates).
atom_infos_list = corpus.atom_infos or {},
-- Corpus-wide collisions (recorded by scan_source.merge_corpus_registries).
collisions = corpus.collisions or {},
-- GTE control-register alias groups (from `duffle.GTE_CR_ALIAS_GROUPS`).
-- The three new per_atom checks (gte_cr_alias_writes, rtdiagonal_completeness,
-- gte_cr_TR_naming) read from this view. `duffle` is exposed alongside so
-- `find_alias_pair_for` can resolve alias → group without a separate registry.
gte_cr_alias_groups = duffle.GTE_CR_ALIAS_GROUPS or {},
}
end
Binary file not shown.
+49 -25
View File
@@ -16,36 +16,60 @@
-- Companion: scripts/gdb/gdb_tape_atoms.gdb (covers GPRs + atom-aware stepping).
local function register_handlers()
if not PCSX.WebServer then PCSX.WebServer = {} end
if not PCSX.WebServer.Handlers then PCSX.WebServer.Handlers = {} end
if not PCSX.WebServer then PCSX.WebServer = {} end
if not PCSX.WebServer.Handlers then PCSX.WebServer.Handlers = {} end
-- ── GTE state ──
PCSX.WebServer.Handlers.gte = function(req)
local r = PCSX.getRegisters()
local out = { "pc=0x" .. string.format("%x", r.pc) }
for i = 0, 31 do
out[#out + 1] = string.format("D[%d]=0x%08x C[%d]=0x%08x",
i, r.CP2D.r[i], i, r.CP2C.r[i])
end
return table.concat(out, "\n")
end
-- ── GTE state ──
PCSX.WebServer.Handlers.gte = function(req)
local r = PCSX.getRegisters()
local out = { "pc=0x" .. string.format("%x", r.pc) }
for i = 0, 31 do
out[#out + 1] = string.format("D[%d]=0x%08x C[%d]=0x%08x",
i, r.CP2D.r[i], i, r.CP2C.r[i])
end
return table.concat(out, "\n")
end
-- ── GP state (pointer to existing endpoints) ──
-- pcsx-redux's Lua GPU API exposes only takeScreenShot(); no GPUSTAT / GP0 / GP1 command log / display state.
-- We point to the existing web endpoints that DO expose those (when the emulator is actually rendering. Paused-at-BP frames won't have a fresh frame).
PCSX.WebServer.Handlers.gp = function(req)
local out = {
"gpu_screenshot_png=http://localhost:8080/api/v1/state/still",
"vram_raw=http://localhost:8080/api/v1/gpu/vram/raw (1MB VRAM)",
"gpustat=NOT_AVAILABLE_VIA_LUA",
"gp_command_log=NOT_AVAILABLE_VIA_LUA (use pcsx-redux Debug > GPU Logger)",
"hint_run_emulator_unpaused_for_screenshot",
}
return table.concat(out, "\n")
end
-- ── GP state (pointer to existing endpoints) ──
-- pcsx-redux's Lua GPU API exposes only takeScreenShot(); no GPUSTAT / GP0 / GP1 command log / display state.
-- We point to the existing web endpoints that DO expose those (when the emulator is actually rendering. Paused-at-BP frames won't have a fresh frame).
PCSX.WebServer.Handlers.gp = function(req)
local out = {
"gpu_screenshot_png=http://localhost:8080/api/v1/state/still",
"vram_raw=http://localhost:8080/api/v1/gpu/vram/raw (1MB VRAM)",
"gpustat=NOT_AVAILABLE_VIA_LUA",
"gp_command_log=NOT_AVAILABLE_VIA_LUA (use pcsx-redux Debug > GPU Logger)",
"hint_run_emulator_unpaused_for_screenshot",
}
return table.concat(out, "\n")
end
end
local ok, err = pcall(register_handlers)
if ok then print("[pcsx_debug_helper] handlers registered: gte, gp")
else print("[pcsx_debug_helper] registration failed: " .. tostring(err))
end
-- ── reload handler (Task 6) ──
-- After gte and gp register successfully, load reload.lua through Support.extra.dofile and call its install(pcsx, support).
-- The whole sequence runs inside pcall so a missing zip, missing module table,
-- or throwing install never disturbs the gte and gp handlers already registered above (handler isolation).
--
-- The failure messages are intentionally single-line so the helper's boot log stays scannable.
if type(Support) == "table"
and type(Support.extra) == "table"
and type(Support.extra.dofile) == "function" then
local load_ok, reload_mod = pcall(Support.extra.dofile, "reload.lua")
if load_ok and type(reload_mod) == "table" and type(reload_mod.install) == "function" then
local install_ok, install_err = pcall(reload_mod.install, PCSX, Support)
if install_ok then
print("[pcsx_debug_helper] reload handler registered")
else
print("[pcsx_debug_helper] reload registration failed: " .. tostring(install_err))
end
else
print("[pcsx_debug_helper] reload load failed: " .. tostring(reload_mod))
end
else
print("[pcsx_debug_helper] reload load failed: Support.extra.dofile unavailable")
end
+902
View File
@@ -0,0 +1,902 @@
-- reload.lua - Side-effect-free hot-reload helper for the
-- pcsx_redux_hot_reload track (Task 2). This file owns the HTTP request
-- surface that the launch / reload client targets:
--
-- POST /api/v1/lua/reload?mode=prime&target=hello_camera&path=<encoded-elf>
-- POST /api/v1/lua/reload?mode=elf&target=hello_camera&path=<encoded-elf>
-- POST /api/v1/lua/reload?mode=patch&target=hello_camera&addr=...&hex=...
--
-- This module exposes the public surface used by the contract harness
-- (tests/reload_helper_contract.lua) and the runtime installed by
-- scripts/pcsx_debug_helper/autoexec.lua. The module must not reference
-- the global PCSX table at load time; the host is passed in explicitly
-- through M.new(host) and M.install(pcsx, support).
--
-- Public surface:
-- M.parse_query(query) -> table, nil OR nil, err_string
-- M.json_response(fields) -> string (sorted keys)
-- M.parse_manifest(...) -> Task 3 (real impl uses elf32.lua)
-- M.new(host) -> runtime object (Task 4; stub here)
-- M.install(pcsx, support) -> registers web handler (Task 6; stub here)
--
-- Companion: scripts/pcsx_debug_helper/autoexec.lua.
-- ---------------------------------------------------------------------------
-- Load the shared ELF32 helpers.
--
-- **The bane of this refactor:** the helper VM (PCSX-Redux) does not expose
-- `require` for paths outside the helper zip. The production loader is
-- `Support.extra.dofile("elf32.lua")` — Support.extra.dofile resolves the
-- name against the helper zip's contents (the zip is generated by the
-- build script and includes both `reload.lua` and `elf32.lua` after Task 6).
--
-- The test harness at `tests/reload_helper_contract.lua` loads `reload.lua`
-- via standard Lua `dofile` with an absolute path; it does not install a
-- `Support` object. We detect the runtime context: if `Support.extra.dofile`
-- exists, use it (production path); otherwise fall back to standard `dofile`
-- with an absolute path (test harness path).
-- ---------------------------------------------------------------------------
local function load_elf32()
if type(Support) == "table"
and type(Support.extra) == "table"
and type(Support.extra.dofile) == "function" then
return Support.extra.dofile("elf32.lua")
end
-- Test harness + any other context that supplies standard Lua dofile.
return dofile("C:/projects/Pikuma/ps1/scripts/elf32.lua")
end
local E = load_elf32()
local M = {}
-- ---------------------------------------------------------------------------
-- parse_query(query)
--
-- Parses an application/x-www-form-urlencoded query string into a table.
--
-- Rules (per spec §8 + plan.md Task 2 Step 3):
-- * Each pair is split on the first '='; the key is to the left, the value
-- to the right. A pair without '=' is a malformed_pair.
-- * Percent escapes '%HH' (HH = two hex digits) decode to the corresponding
-- byte. A '%' not followed by two hex digits is a malformed_escape.
-- * '+' decodes to a literal space (applied after percent decode).
-- * A key appearing more than once is a duplicate_key error.
--
-- Returns the parsed table on success. On failure returns nil and a stable
-- error string suitable for the JSON error envelope. An empty / nil query
-- returns an empty table (not an error).
-- ---------------------------------------------------------------------------
local function percent_decode(s)
-- Walk the string once, byte by byte. A '%' must be followed by exactly
-- two hex digits; '+' decodes to ' '; everything else is passed through.
local out = {}
local i = 1
local len = #s
while i <= len do
local c = s:sub(i, i)
if c == "%" then
if i + 2 > len then
return nil -- truncated escape (e.g., '%' at end or '%X')
end
local hex = s:sub(i + 1, i + 2)
local hd1, hd2 = hex:sub(1, 1), hex:sub(2, 2)
-- Validate both characters are hex digits.
if not (hd1:match("[0-9A-Fa-f]") and hd2:match("[0-9A-Fa-f]")) then
return nil -- malformed escape
end
out[#out + 1] = string.char(tonumber(hex, 16))
i = i + 3
else
out[#out + 1] = c
i = i + 1
end
end
return table.concat(out)
end
local function plus_to_space(s)
-- Standalone helper so callers can decode '+' after percent decoding.
return (s:gsub("+", " "))
end
function M.parse_query(query)
if query == nil or query == "" then
return {}, nil
end
local result = {}
local seen = {}
for pair in query:gmatch("[^&]+") do
-- Split on the first '=' only.
local eq = pair:find("=", 1, true)
if not eq then
return nil, "malformed_pair"
end
local raw_key = pair:sub(1, eq - 1)
local raw_value = pair:sub(eq + 1)
-- Percent-decode first, then convert '+' to space. The order matters:
-- a '%2B' should decode to '+' (literal plus), not be re-converted to a
-- space. Per RFC 1866 §8.2.1, '+' is a literal plus in the encoded form
-- only when it represents a space.
local key = percent_decode(raw_key)
if key == nil then
return nil, "malformed_escape"
end
key = plus_to_space(key)
local val = percent_decode(raw_value)
if val == nil then
return nil, "malformed_escape"
end
val = plus_to_space(val)
if seen[key] then
return nil, "duplicate_key"
end
seen[key] = true
result[key] = val
end
return result, nil
end
-- ---------------------------------------------------------------------------
-- json_response(fields)
--
-- Deterministic JSON object encoder. Returns a string. Keys are sorted
-- alphabetically before emission so byte-for-byte equality is testable
-- across runs and across PS1 captures.
--
-- Supported value types: string, number, boolean, nil (encoded as null).
-- Strings escape '\', '"', and the C0 control range (0x00..0x1F). The
-- named escapes use the conventional single-char forms: \\, \", \b, \f,
-- \n, \r, \t. Everything else in 0x00..0x1F is \uXXXX.
-- ---------------------------------------------------------------------------
local function json_escape_string(s)
-- Two passes: first the named escapes, then the catch-all C0 range
-- (%c covers 0x00..0x1F in Lua patterns). Using plain string.gsub
-- with a literal replacement table covers the named escapes; a
-- second gsub handles the rest.
s = s:gsub('[\\"]', {
["\\"] = "\\\\",
['"'] = '\\"',
})
s = s:gsub("\b", "\\b")
s = s:gsub("\f", "\\f")
s = s:gsub("\n", "\\n")
s = s:gsub("\r", "\\r")
s = s:gsub("\t", "\\t")
-- Remaining C0 control characters (0x00..0x1F) become \uXXXX. We
-- intentionally keep the named escapes above (which are already
-- single backslashes in the output) from being re-escaped: gsub on
-- the literal control char bytes doesn't match the backslashes we
-- already inserted.
s = s:gsub("([%c])", function(c)
return string.format("\\u%04x", string.byte(c))
end)
return s
end
function M.json_response(fields)
if type(fields) ~= "table" then
error("json_response: expected table, got " .. type(fields))
end
-- Sort keys for deterministic output. Lua's table.sort is byte-wise
-- and stable for strings; JSON object key order is not significant
-- but tests rely on a fixed order to compare against fixtures.
local keys = {}
for k in pairs(fields) do
keys[#keys + 1] = k
end
table.sort(keys)
local parts = {}
parts[#parts + 1] = "{"
for i = 1, #keys do
local k = keys[i]
if i > 1 then
parts[#parts + 1] = ","
end
parts[#parts + 1] = '"'
parts[#parts + 1] = json_escape_string(k)
parts[#parts + 1] = '":'
local v = fields[k]
local tv = type(v)
if tv == "string" then
parts[#parts + 1] = '"'
parts[#parts + 1] = json_escape_string(v)
parts[#parts + 1] = '"'
elseif tv == "number" then
parts[#parts + 1] = tostring(v)
elseif tv == "boolean" then
parts[#parts + 1] = v and "true" or "false"
elseif v == nil then
parts[#parts + 1] = "null"
else
error("json_response: unsupported value type " .. tv .. " for key " .. tostring(k))
end
end
parts[#parts + 1] = "}"
return table.concat(parts)
end
-- ---------------------------------------------------------------------------
-- ELF32 manifest parser (Task 3).
--
-- Parses a little-endian ELF32 file exposed through a file_adapter that
-- provides read_u8_at/read_u16_at/read_u32_at/read_size. The parser validates the
-- magic, class, data encoding, and machine before reading anything else.
-- It resolves section names through the .shstrtab table and symbols
-- through every SHT_SYMTAB section (and its linked string table).
--
-- The output manifest contains the state ABI the reload gate must
-- preserve plus the addresses the helper writes to the CPU on a reload.
-- Loaded sections (SHF_ALLOC, non-SHT_NOBITS) are recorded so the runtime
-- can reject any ELF whose loaded range overlaps the preserved smem.
--
-- **Refactor:** the format-constant tables + the byte-level walker live in
-- scripts/elf32.lua (loaded above via `load_elf32()`). This module retains
-- only the manifest-specific validation: required symbols, smem size, stack
-- alignment, loaded-section overlap. The net effect is ~80 lines shorter.
--
-- Stable error codes (returned as the second value):
-- bad_magic, unsupported_elf_class, unsupported_elf_data,
-- non_mips_machine, truncated_header, truncated_section_headers,
-- missing_shstrtab, missing_symtab_strtab, missing_smem,
-- missing_data_start, missing_data_end, missing_bss_start,
-- missing_bss_end, missing_stack_top, missing_hot_reload_entry,
-- zero_smem_size, stack_misaligned, stack_out_of_main_ram,
-- section_overlaps_smem, bad_file_adapter
-- ---------------------------------------------------------------------------
-- Convert a KSEG0/KSEG1/physical address to its physical main-RAM offset.
local function to_physical(addr)
if addr >= 0x80000000 and addr < 0x80200000 then
return addr - 0x80000000
elseif addr >= 0xa0000000 and addr < 0xa0200000 then
return addr - 0xa0000000
end
return addr
end
-- Strip KSEG0 / KSEG1 alias from an address and return the physical main-RAM
-- offset. Used by M.elf_reload and M.patch_handler. Returns nil when the
-- address falls outside physical main RAM (0..0x1fffff), KSEG0 main RAM
-- (0x80000000..0x801fffff), or KSEG1 main RAM (0xa0000000..0xa01fffff).
-- Per spec §7 the patch path MUST reject scratchpad (0x1F800000+), BIOS
-- (0x1FC00000+), MMIO, and expansion aliases; this helper centralizes the
-- strip + range check so callers cannot forget the upper bound.
local function strip_kseg(addr)
if type(addr) ~= "number" then return nil end
if addr >= 0x80000000 and addr < 0x80200000 then
return addr - 0x80000000
elseif addr >= 0xa0000000 and addr < 0xa0200000 then
return addr - 0xa0000000
elseif addr >= 0 and addr < 0x200000 then
return addr
end
return nil
end
-- Parse a hex string ("0xHHHH..." or "HHHH...") into a 32-bit unsigned
-- integer. Returns nil + stable error on absent / non-hex / out-of-range.
-- Used for both the patch path's addr/hex query parameters and any other
-- 32-bit hex field the API may add. Accepts up to 8 hex digits.
local function parse_hex_u32(s, missing_err, badhex_err)
if type(s) ~= "string" or #s == 0 then
return nil, missing_err or "missing_hex"
end
local clean = s:match("^0[xX]([0-9A-Fa-f]+)$")
or s:match("^([0-9A-Fa-f]+)$")
if not clean then return nil, badhex_err or "non_hex" end
if #clean > 8 then return nil, badhex_err or "non_hex" end
return tonumber(clean, 16), nil
end
-- Trap on a missing E.* — keeps the existing one-line-error pattern when
-- the helper zip is stale or absent.
local function stack()
io.stderr:write("[reload.parse_manifest] FATAL: scripts/elf32.lua not loaded; aborting\n")
error("elf32 module not loaded")
end
local function parse_manifest_impl(file_adapter, target, path, require_entry)
-- Wrap the body in a pcall so any thrown exception (e.g. a bad
-- adapter method or a malformed section header) surfaces as a
-- parse_error with the message and traceback instead of being lost
-- into the with_busy_guard xpcall as a generic internal_error.
local inner_ok, inner_result, inner_err = pcall(function()
-- Validate the adapter surface. E.validate_adapter returns the same
-- "bad_file_adapter" error code the prior implementation used.
local ok, err = E.validate_adapter(file_adapter)
if not ok then return nil, err end
-- Magic, class, data encoding. E.parse_elf32_headers reads fields at
-- the wire offsets specified in E.ELF32_HEADER.
local hdr, hdr_err = E.parse_elf32_headers(file_adapter)
if not hdr then return nil, hdr_err end
-- Machine check (e.g. EM_MIPS = 8). e_machine is at offset 0x12 (18).
-- The reload helper rejects non-MIPS ELFs before any symbol work.
-- Explicit pass style: E.read_u16(adapter, off). The helper wraps the
-- Support.File adapter once to strip its implicit `self` so the
-- parser shape stays flat-function, not colon-dispatch.
local machine = E.read_u16(file_adapter, 0x12)
if not machine then return nil, "truncated_header" end
if machine ~= E.EM_MIPS then
return nil, "non_mips_machine"
end
-- Walk sections. E.walk_sections also resolves .shstrtab names.
local sections, walk_err = E.walk_sections(file_adapter, hdr)
if not sections then return nil, walk_err end
-- Walk symbols. E.collect_symbols includes both STB_LOCAL and STB_GLOBAL
-- (the live ELF stores smem as a local symbol).
local symbols, sym_err = E.collect_symbols(file_adapter, sections)
if not symbols then return nil, sym_err end
-- Required symbols.
local smem = symbols["smem"]
local data_start = symbols["__data_start"]
local data_end = symbols["__data_end"]
local bss_start = symbols["__bss_start"]
local bss_end = symbols["__bss_end"]
local stack_top_s = symbols["__sp"]
local entry_s = symbols["hot_reload_entry"]
if not smem then return nil, "missing_smem" end
if not data_start then return nil, "missing_data_start" end
if not data_end then return nil, "missing_data_end" end
if not bss_start then return nil, "missing_bss_start" end
if not bss_end then return nil, "missing_bss_end" end
if not stack_top_s then return nil, "missing_stack_top" end
if require_entry and not entry_s then
return nil, "missing_hot_reload_entry"
end
-- Validate smem size.
if smem.size == 0 then
return nil, "zero_smem_size"
end
-- Validate stack alignment and range.
local stack_top = stack_top_s.value
if stack_top % 8 ~= 0 then
return nil, "stack_misaligned"
end
local p = to_physical(stack_top)
if p < 0 or p > 0x1fffff then
return nil, "stack_out_of_main_ram"
end
-- Collect loaded (SHF_ALLOC, non-SHT_NOBITS) sections and check overlap.
local loaded = {}
local smem_lo = smem.value
local smem_hi = smem.value + smem.size
for _, s in ipairs(sections) do
-- bit 1 (SHF_ALLOC = 0x2) of sh_flags. The modulo-4 trick matches
-- the prior implementation; canonicalising on E.SHF_ALLOC would
-- gain readability but lose the exact prior behavior.
local is_alloc = (s.sh_flags % 4) >= 2
if is_alloc and s.sh_type ~= E.SHT_NOBITS and s.sh_size > 0 then
loaded[#loaded + 1] = { name = s.name, addr = s.sh_addr, size = s.sh_size }
local lo = s.sh_addr
local hi = s.sh_addr + s.sh_size
if lo < smem_hi and hi > smem_lo then
return nil, "section_overlaps_smem"
end
end
end
return {
target = target,
elf_path = path,
elf_entry = hdr.e_entry,
smem_addr = smem.value,
smem_size = smem.size,
bss_start = bss_start.value,
bss_end = bss_end.value,
data_start = data_start.value,
data_end = data_end.value,
hot_reload_entry = entry_s and entry_s.value or nil,
stack_top = stack_top,
loaded_sections = loaded,
}
end)
if inner_ok then
return inner_result, inner_err
end
-- pcall captured a thrown error; surface as parse_error with the
-- message + traceback so the caller can render it.
local tb = debug.traceback(inner_result, 2)
local err = {
parse_error = true,
detail = tostring(inner_result),
tb = tb,
}
return nil, err
end
function M.parse_manifest(file_adapter, target, path, require_entry)
if type(E) ~= "table" or type(E.parse_elf32_headers) ~= "function" then
stack()
end
return parse_manifest_impl(file_adapter, target, path, require_entry)
end
-- ---------------------------------------------------------------------------
-- Runtime + dispatch (Task 4)
--
-- M.new(host) returns a runtime object that owns:
-- active -- the most recently primed manifest, or nil
-- busy -- boolean guard; only one request runs at a time
-- host -- the bound host surface (pause / memory_file / open_file
-- / binary_load / invalidate_cache / get_registers)
--
-- runtime:handle(req) parses the query through M.parse_query, validates
-- the mode against a dispatch table, then acquires the busy guard through
-- xpcall so any error inside the handler releases the guard. The response
-- is always a JSON string built by M.json_response.
--
-- M.prime_active and M.elf_reload are the two handler bodies Task 4 ships.
-- prime_active always parses with require_entry=false (Phase 0 binary
-- compatibility). elf_reload always parses with require_entry=true (the
-- new binary must expose hot_reload_entry). Both validate the parsed
-- manifest; elf_reload runs the five-field ABI gate before declaring
-- success. Full host.pause / memory_file / binary_load / invalidate_cache
-- / get_registers sequencing is Task 5.
-- ---------------------------------------------------------------------------
-- Convert a manifest into the JSON-serializable field subset. loaded_sections
-- is excluded because json_response only supports scalars + nil.
local function manifest_to_response(m)
local fields = {
ok = true,
target = m.target,
elf_path = m.elf_path,
elf_entry = m.elf_entry,
smem_addr = m.smem_addr,
smem_size = m.smem_size,
bss_start = m.bss_start,
bss_end = m.bss_end,
data_start = m.data_start,
data_end = m.data_end,
stack_top = m.stack_top,
}
if m.hot_reload_entry then
fields.hot_reload_entry = m.hot_reload_entry
end
return fields
end
-- Open the new ELF through the host and parse its manifest.
-- Returns manifest on success; nil + stable error on failure.
local function parse_manifest_via_host(host, target, path, require_entry)
local adapter = host.open_file(path)
if not adapter then
return nil, "open_file_failed"
end
return M.parse_manifest(adapter, target, path, require_entry)
end
-- prime_active: parse with require_entry=false. Accepts Phase 0 binaries
-- that lack hot_reload_entry. Stores the manifest in runtime.active.
function M.prime_active(runtime, parsed)
local manifest, err = parse_manifest_via_host(
runtime.host, parsed.target, parsed.path, false)
if not manifest then
return M.json_response({ ok = false, error = err, restart_required = true })
end
runtime.active = manifest
return M.json_response(manifest_to_response(manifest))
end
-- elf_reload: full host-driven reload sequence.
--
-- Per conductor/tracks/ps1_pcsx_redux_hot_reload_20260802/spec.md §5 +
-- plan.md Task 5 Step 4. The canonical 11-entry success log is:
--
-- pause, memory_file, state_read, open_new_elf, binary_load,
-- state_restore, invalidate_cache, get_registers, write_sp,
-- write_ra, write_pc
--
-- Sequencing:
--
-- 1. Validate the request (target == active.target, path present).
-- 2. Compute the physical address of `active.smem_addr` via
-- strip_kseg; reject if outside physical main RAM.
-- 3. PARSE PHASE (before pause):
-- a. elf_handle = host.open_file(parsed.path)
-- b. manifest = M.parse_manifest(elf_handle, ..., require_entry=true)
-- c. Run the five-field ABI gate against runtime.active.
-- d. On any rejection here, return BEFORE pause — the runtime
-- has invoked host.open_file once (logging "open_file") and
-- no other host methods.
-- 4. Pause + snapshot:
-- host.pause()
-- mem = host.memory_file()
-- saved = mem:readAtToSlice(active.smem_size, smem_phys)
-- 5. RELOAD PHASE:
-- elf_handle = host.open_new_elf(parsed.path) -- second open
-- loaded = host.binary_load(elf_handle, mem)
-- if loaded == nil then return binary_load_failed
-- 6. Restore state: mem:writeAtMoveSlice(saved, smem_phys)
-- 7. host.invalidate_cache()
-- 8. Rewrite SP / RA / PC through the FFI register pointer.
-- 9. Replace runtime.active last.
-- 10. Return the JSON envelope.
--
-- The two opens are an intentional test-discoverability choice. The
-- PARSE phase uses host.open_file (it is an existing Task 4 surface
-- also used by prime); the RELOAD phase uses host.open_new_elf (a
-- dedicated Task 5 method). In production both methods bind to
-- Support.File.open so the runtime cost is identical to a single open
-- — the distinction lives in the test log for ordering verification.
local function abi_mismatch_response(field, expected, actual)
return M.json_response({
ok = false, error = "state_abi_mismatch", field = field,
expected = expected, actual = actual,
restart_required = true,
})
end
function M.elf_reload(runtime, parsed)
-- 1. Pre-pause request validation. Pure-Lua, no host calls.
if not runtime.active then
return M.json_response({
ok = false, error = "not_primed", restart_required = false })
end
if parsed.target ~= runtime.active.target then
return M.json_response({
ok = false, error = "target_mismatch",
expected = runtime.active.target, actual = parsed.target,
restart_required = true })
end
if type(parsed.path) ~= "string" or parsed.path == "" then
return M.json_response({
ok = false, error = "missing_path",
restart_required = false })
end
-- 2. SMEM range check on `active` (the new ELF has not been
-- parsed yet; the ABI gate below enforces it cannot relocate).
local smem_phys = strip_kseg(runtime.active.smem_addr)
if smem_phys == nil or smem_phys < 0 or smem_phys > 0x1fffff then
return M.json_response({
ok = false, error = "smem_out_of_main_ram",
restart_required = true })
end
-- 3. PARSE PHASE — open + parse + ABI gate. On any rejection here,
-- only host.open_file has been called. Pause and downstream
-- mutations do NOT occur.
local elf_handle_for_parse = runtime.host.open_file(parsed.path)
if not elf_handle_for_parse then
return M.json_response({
ok = false, error = "open_file_failed",
restart_required = true })
end
local manifest, parse_err = M.parse_manifest(
elf_handle_for_parse, parsed.target, parsed.path, true)
if not manifest then
return M.json_response({
ok = false, error = parse_err,
restart_required = true })
end
local active = runtime.active
if manifest.smem_addr ~= active.smem_addr then
return abi_mismatch_response(
"smem_addr", active.smem_addr, manifest.smem_addr)
end
if manifest.smem_size ~= active.smem_size then
return abi_mismatch_response(
"smem_size", active.smem_size, manifest.smem_size)
end
if manifest.bss_start ~= active.bss_start then
return abi_mismatch_response(
"bss_start", active.bss_start, manifest.bss_start)
end
if manifest.bss_end ~= active.bss_end then
return abi_mismatch_response(
"bss_end", active.bss_end, manifest.bss_end)
end
-- 4. Pause + snapshot smem bytes.
runtime.host.pause()
local mem = runtime.host.memory_file()
local saved = mem:readAtToSlice(active.smem_size, smem_phys)
-- 5. RELOAD PHASE — second open for binary_load.
local elf_handle = runtime.host.open_new_elf(parsed.path)
if not elf_handle then
return M.json_response({
ok = false, error = "open_file_failed",
restart_required = true })
end
local loaded = runtime.host.binary_load(elf_handle, mem)
if loaded == nil then
-- Do NOT restore state; PCSX.Binary.load may have partially
-- written RAM. Keep ACTIVE untouched and tell the caller to
-- restart the emulator.
return M.json_response({
ok = false, error = "binary_load_failed",
restart_required = true })
end
-- 6. Restore the smem snapshot over the freshly-loaded code.
mem:writeAtMoveSlice(saved, smem_phys)
-- 7. Flush the CPU instruction cache (.text/.rodata changed).
runtime.host.invalidate_cache()
-- 8. Rewrite SP / RA / PC through the FFI register pointer. The
-- PC write must happen last; the CPU starts consuming
-- instructions at the new PC the moment the emulator resumes.
local regs = runtime.host.get_registers()
regs.GPR.n.sp = manifest.stack_top
regs.GPR.n.ra = 0
regs.pc = manifest.hot_reload_entry
-- 9. Replace ACTIVE last so a failed reload cannot poison the
-- next request's gate.
runtime.active = manifest
-- 10. Return the JSON envelope.
return M.json_response({
ok = true,
target = manifest.target,
elf_path = manifest.elf_path,
elf_entry = manifest.elf_entry,
smem_addr = manifest.smem_addr,
smem_size = manifest.smem_size,
bss_start = manifest.bss_start,
bss_end = manifest.bss_end,
data_start = manifest.data_start,
data_end = manifest.data_end,
hot_reload_entry = manifest.hot_reload_entry,
stack_top = manifest.stack_top,
})
end
-- patch_handler: one-word RAM patch through MemoryAsFile.
--
-- Per spec §7 + plan.md Task 5 Step 5, the order is:
-- 1. Parse addr and hex query parameters
-- 2. Reject non-hex / missing inputs
-- 3. Reject unaligned addresses (addr & 3)
-- 4. Normalize through strip_kseg; reject out-of-main-RAM
-- (scratchpad 0x1F800000+, BIOS 0x1FC00000+, MMIO, expansion)
-- 5. host.pause()
-- 6. mem = host.memory_file()
-- 7. mem:writeU32At(value, physical_offset)
-- 8. host.invalidate_cache()
-- 9. Return JSON envelope ok=true with the requested addr and value.
local function patch_error(err, restart)
return M.json_response({
ok = false, error = err,
restart_required = restart or false,
})
end
function M.patch_handler(runtime, parsed)
local addr_str = parsed.addr
local hex_str = parsed.hex
-- 1. Presence checks.
if type(addr_str) ~= "string" or addr_str == "" then
return patch_error("missing_addr", false)
end
if type(hex_str) ~= "string" or hex_str == "" then
return patch_error("missing_value", false)
end
-- 2. Hex parse.
local addr = parse_hex_u32(addr_str, "missing_addr", "non_hex_addr")
if not addr then
return patch_error(
addr == false and "missing_addr" or "non_hex_addr", false)
end
local value = parse_hex_u32(hex_str, "missing_value", "non_hex_value")
if not value then
return patch_error(
value == false and "missing_value" or "non_hex_value", false)
end
-- 3. Alignment (checked on the canonical KSEG/physical addr).
if addr % 4 ~= 0 then
return patch_error("addr_unaligned", false)
end
-- 4. Range check via strip_kseg (rejects KSEG0 > 0x801fffff, KSEG1 >
-- 0xa01fffff, scratchpad, BIOS, MMIO, expansion, etc.).
local phys = strip_kseg(addr)
if phys == nil then
return patch_error("addr_out_of_main_ram", false)
end
-- 5-8. Pause / write / cache invalidate.
runtime.host.pause()
local mem = runtime.host.memory_file()
mem:writeU32At(value, phys)
runtime.host.invalidate_cache()
-- 9. Return the JSON envelope. Echo the requested address and the
-- value in normalized hex so log captures stay stable across runs.
return M.json_response({
ok = true,
addr = addr_str,
value = "0x" .. string.format("%x", value),
})
end
-- Mode dispatch table. Each handler is invoked with (runtime, parsed).
-- Tasks 5 adds patch (M.patch_handler); the previous placeholder removed.
local DISPATCH = {
prime = M.prime_active,
elf = M.elf_reload,
patch = M.patch_handler,
}
-- Wrap a handler call with the busy guard. The guard is acquired only
-- after the mode is validated, so unknown-mode requests do not deadlock
-- the runtime. xpcall guarantees the guard is released even if the
-- handler throws.
local function with_busy_guard(runtime, fn)
if runtime.busy then
return M.json_response({
ok = false, error = "reload_busy", restart_required = false })
end
runtime.busy = true
-- Capture both the error text and a full Lua traceback so the user
-- can see the actual failing call site instead of a generic
-- "internal_error". debug.traceback("", 2) skips this xpcall frame
-- and the json_response frame so the trace starts at the handler.
local ok, result = xpcall(fn, function(e)
return { msg = tostring(e), tb = debug.traceback("", 2) }
end)
runtime.busy = false
if not ok then
return M.json_response({
ok = false, error = "internal_error",
detail = result.msg, tb = result.tb,
restart_required = true })
end
return result
end
function M.new(host)
if type(host) ~= "table" then
error("M.new: host must be a table, got " .. type(host))
end
local runtime = {
active = nil,
busy = false,
host = host,
}
function runtime:handle(req)
-- 1. Parse the query (M.parse_query returns nil, err on failure).
local query = req and req.urlData and req.urlData.query or ""
local parsed, parse_err = M.parse_query(query)
if not parsed then
return M.json_response({
ok = false, error = parse_err, restart_required = false })
end
-- 2. Validate the mode against the dispatch table.
local mode = parsed.mode
local handler = DISPATCH[mode]
if not handler then
return M.json_response({
ok = false, error = "unknown_mode", restart_required = false })
end
-- 3. Acquire busy and dispatch via xpcall. Mode validation
-- happens BEFORE busy is acquired so unknown-mode requests
-- cannot deadlock the runtime.
return with_busy_guard(self, function()
return handler(self, parsed)
end)
end
return runtime
end
-- Install the reload handler on a PCSX-Redux instance.
--
-- Per plan.md Task 5 Step 5 the adapter binds the canonical host method
-- names to the PCSX-Lua FFI surface:
--
-- pause -> PCSX.pauseEmulator
-- memory_file -> PCSX.getMemoryAsFile
-- open_file -> Support.File.open(path, "READ")
-- binary_load -> PCSX.Binary.load
-- invalidate_cache -> PCSX.invalidateCache
-- get_registers -> PCSX.getRegisters
--
-- The returned closure dispatches each request through M.new(host)'s
-- runtime:handle so the same prime/elf/patch dispatch machinery is used
-- (including the busy guard from Task 4).
--
-- Missing `PCSX.WebServer.Handlers` is created on demand so callers do
-- not have to wire that themselves; if `PCSX` or `Support` is absent a
-- single line is printed and the function returns without registering
-- a handler.
function M.install(pcsx, support)
if type(pcsx) ~= "table" then
print("[reload] install failed: PCSX is not a table")
return
end
if type(support) ~= "table"
or type(support.File) ~= "table"
or type(support.File.open) ~= "function" then
print("[reload] install failed: Support.File.open unavailable")
return
end
if type(pcsx.pauseEmulator) ~= "function" then print("[reload] install failed: PCSX.pauseEmulator missing"); return end
if type(pcsx.getMemoryAsFile) ~= "function" then print("[reload] install failed: PCSX.getMemoryAsFile missing"); return end
if type(pcsx.Binary) ~= "table"
or type(pcsx.Binary.load) ~= "function" then print("[reload] install failed: PCSX.Binary.load missing"); return end
if type(pcsx.invalidateCache) ~= "function" then print("[reload] install failed: PCSX.invalidateCache missing"); return end
if type(pcsx.getRegisters) ~= "function" then print("[reload] install failed: PCSX.getRegisters missing"); return end
-- ---------------------------------------------------------------------------
-- File adapter wrap.
--
-- The production pcsx-redux Support.File wrapper (see
-- toolchain/pcsx-redux/src/lua/fileffi.lua:225-232 + size() around line 203)
-- exposes byte-read methods as colon-syntax closures with camelCase names:
-- readU8At = function(self, pos) ... end
-- readU16At = function(self, pos) ... end
-- readU32At = function(self, pos) ... end
-- size = function(self) ... end
--
-- The ELF32 parser (scripts/elf32.lua) uses an explicit-pass shape with
-- snake_case names:
-- adapter.read_u8_at(off) / adapter.read_u16_at(off) /
-- adapter.read_u32_at(off) / adapter.read_size()
--
-- The install boundary wraps the Support.File return value in a thin
-- adapter whose methods forward to the production closures, stripping
-- the implicit `self` and re-exporting the names the parser validates.
-- Without this wrap, E.validate_adapter returns "bad_file_adapter"
-- because adapter.read_u8_at / read_u16_at / read_u32_at / read_size
-- are not present on the raw Support.File return.
local function wrap_file(f)
return {
read_u8_at = function(off) return f:readU8At(off) end,
read_u16_at = function(off) return f:readU16At(off) end,
read_u32_at = function(off) return f:readU32At(off) end,
read_size = function() return f:size() end,
}
end
local host = {
pause = function() pcsx.pauseEmulator() end,
memory_file = function() return pcsx.getMemoryAsFile() end,
open_file = function(path) return wrap_file(support.File.open(path, "READ")) end,
-- open_new_elf returns the raw Support.File object because the
-- RELOAD phase passes it directly to PCSX.Binary.load which
-- expects a real File (with readAt / size), NOT the elf32
-- parser adapter (read_u8_at / read_u16_at / read_u32_at /
-- read_size). Wrapping it in the adapter here triggers the
-- binffi.lua "Expected a File object as first argument" error.
open_new_elf = function(path) return support.File.open(path, "READ") end,
binary_load = function(elf, mem) return pcsx.Binary.load(elf, mem) end,
invalidate_cache = function() pcsx.invalidateCache() end,
get_registers = function() return pcsx.getRegisters() end,
}
local runtime = M.new(host)
if type(pcsx.WebServer) ~= "table" then pcsx.WebServer = {} end
if type(pcsx.WebServer.Handlers) ~= "table" then pcsx.WebServer.Handlers = {} end
pcsx.WebServer.Handlers.reload = function(req)
return runtime:handle(req)
end
print("[reload] handler installed: reload")
end
return M
-6
View File
@@ -118,12 +118,6 @@ local PASSES = {
kind = "header-output",
deps = {"scan-source", "word-counts"},
},
auto_reg = {
module = "passes.auto_reg",
kind = "header-output",
deps = {"components"},
groups = { "pre-link" },
},
["emission-model"] = {
module = "passes.emission_model",
kind = "validation",
+82
View File
@@ -0,0 +1,82 @@
# scripts/reload.ps1
#
# PCSX-Redux Lua helper reload client.
#
# Modes:
# elf - Request a full ELF reload. Requires -ElfPath.
# patch - Request a single-word RAM patch. Requires -Address and -Word.
#
# -RequestOnly prints the URI and exits before any network I/O.
# -Quiet suppresses the compact-JSON printout on the real path.
[CmdletBinding()]
param(
[ValidateSet('elf', 'patch')][string]$Mode = 'elf',
[string]$Target = 'hello_camera',
[string]$ElfPath = '',
[string]$Address = '',
[string]$Word = '',
[int]$Port = 8080,
[switch]$RequestOnly,
[switch]$Quiet
)
# mode-specific argument guards
switch ($Mode) {
'patch' {
if ([string]::IsNullOrEmpty($Address) -or [string]::IsNullOrEmpty($Word)) {
Write-Error "patch mode requires both -Address and -Word"
exit 1
}
}
'elf' {
if ([string]::IsNullOrEmpty($ElfPath)) {
Write-Error "elf mode requires -ElfPath"
exit 1
}
}
}
# Build the URL-encoded query string.
$queryParts = New-Object System.Collections.Generic.List[string]
[void]$queryParts.Add("mode=$([uri]::EscapeDataString($Mode))")
[void]$queryParts.Add("target=$([uri]::EscapeDataString($Target))")
switch ($Mode) {
'elf' {
[void]$queryParts.Add("path=$([uri]::EscapeDataString($ElfPath))")
}
'patch' {
[void]$queryParts.Add("addr=$([uri]::EscapeDataString($Address))")
[void]$queryParts.Add("hex=$([uri]::EscapeDataString($Word))")
}
}
$uri = "http://localhost:$Port/api/v1/lua/reload?$($queryParts -join '&')"
# RequestOnly path: emit URI and return before any network I/O.
if ($RequestOnly) {
Write-Output $uri
return
}
# Real request path: POST, decode body if it is a byte array, parse JSON.
$response = Invoke-WebRequest -Method Post -Uri $uri
if ($response.Content -is [byte[]]) {
$text = [System.Text.Encoding]::UTF8.GetString([byte[]]$response.Content)
}
else {
$text = [string]$response.Content
}
$obj = $text | ConvertFrom-Json
if (-not $Quiet) {
$obj | ConvertTo-Json -Compress | Write-Output
}
if (-not $obj.ok) {
$errCode = if ($obj.error) { [string]$obj.error } else { 'unknown' }
throw "Reload failed: $errCode"
}
-129
View File
@@ -14,21 +14,16 @@ $url_armips = 'https://github.com/Kingcom/armips.git'
$url_pcsx_redux = 'https://github.com/grumpycoders/pcsx-redux.git'
$url_psyq_iwyu = 'https://github.com/johnbaumann/psyq_include_what_you_use.git'
$url_lpeg = 'https://github.com/roberto-ieru/LPeg.git'
# $url_mkpsxiso = 'https://github.com/Lameguy64/mkpsxiso.git'
$url_mkpsxiso_win64 = 'https://github.com/Lameguy64/mkpsxiso/releases/download/v2.30/mkpsxiso-2.30-win64.zip'
$path_armips = join-path $path_toolchain 'armips'
$path_pcsx_redux = join-path $path_toolchain 'pcsx-redux'
$path_psyq_iwyu = join-path $path_toolchain 'psyq_iwyu'
$path_lpeg = join-path $path_toolchain 'lpeg'
$path_mkpsxiso = join-path $path_toolchain 'mkpsxiso'
clone-gitrepo $path_armips $url_armips
clone-gitrepo $path_lpeg $url_lpeg
clone-gitrepo $path_pcsx_redux $url_pcsx_redux
clone-gitrepo $path_psyq_iwyu $url_psyq_iwyu
# clone-gitrepo $path_mkpsxiso $url_mkpsxiso
$path_armips_build = join-path $path_armips 'build'
verify-path $path_armips_build
@@ -61,119 +56,6 @@ if (-not $msbuild_exe) {
}
$path_pcsx_sln = join-path $path_pcsx_redux 'vsprojects\pcsx-redux.sln'
# ════════════════════════════════════════════════════════════════════════════
# NuGet restore — required before MSBuild.
# pcsx-redux's .vcxproj files use the legacy packages.config style with
# hardcoded `<Import Project="..\packages\{id}.{ver}\...">` directives.
# MSBuild's `/t:Restore` won't fetch missing packages here (the local
# packages\ dir is checked but no package-source lookup happens), and
# `dotnet restore` errors on packages.config projects, so we walk every
# packages.config, parse out the <package id version/> entries, and pull
# any missing .nupkg directly from api.nuget.org's flat container.
# ════════════════════════════════════════════════════════════════════════════
$path_pcsx_packages = join-path $path_pcsx_redux 'vsprojects\packages'
$nuget_flat_container = 'https://api.nuget.org/v3-flatcontainer'
# Collect required (id, version) pairs from every packages.config.
$required_packages = @{}
Get-ChildItem -Path (join-path $path_pcsx_redux 'vsprojects') -Filter 'packages.config' -Recurse -ErrorAction SilentlyContinue |
ForEach-Object {
[xml]$xml = Get-Content -LiteralPath $_.FullName -Raw
foreach ($pkg in $xml.packages.package) {
$key = '{0}|{1}' -f $pkg.id, $pkg.version
$required_packages[$key] = @{ id = $pkg.id; version = $pkg.version }
}
}
# Ensure the packages root exists.
if (-not (Test-Path -LiteralPath $path_pcsx_packages)) {
New-Item -ItemType Directory -Path $path_pcsx_packages -Force | Out-Null
}
# Download anything missing. Skip the package entirely if its dir already has
# any contents (the legacy packages.config style means the targets file
# location varies per package — `luajit.native` puts it at build/native/,
# `glfw` puts it elsewhere — so we can't probe a specific path; just check
# whether the dir is non-empty).
Add-Type -AssemblyName System.IO.Compression.FileSystem
foreach ($pkg in $required_packages.Values) {
$pkgDir = Join-Path $path_pcsx_packages ('{0}.{1}' -f $pkg.id, $pkg.version)
if ((Test-Path -LiteralPath $pkgDir) -and `
(@(Get-ChildItem -LiteralPath $pkgDir -Recurse -ErrorAction SilentlyContinue).Count -gt 0)) {
continue
}
$url = '{0}/{1}/{2}/{1}.{2}.nupkg' -f $nuget_flat_container, $pkg.id, $pkg.version
$nupkg = Join-Path $pkgDir ('{0}.{1}.nupkg' -f $pkg.id, $pkg.version)
New-Item -ItemType Directory -Path $pkgDir -Force | Out-Null
Write-Host "Fetching NuGet package: $($pkg.id) $($pkg.version)"
try {
Invoke-WebRequest -Uri $url -OutFile $nupkg -UseBasicParsing -ErrorAction Stop
[System.IO.Compression.ZipFile]::ExtractToDirectory($nupkg, $pkgDir)
Remove-Item -LiteralPath $nupkg -Force
} catch {
$msg = $_.Exception.Message
if ($msg -match '404') {
Write-Host " Not on nuget.org (vendored?) — skipping $url"
} else {
Write-Warning "Failed to fetch $url$msg"
}
if (Test-Path -LiteralPath $nupkg) { Remove-Item -LiteralPath $nupkg -Force }
}
}
# ════════════════════════════════════════════════════════════════════════════
# isoffi.lua size guard — `core.vcxproj` #includes src/core/isoffi.lua into
# luaiso.cc via the `-- lualoader, R"EOF(...)EOF"` trick. The raw string
# literal between R"EOF(-- and -- )EOF" must stay under ~16,379 bytes or
# MSVC (19.44) fails with C2026 (its actual raw-string limit is 16,384,
# minus 5 bytes for the `-- lualoader, ` prefix). If the upstream file
# grows past that, trim it: remove license header, trailing whitespace,
# blank separators, inline comments, and shrink 4-space indent to 2-space.
# Idempotent — only writes when the raw string exceeds the limit.
# ════════════════════════════════════════════════════════════════════════════
$path_isoffi = join-path $path_pcsx_redux 'src\core\isoffi.lua'
if (Test-Path -LiteralPath $path_isoffi) {
$content = Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8
$startMarker = $content.IndexOf('R"EOF(--')
$endMarker = $content.IndexOf('-- )EOF"')
$literalLen = if ($startMarker -ge 0 -and $endMarker -gt $startMarker) {
$endMarker - ($startMarker + 8)
} else { -1 }
# Effective MSVC raw-string limit for the lualoader prefix is 16379 bytes.
if ($literalLen -gt 16379) {
Write-Host "isoffi.lua raw string is $literalLen bytes (>16379); trimming for MSVC C2026 limit."
$lines = $content -split "`n"
$markerIdx = -1
for ($i = 0; $i -lt $lines.Length; $i++) {
if ($lines[$i] -match '^-- \)EOF"') { $markerIdx = $i; break }
}
$newLines = @()
for ($i = 0; $i -lt $lines.Length; $i++) {
$lineNum = $i + 1
$line = $lines[$i]
# Keep the first line and the EOF-marker line untouched.
if ($i -eq 0 -or $i -eq $markerIdx) { $newLines += $line; continue }
# Drop the GPL license header (lines 2-17).
if ($lineNum -ge 2 -and $lineNum -le 17) { continue }
# Drop blank separator lines.
if ($line -match '^\s*$') { continue }
# Drop trailing whitespace.
$line = $line -replace '\s+$', ''
# Drop inline comments (anything from `--` to end of line).
$line = $line -replace '\s*--.*$', ''
# Shrink 4-space indent to 2-space.
$line = $line -replace '^( )', ' '
if ($line -match '^\s*$') { continue }
$newLines += $line
}
($newLines -join "`n") | Out-File -LiteralPath $path_isoffi -Encoding utf8 -NoNewline
$newLen = ((Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8) `
-replace '.*R"EOF\(--', '' -replace '-- \)EOF".*', '').Length
Write-Host "isoffi.lua trimmed: $literalLen -> $newLen bytes of raw string content."
}
}
& $msbuild_exe $path_pcsx_sln /p:Configuration=Release /p:Platform=x64 /p:PlatformToolset=v143 /m /v:minimal
# Locate luajit via scoop. `luajit.exe` is on PATH via scoop's shim;
@@ -230,17 +112,6 @@ $lfs_dll_import = join-path $luajit_lib_dir 'libluajit-5.1.dll.a'
# ════════════════════════════════════════════════════════════════════════════
$path_openbios = join-path $path_pcsx_redux 'src\mips\openbios'
# Wipe stale *.dep files across src\mips. These cache absolute paths to the
# GCC headers directory; if the toolchain was upgraded (e.g. v14.2.0 → v16.1.0)
# Make reads the stale paths and aborts with "no rule to make target .../stddef.h".
# `make clean` in openbios only clears its own dir — subdirs like
# common/crt0/, modplayer/, and shell/ keep their stale .dep files. Easier to
# just delete the lot before each build than to teach every Makefile about
# deepclean recursion.
Get-ChildItem -Path (join-path $path_pcsx_redux 'src\mips') -Recurse -Filter '*.dep' -ErrorAction SilentlyContinue |
ForEach-Object { Remove-Item -LiteralPath $_.FullName -Force }
push-location $path_openbios
& make clean
& make