mirror of
https://github.com/Ed94/pikuma_ps1.git
synced 2026-09-22 17:20:04 +00:00
Compare commits
15
Commits
e2ffe538b6
..
master
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5f49c071d1 | ||
|
|
e6cd2539d8 | ||
|
|
0e9034cbd4 | ||
|
|
1faf3539d8 | ||
|
|
6b3fbab387 | ||
|
|
2c38a73709 | ||
|
|
159ead625e | ||
|
|
52888015de | ||
|
|
a37ffe6f58 | ||
|
|
b61610d819 | ||
|
|
1b950ab5b5 | ||
|
|
b2858b3c73 | ||
|
|
f1801343e2 | ||
|
|
85b2205603 | ||
|
|
2d754650c9 |
@@ -30,7 +30,7 @@
|
||||
"editorHoverWidget.background": "#2c334b"
|
||||
},
|
||||
"semanticTokenColors": {
|
||||
"comment": { "foreground": "#868686", "fontStyle": "italic" },
|
||||
"comment": { "foreground": "#868686", }, //"fontStyle": "italic" },
|
||||
"keyword": { "foreground": "#d8bd5b" },
|
||||
"string": { "foreground": "#d46a54" },
|
||||
"number": { "foreground": "#b5cea8" },
|
||||
@@ -81,7 +81,7 @@
|
||||
"tapeDelaySlot": { "foreground": "#ff5647" }
|
||||
},
|
||||
"tokenColors": [
|
||||
{ "scope": ["comment", "comment.block", "comment.line", "comment.block.documentation"], "settings": { "foreground": "#868686", "fontStyle": "italic" } },
|
||||
{ "scope": ["comment", "comment.block", "comment.line", "comment.block.documentation"], "settings": { "foreground": "#868686", } }, //"fontStyle": "italic" } },
|
||||
{ "scope": ["keyword", "keyword.control", "keyword.other"], "settings": { "foreground": "#d8bd5b" } },
|
||||
{ "scope": ["string", "string.quoted"], "settings": { "foreground": "#d46a54" } },
|
||||
{ "scope": ["string.quoted.other"], "settings": { "foreground": "#d69d85" } },
|
||||
|
||||
+119
-74
@@ -21,72 +21,108 @@ const TOKEN_TYPES = [
|
||||
"tapeGprRegister",
|
||||
"tapeCop2Register",
|
||||
"tapeDuffleType",
|
||||
"tapeAttribute",
|
||||
"tapeAt__ibute",
|
||||
"keyword",
|
||||
"macro",
|
||||
];
|
||||
|
||||
const TOKEN_MODIFIERS = ["declaration", "tapeRead", "tapeWrite", "tapeAuto"];
|
||||
const TOKEN_TYPE_INDEX = new Map(TOKEN_TYPES.map((name, index) => [name, index]));
|
||||
const TOKEN_MODIFIERS = ["declaration", "tapeRead", "tapeWrite", "tapeAuto"];
|
||||
const TOKEN_TYPE_INDEX = new Map(TOKEN_TYPES.map((name, index) => [name, index]));
|
||||
const TOKEN_MODIFIER_INDEX = new Map(TOKEN_MODIFIERS.map((name, index) => [name, index]));
|
||||
|
||||
const ATOM_KEYWORDS = new Set(["MipsAtom_", "MipsAtom_Proc_"]);
|
||||
const COMPONENT_KEYWORDS = new Set(["MipsAtomComp_", "MipsAtomComp_Proc_"]);
|
||||
const ANNOTATIONS = new Set([
|
||||
"atom_info", "atom_bind", "atom_reads", "atom_writes", "atom_label",
|
||||
"atom_offset", "atom_reg", "atom_type", "atom_ctx", "atom_phase",
|
||||
"atom_auto_reg", "phase_auto_reg", "atom_dbg_skip",
|
||||
"atom_info",
|
||||
"atom_bind",
|
||||
"atom_reads",
|
||||
"atom_writes",
|
||||
"atom_label",
|
||||
"atom_offset",
|
||||
"atom_reg",
|
||||
"atom_type",
|
||||
"atom_ctx",
|
||||
"atom_phase",
|
||||
"atom_auto_reg",
|
||||
"phase_auto_reg",
|
||||
"atom_dbg_skip",
|
||||
]);
|
||||
|
||||
const DSL_KEYWORDS = new Set([
|
||||
"FI_", "I_", "NI_", "Relative_", "Struct_", "Enum_", "Union_", "Array_",
|
||||
"Slice_", "TypeR_", "TypeV_", "align_", "internal", "local_persist", "global",
|
||||
"RO_", "LP_", "gknown", "expect_", "cexpr_",
|
||||
"enum", "struct", "union",
|
||||
"offset_of", "static_assert", "typeof", "typeof_ptr", "typeof_same",
|
||||
"glue", "tmpl",
|
||||
"A_", "FI_", "I_", "NI_",
|
||||
"Array_", "Enum_", "Proc_", "Relative_", "Struct_", "Union_", "Slice_",
|
||||
|
||||
// "TypeR_", "TypeV_",
|
||||
|
||||
"align_",
|
||||
"internal", "local_persist", "global",
|
||||
|
||||
"RO_", "LP_",
|
||||
"gknown", "expect_", "cexpr_",
|
||||
|
||||
"O_", "OA_", "S_", "C_", "T_", "T_same", "R_", "V_",
|
||||
"r_", "v_", "rt_", "vt_",
|
||||
|
||||
"asm", "asm_words", "asm_rpins", "asm_clobber",
|
||||
"O_", "S_", "C_", "T_", "tmpl", "glue", "r_", "v_", "tr_", "tv_",
|
||||
"rgcc", "r_use", "r_set", "r_mod", "r_imm", "r_mem",
|
||||
"u1_", "u2_", "u4_", "u8_", "s1_", "s2_", "s4_", "s8_",
|
||||
"u1_r", "u2_r", "u4_r", "u8_r", "u1_v", "u2_v", "u4_v", "u8_v",
|
||||
|
||||
"b1_", "b2_", "b4_", "b8_",
|
||||
"u1_", "u2_", "u4_", "u8_",
|
||||
"s1_", "s2_", "s4_", "s8_",
|
||||
"b1_r", "b2_r", "b4_r", "b8_r",
|
||||
"b1_v", "b2_v", "b4_v", "b8_v",
|
||||
"u1_r", "u2_r", "u4_r", "u8_r",
|
||||
"u1_v", "u2_v", "u4_v", "u8_v",
|
||||
|
||||
"u4_lo", "u4_hi",
|
||||
]);
|
||||
|
||||
const DELAY_SLOT_KEYWORDS = new Set(["LdSlot_", "BdSlot_", "DmaSlot_", "GteDelay_"]);
|
||||
const DELAY_SLOT_KEYWORDS = new Set([
|
||||
"LdSlot_",
|
||||
"BdSlot_",
|
||||
"DmaSlot_",
|
||||
"GteDelay_"
|
||||
]);
|
||||
|
||||
const CONTROL_FLOW_PREFIXES = /^(?:branch_|jump_|call_)/;
|
||||
|
||||
const ROLE_TO_TYPE = {
|
||||
atomName: "tapeAtomName",
|
||||
atomName: "tapeAtomName",
|
||||
componentName: "tapeComponentName",
|
||||
bindType: "tapeBindType",
|
||||
duffleType: "tapeDuffleType",
|
||||
gprRegister: "tapeGprRegister",
|
||||
cop2Register: "tapeCop2Register",
|
||||
bindType: "tapeBindType",
|
||||
duffleType: "tapeDuffleType",
|
||||
gprRegister: "tapeGprRegister",
|
||||
cop2Register: "tapeCop2Register",
|
||||
};
|
||||
|
||||
function registerType(name, index) {
|
||||
const kind = index.registers.get(name);
|
||||
if (kind === "gpr" || /^R_[A-Za-z0-9_]+$/.test(name)) return "tapeGprRegister";
|
||||
if (kind === "cop2" || /^(?:C2_|gte_cr_)[A-Za-z0-9_]+$/.test(name)) return "tapeCop2Register";
|
||||
if (kind === "gpr" || /^R_[A-Za-z0-9_] + $/.test(name)) return "tapeGprRegister";
|
||||
if (kind === "cop2" || /^(?:C2_|gte_cr_)[A-Za-z0-9_] + $/.test(name)) return "tapeCop2Register";
|
||||
return null;
|
||||
}
|
||||
|
||||
function instructionType(name, index) {
|
||||
const domain = index.macros.get(name);
|
||||
if (domain === "control") return "tapeControlFlow";
|
||||
if (domain === "cpu") return "tapeCpuInstruction";
|
||||
if (domain === "gte") return "tapeGteInstruction";
|
||||
if (domain === "gpu") return "tapeGpuInstruction";
|
||||
if (domain === "cpu") return "tapeCpuInstruction";
|
||||
if (domain === "gte") return "tapeGteInstruction";
|
||||
if (domain === "gpu") return "tapeGpuInstruction";
|
||||
if (domain === "component") {
|
||||
if (/^mac_gte_/.test(name)) return "tapeGteInstruction";
|
||||
if (/^mac_gp/.test(name)) return "tapeGpuInstruction";
|
||||
if (/^mac_/.test(name)) return "tapeComponentInstruction";
|
||||
if (/^mac_gp/.test(name)) return "tapeGpuInstruction";
|
||||
if (/^mac_/.test(name)) return "tapeComponentInstruction";
|
||||
return "macro";
|
||||
}
|
||||
if (domain === "utility") return "macro";
|
||||
if (domain === "utility") return "macro";
|
||||
if (/^gte_(?!cr_)/.test(name)) return "tapeGteInstruction";
|
||||
if (/^gp[01]_/.test(name)) return "tapeGpuInstruction";
|
||||
if (/^mac_gte_/.test(name)) return "tapeGteInstruction";
|
||||
if (/^mac_gp/.test(name)) return "tapeGpuInstruction";
|
||||
if (/^mac_/.test(name)) return "tapeComponentInstruction";
|
||||
if (/^gp[01]_/.test(name)) return "tapeGpuInstruction";
|
||||
if (/^mac_gte_/.test(name)) return "tapeGteInstruction";
|
||||
if (/^mac_gp/.test(name)) return "tapeGpuInstruction";
|
||||
if (/^mac_/.test(name)) return "tapeComponentInstruction";
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -100,13 +136,11 @@ function modifierMask(modifiers) {
|
||||
}
|
||||
|
||||
function isRegUseAccess(tokens, tokenIndex) {
|
||||
const prev = tokens[tokenIndex - 1];
|
||||
if (!prev || prev.text !== ".") return false;
|
||||
const prevPrev = tokens[tokenIndex - 2];
|
||||
if (!prevPrev || prevPrev.kind !== "identifier") return false;
|
||||
const next = tokens[tokenIndex + 1];
|
||||
const prev = tokens[tokenIndex - 1]; if (!prev || prev.text !== ".") return false;
|
||||
const prevPrev = tokens[tokenIndex - 2]; if (!prevPrev || prevPrev.kind !== "identifier") return false;
|
||||
const next = tokens[tokenIndex + 1];
|
||||
if (next && next.text === ".") return false;
|
||||
if (prevPrev.text === "r") return true;
|
||||
if (prevPrev.text === "r") return true;
|
||||
const prev3 = tokens[tokenIndex - 3];
|
||||
const prev4 = tokens[tokenIndex - 4];
|
||||
if (prev3 && prev3.text === "." && prev4 && prev4.kind === "identifier" && prev4.text === "r") return true;
|
||||
@@ -115,81 +149,92 @@ function isRegUseAccess(tokens, tokenIndex) {
|
||||
|
||||
function classifyDocument(source, filePath, workspaceIndex, shouldCancel = () => false) {
|
||||
const scanned = scanSource(source, filePath);
|
||||
const index = mergeIndexes(workspaceIndex, scanned.index);
|
||||
const spans = [];
|
||||
const index = mergeIndexes(workspaceIndex, scanned.index);
|
||||
const spans = [];
|
||||
|
||||
for (let tokenIndex = 0; tokenIndex < scanned.tokens.length; tokenIndex += 1) {
|
||||
if (shouldCancel()) break;
|
||||
const token = scanned.tokens[tokenIndex];
|
||||
if (token.kind !== "identifier") continue;
|
||||
const token = scanned.tokens[tokenIndex]; if (token.kind !== "identifier") continue;
|
||||
|
||||
let type = null;
|
||||
let modifiers = [];
|
||||
let type = null;
|
||||
let modifiers = [];
|
||||
const declaration = scanned.declarations.get(token.start);
|
||||
const context = nearestCall(scanned.contexts, tokenIndex);
|
||||
const context = nearestCall(scanned.contexts, tokenIndex);
|
||||
|
||||
if (declaration) {
|
||||
type = ROLE_TO_TYPE[declaration.role] || null;
|
||||
type = ROLE_TO_TYPE[declaration.role] || null;
|
||||
modifiers = declaration.modifiers.slice();
|
||||
} else if (ATOM_KEYWORDS.has(token.text)) {
|
||||
}
|
||||
else if (ATOM_KEYWORDS.has(token.text)) {
|
||||
type = "tapeAtomKeyword";
|
||||
} else if (COMPONENT_KEYWORDS.has(token.text)) {
|
||||
}
|
||||
else if (COMPONENT_KEYWORDS.has(token.text)) {
|
||||
type = "keyword";
|
||||
} else if (ANNOTATIONS.has(token.text)) {
|
||||
}
|
||||
else if (ANNOTATIONS.has(token.text)) {
|
||||
type = "tapeAnnotation";
|
||||
} else if (context && context.callee === "atom_bind" && context.argIndex === 0) {
|
||||
}
|
||||
else if (context && context.callee === "atom_bind" && context.argIndex === 0) {
|
||||
type = "tapeBindType";
|
||||
} else if (context && context.callee === "atom_phase" && context.argIndex === 0) {
|
||||
}
|
||||
else if (context && context.callee === "atom_phase" && context.argIndex === 0) {
|
||||
type = "tapePhase";
|
||||
modifiers = ["declaration"];
|
||||
} else if (context && context.callee === "atom_ctx" && context.argIndex === 0) {
|
||||
}
|
||||
else if (context && context.callee === "atom_ctx" && context.argIndex === 0) {
|
||||
type = "tapeAtomName";
|
||||
} else if (context && context.callee === "atom_label" && context.argIndex === 0) {
|
||||
}
|
||||
else if (context && context.callee === "atom_label" && context.argIndex === 0) {
|
||||
type = "tapeLabel";
|
||||
modifiers = ["declaration"];
|
||||
} else if (context && context.callee === "atom_offset" && context.argIndex <= 1) {
|
||||
}
|
||||
else if (context && context.callee === "atom_offset" && context.argIndex <= 1) {
|
||||
type = "tapeLabel";
|
||||
} else if (context && context.callee === "atom_reads") {
|
||||
}
|
||||
else if (context && context.callee === "atom_reads") {
|
||||
type = registerType(token.text, index);
|
||||
if (type) modifiers = ["tapeRead"];
|
||||
} else if (context && context.callee === "atom_writes") {
|
||||
}
|
||||
else if (context && context.callee === "atom_writes") {
|
||||
type = registerType(token.text, index);
|
||||
if (type) modifiers = ["tapeWrite"];
|
||||
} else if (context && context.callee === "atom_auto_reg") {
|
||||
}
|
||||
else if (context && context.callee === "atom_auto_reg") {
|
||||
if (context.argIndex === 0) type = "tapeAtomName";
|
||||
if (context.argIndex === 1) {
|
||||
type = "tapeGprRegister";
|
||||
type = "tapeGprRegister";
|
||||
modifiers = ["declaration", "tapeAuto"];
|
||||
}
|
||||
} else if (context && context.callee === "phase_auto_reg") {
|
||||
}
|
||||
else if (context && context.callee === "phase_auto_reg") {
|
||||
if (context.argIndex === 0) type = "tapePhase";
|
||||
if (context.argIndex === 1) {
|
||||
type = "tapeGprRegister";
|
||||
type = "tapeGprRegister";
|
||||
modifiers = ["declaration", "tapeAuto"];
|
||||
}
|
||||
}
|
||||
|
||||
if (!type && index.bindTypes.has(token.text)) type = "tapeBindType";
|
||||
if (!type && DSL_KEYWORDS.has(token.text)) type = "keyword";
|
||||
if (!type && index.types.has(token.text)) type = "tapeDuffleType";
|
||||
if (!type && index.attributes.has(token.text)) type = "tapeAttribute";
|
||||
if (!type) type = registerType(token.text, index);
|
||||
if (!type && DELAY_SLOT_KEYWORDS.has(token.text)) type = "tapeDelaySlot";
|
||||
if (!type) {
|
||||
if (! type && index.bindTypes.has(token.text)) type = "tapeBindType";
|
||||
if (! type && DSL_KEYWORDS.has(token.text)) type = "keyword";
|
||||
if (! type && index.types.has(token.text)) type = "tapeDuffleType";
|
||||
if (! type && index.attributes.has(token.text)) type = "tapeAttribute";
|
||||
if (! type) type = registerType(token.text, index);
|
||||
if (! type && DELAY_SLOT_KEYWORDS.has(token.text)) type = "tapeDelaySlot";
|
||||
if (! type) {
|
||||
const domain = index.macros.get(token.text);
|
||||
if (domain === "control" || (domain && CONTROL_FLOW_PREFIXES.test(token.text))) {
|
||||
type = "tapeControlFlow";
|
||||
}
|
||||
}
|
||||
if (!type && isRegUseAccess(scanned.tokens, tokenIndex)) type = "tapeGprRegister";
|
||||
if (!type) type = instructionType(token.text, index);
|
||||
if (!type && /^(?:Slice_|A[0-9]+_)/.test(token.text)) type = "tapeDuffleType";
|
||||
if (!type && /_[RV]$/.test(token.text)) type = "tapeDuffleType";
|
||||
if (!type && index.atoms.has(token.text)) type = "tapeAtomName";
|
||||
if (!type && index.components.has(token.text)) type = "tapeComponentName";
|
||||
if (!type && index.phases.has(token.text)) type = "tapePhase";
|
||||
if (!type && index.labels.has(token.text)) type = "tapeLabel";
|
||||
if (!type) continue;
|
||||
if (! type && isRegUseAccess(scanned.tokens, tokenIndex)) type = "tapeGprRegister";
|
||||
if (! type) type = instructionType(token.text, index);
|
||||
if (! type && /^(?:Slice_|A[0-9]+_)/.test(token.text)) type = "tapeDuffleType";
|
||||
if (! type && /_[RV]$/.test(token.text)) type = "tapeDuffleType";
|
||||
if (! type && index.atoms.has(token.text)) type = "tapeAtomName";
|
||||
if (! type && index.components.has(token.text)) type = "tapeComponentName";
|
||||
if (! type && index.phases.has(token.text)) type = "tapePhase";
|
||||
if (! type && index.labels.has(token.text)) type = "tapeLabel";
|
||||
if (! type) continue;
|
||||
|
||||
spans.push({
|
||||
text: token.text,
|
||||
|
||||
+22
-18
@@ -2,10 +2,10 @@
|
||||
|
||||
const vscode = require("vscode");
|
||||
const { TOKEN_MODIFIERS, TOKEN_TYPES, classifyDocument } = require("./classifier");
|
||||
const { createIndex, mergeIndexes, scanSource } = require("./source-index");
|
||||
const { createIndex, mergeIndexes, scanSource } = require("./source-index");
|
||||
|
||||
const SOURCE_GLOB = "**/*.{c,h,cc,cpp,cxx,hh,hpp,hxx}";
|
||||
const EXCLUDE_GLOB = "**/{gen,build,.slop_cache,toolchain,node_modules}/**";
|
||||
const SOURCE_GLOB = "**/*.{c,h,cc,cpp,cxx,hh,hpp,hxx}";
|
||||
const EXCLUDE_GLOB = "**/{gen,build,.slop_cache,toolchain,node_modules}/**";
|
||||
const EXCLUDED_SEGMENTS = new Set(["gen", "build", ".slop_cache", "toolchain", "node_modules"]);
|
||||
|
||||
function isExcluded(uri) {
|
||||
@@ -18,28 +18,29 @@ function formatError(filePath, error) {
|
||||
}
|
||||
|
||||
async function activate(context) {
|
||||
const output = vscode.window.createOutputChannel("Tape Atom DSL");
|
||||
const output = vscode.window.createOutputChannel("Tape Atom DSL");
|
||||
const emitter = new vscode.EventEmitter();
|
||||
const legend = new vscode.SemanticTokensLegend(TOKEN_TYPES, TOKEN_MODIFIERS);
|
||||
let workspaceIndex = createIndex();
|
||||
const legend = new vscode.SemanticTokensLegend(TOKEN_TYPES, TOKEN_MODIFIERS);
|
||||
let workspaceIndex = createIndex();
|
||||
let rebuildGeneration = 0;
|
||||
let debounceHandle = null;
|
||||
let debounceHandle = null;
|
||||
|
||||
async function rebuildIndex() {
|
||||
const generation = ++rebuildGeneration;
|
||||
const files = await vscode.workspace.findFiles(SOURCE_GLOB, EXCLUDE_GLOB);
|
||||
let nextIndex = createIndex();
|
||||
const files = await vscode.workspace.findFiles(SOURCE_GLOB, EXCLUDE_GLOB);
|
||||
let nextIndex = createIndex();
|
||||
|
||||
for (const uri of files) {
|
||||
if (generation !== rebuildGeneration) return;
|
||||
if (isExcluded(uri)) continue;
|
||||
try {
|
||||
const bytes = await vscode.workspace.fs.readFile(uri);
|
||||
const bytes = await vscode.workspace.fs.readFile(uri);
|
||||
const source = Buffer.from(bytes).toString("utf8");
|
||||
const result = scanSource(source, uri.fsPath);
|
||||
nextIndex = mergeIndexes(nextIndex, result.index);
|
||||
nextIndex = mergeIndexes(nextIndex, result.index);
|
||||
for (const error of result.errors) output.appendLine(formatError(uri.fsPath, error));
|
||||
} catch (error) {
|
||||
}
|
||||
catch (error) {
|
||||
output.appendLine(`${uri.fsPath}: ${error.stack || error.message || error}`);
|
||||
}
|
||||
}
|
||||
@@ -53,9 +54,11 @@ async function activate(context) {
|
||||
if (uri && isExcluded(uri)) return;
|
||||
if (debounceHandle !== null) clearTimeout(debounceHandle);
|
||||
debounceHandle = setTimeout(() => {
|
||||
debounceHandle = null;
|
||||
rebuildIndex().catch((error) => output.appendLine(error.stack || String(error)));
|
||||
}, 100);
|
||||
debounceHandle = null;
|
||||
rebuildIndex().catch((error) => output.appendLine(error.stack || String(error)));
|
||||
},
|
||||
100
|
||||
);
|
||||
}
|
||||
|
||||
const provider = {
|
||||
@@ -77,7 +80,8 @@ async function activate(context) {
|
||||
output.appendLine(formatError(document.uri.fsPath || document.uri.toString(), error));
|
||||
}
|
||||
return builder.build();
|
||||
} catch (error) {
|
||||
}
|
||||
catch (error) {
|
||||
output.appendLine(`${document.uri}: ${error.stack || error.message || error}`);
|
||||
return new vscode.SemanticTokensBuilder(legend).build();
|
||||
}
|
||||
@@ -85,8 +89,8 @@ async function activate(context) {
|
||||
};
|
||||
|
||||
const selector = [
|
||||
{ language: "c", scheme: "file" },
|
||||
{ language: "c", scheme: "untitled" },
|
||||
{ language: "c", scheme: "file" },
|
||||
{ language: "c", scheme: "untitled" },
|
||||
{ language: "cpp", scheme: "file" },
|
||||
{ language: "cpp", scheme: "untitled" },
|
||||
];
|
||||
|
||||
Vendored
+38
-31
@@ -10,29 +10,30 @@ function isIdentifierContinue(code) {
|
||||
return isIdentifierStart(code) || (code >= 48 && code <= 57);
|
||||
}
|
||||
|
||||
function lex(source) {
|
||||
function lex(source)
|
||||
{
|
||||
if (typeof source !== "string") throw new TypeError("source must be a string");
|
||||
|
||||
const tokens = [];
|
||||
const errors = [];
|
||||
let offset = 0;
|
||||
let line = 0;
|
||||
let character = 0;
|
||||
const tokens = [];
|
||||
const errors = [];
|
||||
let offset = 0;
|
||||
let line = 0;
|
||||
let character = 0;
|
||||
|
||||
function advance() {
|
||||
if (source[offset] === "\r" && source[offset + 1] === "\n") {
|
||||
offset += 2;
|
||||
line += 1;
|
||||
offset += 2;
|
||||
line += 1;
|
||||
character = 0;
|
||||
return;
|
||||
}
|
||||
if (source[offset] === "\n") {
|
||||
offset += 1;
|
||||
line += 1;
|
||||
offset += 1;
|
||||
line += 1;
|
||||
character = 0;
|
||||
return;
|
||||
}
|
||||
offset += 1;
|
||||
offset += 1;
|
||||
character += 1;
|
||||
}
|
||||
|
||||
@@ -47,7 +48,8 @@ function lex(source) {
|
||||
});
|
||||
}
|
||||
|
||||
while (offset < source.length) {
|
||||
while (offset < source.length)
|
||||
{
|
||||
const ch = source[offset];
|
||||
|
||||
if (/\s/.test(ch)) {
|
||||
@@ -60,12 +62,14 @@ function lex(source) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (ch === "/" && source[offset + 1] === "*") {
|
||||
if (ch === "/" && source[offset + 1] === "*")
|
||||
{
|
||||
const start = offset;
|
||||
advance();
|
||||
advance();
|
||||
let closed = false;
|
||||
while (offset < source.length) {
|
||||
while (offset < source.length)
|
||||
{
|
||||
if (source[offset] === "*" && source[offset + 1] === "/") {
|
||||
advance();
|
||||
advance();
|
||||
@@ -78,12 +82,14 @@ function lex(source) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (ch === "\"" || ch === "'") {
|
||||
if (ch === "\"" || ch === "'")
|
||||
{
|
||||
const quote = ch;
|
||||
const start = offset;
|
||||
advance();
|
||||
let closed = false;
|
||||
while (offset < source.length) {
|
||||
while (offset < source.length)
|
||||
{
|
||||
if (source[offset] === "\\") {
|
||||
advance();
|
||||
if (offset < source.length) advance();
|
||||
@@ -97,14 +103,14 @@ function lex(source) {
|
||||
if (source[offset] === "\n" || source[offset] === "\r") break;
|
||||
advance();
|
||||
}
|
||||
if (!closed) errors.push({ kind: "unterminated-literal", offset: start });
|
||||
if (! closed) errors.push({ kind: "unterminated-literal", offset: start });
|
||||
continue;
|
||||
}
|
||||
|
||||
const code = source.charCodeAt(offset);
|
||||
if (isIdentifierStart(code)) {
|
||||
const start = offset;
|
||||
const startLine = line;
|
||||
const start = offset;
|
||||
const startLine = line;
|
||||
const startCharacter = character;
|
||||
advance();
|
||||
while (offset < source.length && isIdentifierContinue(source.charCodeAt(offset))) advance();
|
||||
@@ -112,8 +118,8 @@ function lex(source) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const start = offset;
|
||||
const startLine = line;
|
||||
const start = offset;
|
||||
const startLine = line;
|
||||
const startCharacter = character;
|
||||
advance();
|
||||
pushToken("punctuation", start, startLine, startCharacter);
|
||||
@@ -124,9 +130,9 @@ function lex(source) {
|
||||
|
||||
function buildCallContexts(tokens) {
|
||||
const contexts = Array.from({ length: tokens.length }, () => []);
|
||||
const calls = [];
|
||||
const calls = [];
|
||||
const errors = [];
|
||||
const stack = [];
|
||||
const stack = [];
|
||||
|
||||
for (let tokenIndex = 0; tokenIndex < tokens.length; tokenIndex += 1) {
|
||||
const token = tokens[tokenIndex];
|
||||
@@ -134,7 +140,8 @@ function buildCallContexts(tokens) {
|
||||
if (token.text === ")") {
|
||||
if (stack.length === 0) {
|
||||
errors.push({ kind: "unmatched-close-paren", offset: token.start });
|
||||
} else {
|
||||
}
|
||||
else {
|
||||
const frame = stack.pop();
|
||||
if (frame.callee !== null) calls.push({ ...frame, closeTokenIndex: tokenIndex });
|
||||
}
|
||||
@@ -143,20 +150,20 @@ function buildCallContexts(tokens) {
|
||||
contexts[tokenIndex] = stack
|
||||
.filter((frame) => frame.callee !== null)
|
||||
.map((frame) => ({
|
||||
callee: frame.callee,
|
||||
callee: frame.callee,
|
||||
calleeTokenIndex: frame.calleeTokenIndex,
|
||||
openTokenIndex: frame.openTokenIndex,
|
||||
argIndex: frame.argIndex,
|
||||
openTokenIndex: frame.openTokenIndex,
|
||||
argIndex: frame.argIndex,
|
||||
}));
|
||||
|
||||
if (token.text === "(") {
|
||||
const previous = tokens[tokenIndex - 1];
|
||||
const previous = tokens[tokenIndex - 1];
|
||||
const hasCallee = previous && previous.kind === "identifier";
|
||||
stack.push({
|
||||
callee: hasCallee ? previous.text : null,
|
||||
callee: hasCallee ? previous.text : null,
|
||||
calleeTokenIndex: hasCallee ? tokenIndex - 1 : -1,
|
||||
openTokenIndex: tokenIndex,
|
||||
argIndex: 0,
|
||||
openTokenIndex: tokenIndex,
|
||||
argIndex: 0,
|
||||
});
|
||||
continue;
|
||||
}
|
||||
|
||||
+47
-38
@@ -4,8 +4,10 @@ const path = require("node:path");
|
||||
const { buildCallContexts, lex, nearestCall } = require("./lexer");
|
||||
|
||||
const BASE_TYPES = [
|
||||
"B1", "B2", "B4", "B8", "F4", "F8", "S1", "S2", "S4", "S8",
|
||||
"U1", "U2", "U4", "U8", "MipsAtom", "MipsCode", "Reg",
|
||||
"B1", "B2", "B4", "B8",
|
||||
"F4", "F8", "S1", "S2", "S4", "S8",
|
||||
"U1", "U2", "U4", "U8",
|
||||
"MipsAtom", "MipsCode", "Reg",
|
||||
];
|
||||
|
||||
const C_BUILTINS = new Set([
|
||||
@@ -15,11 +17,13 @@ const C_BUILTINS = new Set([
|
||||
]);
|
||||
|
||||
const BASE_ATTRIBUTES = [
|
||||
"FI_", "I_", "NI_", "Relative_", "Struct_", "Enum_", "Union_", "Array_",
|
||||
"Slice_", "TypeR_", "TypeV_", "align_", "internal", "local_persist", "global",
|
||||
"RO_", "LP_", "gknown", "expect_", "cexpr_",
|
||||
"FI_", "I_", "NI_",
|
||||
"Relative_", "Struct_", "Enum_", "Union_", "Array_", "Slice_",
|
||||
"align_", "internal", "local_persist", "global",
|
||||
"RO_", "LP_",
|
||||
"gknown", "expect_", "cexpr_",
|
||||
"asm", "asm_words", "asm_rpins", "asm_clobber",
|
||||
"O_", "S_", "C_", "T_", "tmpl", "glue", "r_", "v_", "tr_", "tv_",
|
||||
"O_", "OA_", "S_", "C_", "T_", "tmpl", "glue", "r_", "v_", "rt_", "vt_",
|
||||
"rgcc", "r_use", "r_set", "r_mod", "r_imm", "r_mem",
|
||||
"u1_", "u2_", "u4_", "u8_", "s1_", "s2_", "s4_", "s8_",
|
||||
"u1_r", "u2_r", "u4_r", "u8_r", "u1_v", "u2_v", "u4_v", "u8_v",
|
||||
@@ -27,16 +31,16 @@ const BASE_ATTRIBUTES = [
|
||||
|
||||
function createIndex() {
|
||||
return {
|
||||
atoms: new Set(),
|
||||
components: new Set(),
|
||||
atoms: new Set(),
|
||||
components: new Set(),
|
||||
componentAliases: new Set(),
|
||||
macros: new Map(),
|
||||
registers: new Map(),
|
||||
bindTypes: new Set(),
|
||||
types: new Set(BASE_TYPES),
|
||||
phases: new Set(),
|
||||
labels: new Set(),
|
||||
attributes: new Set(BASE_ATTRIBUTES),
|
||||
macros: new Map(),
|
||||
registers: new Map(),
|
||||
bindTypes: new Set(),
|
||||
types: new Set(BASE_TYPES),
|
||||
phases: new Set(),
|
||||
labels: new Set(),
|
||||
attributes: new Set(BASE_ATTRIBUTES),
|
||||
componentCallees: new Map(),
|
||||
};
|
||||
}
|
||||
@@ -46,8 +50,8 @@ function cloneIndex(source) {
|
||||
for (const key of ["atoms", "components", "componentAliases", "bindTypes", "types", "phases", "labels", "attributes"]) {
|
||||
for (const value of source[key]) result[key].add(value);
|
||||
}
|
||||
for (const [name, domain] of source.macros) result.macros.set(name, domain);
|
||||
for (const [name, domain] of source.registers) result.registers.set(name, domain);
|
||||
for (const [name, domain] of source.macros) result.macros.set(name, domain);
|
||||
for (const [name, domain] of source.registers) result.registers.set(name, domain);
|
||||
for (const [name, callees] of source.componentCallees) result.componentCallees.set(name, callees.slice());
|
||||
return result;
|
||||
}
|
||||
@@ -63,7 +67,7 @@ function mergeIndexes(...sources) {
|
||||
const existing = result.macros.get(name);
|
||||
if (!existing || domainRank(domain) >= domainRank(existing)) result.macros.set(name, domain);
|
||||
}
|
||||
for (const [name, domain] of source.registers) result.registers.set(name, domain);
|
||||
for (const [name, domain] of source.registers) result.registers.set(name, domain);
|
||||
for (const [name, callees] of source.componentCallees) {
|
||||
const existing = result.componentCallees.get(name) || [];
|
||||
result.componentCallees.set(name, existing.concat(callees));
|
||||
@@ -75,21 +79,21 @@ function mergeIndexes(...sources) {
|
||||
function domainFromPath(filePath) {
|
||||
const base = path.basename(filePath.replaceAll("\\", "/")).toLowerCase();
|
||||
if (base === "mips.h") return "cpu";
|
||||
if (base === "gte.h") return "gte";
|
||||
if (base === "gp.h") return "gpu";
|
||||
if (base === "gte.h") return "gte";
|
||||
if (base === "gp.h") return "gpu";
|
||||
return null;
|
||||
}
|
||||
|
||||
function prefixDomain(name) {
|
||||
if (/^(?:branch_|jump_|call_)/.test(name)) return "control";
|
||||
if (/^gte_(?!cr_)/.test(name) || name.startsWith("mac_gte_") || name.startsWith("ac_gte_")) return "gte";
|
||||
if (/^gp[01]_/.test(name) || name.startsWith("mac_gp_") || name.startsWith("ac_gp_")) return "gpu";
|
||||
if (/^gp[01]_/.test(name) || name.startsWith("mac_gp_") || name.startsWith("ac_gp_")) return "gpu";
|
||||
return null;
|
||||
}
|
||||
|
||||
function collectBraceIdentifiers(tokens, openBraceIndex) {
|
||||
const names = [];
|
||||
let depth = 0;
|
||||
let depth = 0;
|
||||
for (let tokenIndex = openBraceIndex; tokenIndex < tokens.length; tokenIndex += 1) {
|
||||
if (tokens[tokenIndex].text === "{") depth += 1;
|
||||
if (tokens[tokenIndex].text === "}") {
|
||||
@@ -103,17 +107,17 @@ function collectBraceIdentifiers(tokens, openBraceIndex) {
|
||||
|
||||
function resolveComponentDomains(index) {
|
||||
const hardwareRank = { cpu: 1, gpu: 2, gte: 3, control: 4 };
|
||||
let changed = true;
|
||||
let changed = true;
|
||||
while (changed) {
|
||||
changed = false;
|
||||
for (const [alias, callees] of index.componentCallees) {
|
||||
let best = index.macros.get(alias) || "component";
|
||||
let best = index.macros.get(alias) || "component";
|
||||
let bestRank = hardwareRank[best] || 0;
|
||||
for (const callee of callees) {
|
||||
const domain = prefixDomain(callee) || index.macros.get(callee);
|
||||
const rank = hardwareRank[domain] || 0;
|
||||
const rank = hardwareRank[domain] || 0;
|
||||
if (rank > bestRank) {
|
||||
best = domain;
|
||||
best = domain;
|
||||
bestRank = rank;
|
||||
}
|
||||
}
|
||||
@@ -134,7 +138,7 @@ function domainRank(domain) {
|
||||
}
|
||||
|
||||
function registerKind(name) {
|
||||
if (/^R_[A-Za-z0-9_]+$/.test(name)) return "gpr";
|
||||
if (/^R_[A-Za-z0-9_]+$/.test(name)) return "gpr";
|
||||
if (/^(?:C2_|gte_cr_)[A-Za-z0-9_]+$/.test(name)) return "cop2";
|
||||
return null;
|
||||
}
|
||||
@@ -161,14 +165,15 @@ function findFunctionNameBefore(tokens, calleeTokenIndex) {
|
||||
return null;
|
||||
}
|
||||
|
||||
function scanSource(source, filePath) {
|
||||
const lexical = lex(source);
|
||||
const balanced = buildCallContexts(lexical.tokens);
|
||||
const tokens = lexical.tokens;
|
||||
const contexts = balanced.contexts;
|
||||
const index = createIndex();
|
||||
function scanSource(source, filePath)
|
||||
{
|
||||
const lexical = lex(source);
|
||||
const balanced = buildCallContexts(lexical.tokens);
|
||||
const tokens = lexical.tokens;
|
||||
const contexts = balanced.contexts;
|
||||
const index = createIndex();
|
||||
const declarations = new Map();
|
||||
const domain = domainFromPath(filePath);
|
||||
const domain = domainFromPath(filePath);
|
||||
|
||||
function mark(token, role, modifiers = ["declaration"]) {
|
||||
declarations.set(token.start, { role, modifiers });
|
||||
@@ -191,7 +196,8 @@ function scanSource(source, filePath) {
|
||||
if (!index.macros.has(alias)) index.macros.set(alias, "component");
|
||||
}
|
||||
|
||||
for (let tokenIndex = 0; tokenIndex < tokens.length; tokenIndex += 1) {
|
||||
for (let tokenIndex = 0; tokenIndex < tokens.length; tokenIndex += 1)
|
||||
{
|
||||
const token = tokens[tokenIndex];
|
||||
if (token.kind !== "identifier") continue;
|
||||
|
||||
@@ -240,12 +246,14 @@ function scanSource(source, filePath) {
|
||||
mark(token, "gprRegister", ["declaration", "tapeAuto"]);
|
||||
}
|
||||
|
||||
if (token.text === "define" && tokens[tokenIndex - 1] && tokens[tokenIndex - 1].text === "#") {
|
||||
if (token.text === "define" && tokens[tokenIndex - 1] && tokens[tokenIndex - 1].text === "#")
|
||||
{
|
||||
const name = tokens[tokenIndex + 1];
|
||||
if (name && name.kind === "identifier" && name.line === token.line) {
|
||||
if (/^(?:RegUse_|Struct_|Enum_|Union_|TypeR_|TypeV_|Relative_|Binds_)/.test(name.text)) {
|
||||
index.types.add(name.text);
|
||||
} else if (/^(?:ac_|mac_)/.test(name.text)) {
|
||||
}
|
||||
else if (/^(?:ac_|mac_)/.test(name.text)) {
|
||||
const alias = name.text.startsWith("ac_") ? componentAlias(name.text) : name.text;
|
||||
const rest = [];
|
||||
for (let restIndex = tokenIndex + 2; restIndex < tokens.length && tokens[restIndex].line === name.line; restIndex += 1) {
|
||||
@@ -256,7 +264,8 @@ function scanSource(source, filePath) {
|
||||
index.macros.set(alias, prefixDomain(alias) || "component");
|
||||
if (rest.length) index.componentCallees.set(alias, rest);
|
||||
}
|
||||
} else {
|
||||
}
|
||||
else {
|
||||
index.macros.set(name.text, domain || "utility");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -44,7 +44,7 @@
|
||||
*
|
||||
* Annotation rules
|
||||
* ----------------
|
||||
* 1. atom_info(...) is OPTIONAL. Atoms without atom_info are silently skipped by the metaprogram.
|
||||
* 1. atom_info(...) is optional. Atoms without atom_info are silently skipped by the metaprogram.
|
||||
* 2. If present, atom_info takes up to three sub-calls, all order-independent within the arg list:
|
||||
* - atom_bind(Binds_X)
|
||||
* - atom_reads(...)
|
||||
@@ -160,6 +160,8 @@
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_bind(binds_struct) /* atom_bind(binds_struct) */
|
||||
|
||||
#define Binds_(type) (tmpl(Binds,type)) // TODO(Ed): Do we want to use this?
|
||||
|
||||
/* ============================================================================
|
||||
* atom_label / atom_offset — branch target machinery
|
||||
*
|
||||
@@ -177,9 +179,3 @@
|
||||
#define atom_offset(F, T) atom_offset_ ## F ## _ ## T
|
||||
// atom_label is a pure annotation for the metaprogram's offset calculations.
|
||||
#define atom_label(name) /* atom_label anchor: name */
|
||||
|
||||
// WIP: Atoms Assocated closely with each other to form a tape procedure. (Maybe also a phase in a procedure/pipeline?)
|
||||
|
||||
#define AtomBundle_(name) Struct_(tmpl(AtomBundle,name))
|
||||
#define AtomBundle_Len(name) S_(tmpl(AtomBundle,name))/S_(MipsAtom*)
|
||||
#define AtomBundleEntry_(bundle,entry) tmpl(bundle,entry)
|
||||
|
||||
+23
-12
@@ -14,8 +14,8 @@
|
||||
#define glue(A, B) glue_impl(A, B)
|
||||
#define tmpl(prefix, type) prefix ## _ ## type
|
||||
|
||||
#define stringify_impl(S) #S
|
||||
#define stringify(S) stringify_impl(S)
|
||||
#define stringify_impl(S) #S
|
||||
#define stringify(S) stringify_impl(S)
|
||||
|
||||
#define VA_Sel_1( _1, ... ) _1 // <-- Of all th args passed pick _1.
|
||||
#define VA_Sel_2( _1, _2, ... ) _2 // <-- Of all the args passed pick _2.
|
||||
@@ -29,8 +29,7 @@
|
||||
|
||||
#define asm __asm__
|
||||
|
||||
#define A_(data) (& data)
|
||||
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
||||
#define A_(data) (& (data))
|
||||
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
||||
#define C_(type,data) ((type)(data)) // for enforced precedence
|
||||
#define expect_(x, y) __builtin_expect(x, y) // so compiler knows the common path
|
||||
@@ -45,7 +44,8 @@
|
||||
#define R_ restrict
|
||||
#define V_ volatile
|
||||
|
||||
#pragma region Fictional //, used for intiution
|
||||
#pragma region Fictional
|
||||
//, used for intiution
|
||||
|
||||
#define EUB_ restrict // Execute Unit Bound: Data is siloed in the ALU Register File. The Load/Store Unit is bypassed. (Route to Execution Unit. Keep in registers)
|
||||
#define ISO_ restrict // Isolated Provenance: Alternative to Exu_. Guarantees electrical memory isolation,
|
||||
@@ -83,8 +83,8 @@
|
||||
|
||||
#define r_(ptr) C_(T_(ptr[0])*R_, ptr) // Constrain pointer to restrict
|
||||
#define v_(ptr) C_(T_(ptr[0])V_*, ptr) //
|
||||
#define tr_(type, ptr) C_(type *R_, ptr)
|
||||
#define tv_(type, ptr) C_(type V_*, ptr)
|
||||
#define rt_(type, ptr) C_(type *R_, ptr)
|
||||
#define vt_(type, ptr) C_(type V_*, ptr)
|
||||
|
||||
#define TypeR_(type) type *R_ type ## _R // type *restrict type_R
|
||||
#define TypeV_(type) type V_* type ## _V // type volatile* type_V
|
||||
@@ -120,6 +120,10 @@ typedef unsigned char TSet_(B1);
|
||||
typedef __UINT16_TYPE__ TSet_(B2);
|
||||
typedef __UINT32_TYPE__ TSet_(B4);
|
||||
|
||||
#define b1_(value) C_(B1, value)
|
||||
#define b2_(value) C_(B2, value)
|
||||
#define b4_(value) C_(B4, value)
|
||||
|
||||
#define u1_(value) C_(U1, value)
|
||||
#define u2_(value) C_(U2, value)
|
||||
#define u4_(value) C_(U4, value)
|
||||
@@ -127,6 +131,13 @@ typedef __UINT32_TYPE__ TSet_(B4);
|
||||
#define s2_(value) C_(S2, value)
|
||||
#define s4_(value) C_(S4, value)
|
||||
|
||||
#define b1_r(value) C_(B1*R_, value)
|
||||
#define b2_r(value) C_(B2*R_, value)
|
||||
#define b4_r(value) C_(B4*R_, value)
|
||||
#define b1_v(value) C_(B1 V_*, value)
|
||||
#define b2_v(value) C_(B2 V_*, value)
|
||||
#define b4_v(value) C_(B4 V_*, value)
|
||||
|
||||
#define u1_r(value) C_(U1 *R_, value)
|
||||
#define u2_r(value) C_(U2 *R_, value)
|
||||
#define u4_r(value) C_(U4 *R_, value)
|
||||
@@ -183,12 +194,12 @@ def_signed_ops(le, <=)
|
||||
#undef def_generic_sop
|
||||
#endif
|
||||
|
||||
#define alignas _Alignas
|
||||
#define alignof _Alignof
|
||||
#define byte_pad(amount, ...) B1 glue(_PAD_, __VA_ARGS__) [amount]
|
||||
#define C_ptr(type, data) (C_(type*, & (data)) [0])
|
||||
#define alignas _Alignas
|
||||
#define alignof _Alignof
|
||||
#define byte_pad(amount, ...) B1 glue(_PAD_, __VA_ARGS__) [amount]
|
||||
#define C_ptr(type, data) (C_(type*, & (data)) [0])
|
||||
|
||||
#define dbg_args(...) __VA_ARGS__
|
||||
#define dbg_args(...) __VA_ARGS__
|
||||
|
||||
#pragma region Control Flow & Iteration
|
||||
#define each_iter(type, iter, end) (type iter = 0; iter < end; ++ iter)
|
||||
|
||||
@@ -79,14 +79,6 @@
|
||||
* Why bundle the `__asm__()` wrapper?
|
||||
* - The integer R_T4 (= 12, via R_T4_Code) already indicates the register.
|
||||
* - The string "$12" is derived from it via reg_str, so they cannot drift apart.
|
||||
* - Spelling `__asm__(reg_str(R_T4_Code))` at every call site is noise.
|
||||
*
|
||||
* tmpl defined in dsl.h (token-paste glue).
|
||||
* rgcc define here (gcc_asm.h) because the `__asm__` keyword is GCC-specific.
|
||||
* Anyone porting to a different compiler's asm dialect overrides rgcc,
|
||||
* and the integer→string derivation in rlit can be retargeted in one place.
|
||||
*
|
||||
* For clobber lists and asm-template strings, use the bare `rlit(R_T4_Code)`.
|
||||
* ------------------------------------------------------------------------ */
|
||||
#define rgcc(n) __asm__(rlit(n))
|
||||
|
||||
|
||||
+25
-18
@@ -13,7 +13,7 @@
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\gte.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\tape.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
|
||||
@@ -35,6 +35,7 @@
|
||||
* These do NOT yield. They are expanded inline inside Tape Atoms.
|
||||
* ---------------------------------------------------------------------------*/
|
||||
// The 'Yield' sequence for Tape Atoms (mac_yield).
|
||||
// In Forth this is considered the "NEXT" mechanism.
|
||||
#define mac_yield(...) \
|
||||
load_word(R_AtomJmp, R_TapePtr, 0) \
|
||||
LdSlot_ \
|
||||
@@ -55,6 +56,12 @@ WORD_COUNT(mac_yield_load, 1)
|
||||
, BdSlot_ nop
|
||||
WORD_COUNT(mac_yield_tail, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_yield_to(code_ptr) \
|
||||
jump_reg(code_ptr) \
|
||||
, BdSlot_ nop
|
||||
WORD_COUNT(mac_yield_to, 2)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_load_half_v3(tx, ty, tz, base, offset) \
|
||||
load_half(tx, base, offset + OA_(U2,[0])) \
|
||||
@@ -228,21 +235,21 @@ WORD_COUNT(mac_gte_op_cross_v3s4, 13)
|
||||
WORD_COUNT(mac_gte_store_f3, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_gte_load_tri_verts(r_vert_base, r_v0, r_v1, r_v2) \
|
||||
shift_lleft(R_AT, r_v0, v3s2_byteoff) \
|
||||
, add_u_self(R_AT, r_vert_base) \
|
||||
#define mac_gte_load_tri_verts(vbase, v0, v1, v2) \
|
||||
shift_lleft(R_AT, v0, v3s2_byteoff) \
|
||||
, add_u_self(R_AT, vbase) \
|
||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||
, LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY0) \
|
||||
, gte_mv_to_data_r(R_V1, C2_VZ0) \
|
||||
, shift_lleft(R_AT, r_v1, v3s2_byteoff) \
|
||||
, add_u_self(R_AT, r_vert_base) \
|
||||
, shift_lleft(R_AT, v1, v3s2_byteoff) \
|
||||
, add_u_self(R_AT, vbase) \
|
||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||
, LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY1) \
|
||||
, gte_mv_to_data_r(R_V1, C2_VZ1) \
|
||||
, shift_lleft(R_AT, r_v2, v3s2_byteoff) \
|
||||
, add_u_self(R_AT, r_vert_base) \
|
||||
, shift_lleft(R_AT, v2, v3s2_byteoff) \
|
||||
, add_u_self(R_AT, vbase) \
|
||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||
, LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY2) \
|
||||
@@ -270,10 +277,10 @@ WORD_COUNT(mac_gte_store_g4_p3, 1)
|
||||
WORD_COUNT(mac_gte_sqr_v3, 8)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, delay_slot) \
|
||||
gte_mv_to_data_r(r_sx, C2_IR1) \
|
||||
, gte_mv_to_data_r(r_sy, C2_IR2) \
|
||||
, gte_mv_to_data_r(r_sz, C2_IR3) \
|
||||
#define mac_gte_sqr_v3s4(sx, sy, sz, delay_slot) \
|
||||
gte_mv_to_data_r(sx, C2_IR1) \
|
||||
, gte_mv_to_data_r(sy, C2_IR2) \
|
||||
, gte_mv_to_data_r(sz, C2_IR3) \
|
||||
, delay_slot \
|
||||
, gte_cmdw_sqr
|
||||
WORD_COUNT(mac_gte_sqr_v3s4, 5)
|
||||
@@ -304,12 +311,12 @@ WORD_COUNT(mac_gte_gpf_scale, 13)
|
||||
WORD_COUNT(mac_trans_mt3s3s4, 6)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_lzcr_round_even_half_shift(r_shift, r_mag_sq, r_mag_sq_copy) \
|
||||
and_i(r_shift, r_shift, gte_lzcr_even_mask) \
|
||||
, or_u(r_mag_sq_copy, r_mag_sq, 0) \
|
||||
, li_s( r_mag_sq, 31) \
|
||||
, sub_s( r_mag_sq, r_mag_sq, r_shift) \
|
||||
, shift_aright(r_mag_sq, r_mag_sq, 1)
|
||||
#define mac_lzcr_round_even_half_shift(shift, mag_sq, mag_sq_copy) \
|
||||
and_i(shift, shift, gte_lzcr_even_mask) \
|
||||
, or_u(mag_sq_copy, mag_sq, 0) \
|
||||
, li_s( mag_sq, 31) \
|
||||
, sub_s( mag_sq, mag_sq, shift) \
|
||||
, shift_aright(mag_sq, mag_sq, 1)
|
||||
WORD_COUNT(mac_lzcr_round_even_half_shift, 5)
|
||||
|
||||
#define mac_gte_general_purpose_interopolation(to_ir0, to_ir1, to_ir2, to_ir3, fr_mac1, fr_mac2, fr_mac3, nop_slot1, nop_slot2) \
|
||||
|
||||
@@ -10,7 +10,7 @@
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\gte.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\tape.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
|
||||
@@ -33,7 +33,7 @@ enum {
|
||||
atom_offset_example_atom_proc_skip = _atom_offset_example_atom_proc_skip,
|
||||
};
|
||||
|
||||
// --- atom: build_normalize_v3s4 (63 words) ---
|
||||
// --- atom: normalize_v3s4 (62 words) ---
|
||||
|
||||
#define _atom_offset_aligned_done_srav_path 3
|
||||
#define _atom_offset_srav_path_aligned_done 4
|
||||
|
||||
@@ -1,14 +1,14 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# include "dsl.h"
|
||||
# include "gp.h"
|
||||
# include "lottes_tape.h"
|
||||
# include "tape.h"
|
||||
#endif
|
||||
|
||||
ATOM_FILE_DEBUGGER_LINE_MARKER(gp_atom_c);
|
||||
|
||||
#pragma region MACs (Mips Atom Components)
|
||||
|
||||
FI_ Slice_MipsCode ac_gcmd_push(AtomBuilder_R ab, U4 cmd, U4 reg_transfer, U4 reg_base, U2 port)
|
||||
FI_ Slice_MipsCode ac_gcmd_push(AtomBuilder_R ab, U2 cmd, Reg reg_transfer, Reg reg_base, U2 port)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
mac_load_word_imm(reg_transfer, cmd),
|
||||
store_word( reg_transfer, reg_base, port),
|
||||
|
||||
+20
-35
@@ -6,16 +6,6 @@
|
||||
* Primitive commands: gp0_cmd_poly_f3 = 0x20 (byte opcode)
|
||||
* Packed 32-bit cmd: gp0_word_poly_f3(r, g, b) (32-bit, shifted)
|
||||
*
|
||||
* Type ordering: domain?_(direction)?_action_target_modifier_type?
|
||||
* Examples: add_ui (add + unsigned + immediate)
|
||||
* add_s (add + signed, R-type implicit)
|
||||
* shift_lleft (shift + logical + left)
|
||||
* shift_aright (shift + arithmetic + right)
|
||||
* call_reg(rs) (call + register, $ra implicit)
|
||||
* gte_mv_to_data_r (gte + mv + to + data + register)
|
||||
* gte_lw_v0_xy(base) (gte + lw + v0 + xy)
|
||||
* load_upper_i (load-upper + immediate, unique verb)
|
||||
*
|
||||
* --- GPU-domain layer cake ---
|
||||
* Every gp.h macro follows the same 4-layer composition as mips.h and gte.h:
|
||||
* 4. Semantic encoders gp0_word_poly_f3(r,g,b)
|
||||
@@ -23,9 +13,6 @@
|
||||
* 2. Per-field encoders enc_gp0_color_r(r), enc_gp0_color_g(g), ...
|
||||
* 1. Bitfield layout consts gp0_color_red_pos = 0, gp0_color_red_width = 8
|
||||
* 0. Opcode IDs gp0_cmd_poly_f3 = 0x20
|
||||
*
|
||||
* Vendor mnemonics (gte_mtc2, gte_mfc2, etc.) are NOT in this header.
|
||||
* They live in the opt-in `gp_vendor_sym.h` for users who prefer the PSYQ-style names.
|
||||
* ============================================================================ */
|
||||
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
@@ -391,17 +378,15 @@ enum {
|
||||
* Primitive structs (8 polygon variants + tag)
|
||||
* ============================================================================
|
||||
* Each struct follows the GPU-documented memory layout for the corresponding primitive command.
|
||||
* The PolyTag is the OT-link header; the rest of the struct is the primitive's body.
|
||||
* PolyTag is an OT-link header. Rest of the struct is the primitive's body.
|
||||
*
|
||||
* The current working layouts match the existing demo
|
||||
* (floor_tri uses Poly_F3; cube_tri uses Poly_G4).
|
||||
* They are NOT necessarily byte-identical to the PSX-SPX reference layout.
|
||||
* The demo layout uses color+vertex interleaving that doesn't match the standard PSX SDK file format.
|
||||
* For PSX-SDK file compatibility, the textured variants (FT*, GT*) would need layout adjustments.
|
||||
* ============================================================================ */
|
||||
|
||||
/* ---------- RGB8 (3-byte packed color) ---------- */
|
||||
typedef Struct_(RGB8) { B1 r; B1 g; B1 b; };
|
||||
typedef Struct_(RGB8) { U1 r; U1 g; U1 b; };
|
||||
#define rgb8(r,g,b) ((RGB8){r,g,b})
|
||||
|
||||
/* ---------- PolyTag (the OT-link header; 1 word) ---------- */
|
||||
@@ -431,7 +416,7 @@ typedef Struct_(PolyTag) {
|
||||
typedef Struct_(Poly_F3) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
U1 code;
|
||||
union {
|
||||
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; };
|
||||
A3_V2_S2 points;
|
||||
@@ -442,7 +427,7 @@ typedef Struct_(Poly_F3) {
|
||||
typedef Struct_(Poly_F4) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
U1 code;
|
||||
union {
|
||||
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; V2_S2 p3; };
|
||||
A4_V2_S2 points;
|
||||
@@ -451,18 +436,18 @@ typedef Struct_(Poly_F4) {
|
||||
|
||||
/* ---------- Poly_G3 (Gouraud Triangle; 7 words) ---------- */
|
||||
typedef Struct_(Poly_G3) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
U4 tag; RGB8 c0; U1 code;
|
||||
V2_S2 p0; RGB8 c1; U1 pad1;
|
||||
V2_S2 p1; RGB8 c2; U1 pad2;
|
||||
V2_S2 p2;
|
||||
};
|
||||
|
||||
/* ---------- Poly_G4 (Gouraud Quad; 9 words) ---------- */
|
||||
typedef Struct_(Poly_G4) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2; RGB8 c3; B1 pad3;
|
||||
U4 tag; RGB8 c0; U1 code;
|
||||
V2_S2 p0; RGB8 c1; U1 pad1;
|
||||
V2_S2 p1; RGB8 c2; U1 pad2;
|
||||
V2_S2 p2; RGB8 c3; U1 pad3;
|
||||
V2_S2 p3;
|
||||
};
|
||||
|
||||
@@ -471,7 +456,7 @@ typedef Struct_(Poly_G4) {
|
||||
typedef Struct_(Poly_FT3) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
U1 code;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 p0; U1 u0; U1 v0;
|
||||
@@ -483,7 +468,7 @@ typedef Struct_(Poly_FT3) {
|
||||
typedef Struct_(Poly_FT4) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
U1 code;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 p0; U1 u0; U1 v0;
|
||||
@@ -494,9 +479,9 @@ typedef Struct_(Poly_FT4) {
|
||||
|
||||
/* ---------- Poly_GT3 (Gouraud Textured Triangle) ---------- */
|
||||
typedef Struct_(Poly_GT3) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
U4 tag; RGB8 c0; U1 code;
|
||||
V2_S2 p0; RGB8 c1; U1 pad1;
|
||||
V2_S2 p1; RGB8 c2; U1 pad2;
|
||||
V2_S2 p2;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
@@ -507,10 +492,10 @@ typedef Struct_(Poly_GT3) {
|
||||
|
||||
/* ---------- Poly_GT4 (Gouraud Textured Quad) ---------- */
|
||||
typedef Struct_(Poly_GT4) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2; RGB8 c3; B1 pad3;
|
||||
U4 tag; RGB8 c0; U1 code;
|
||||
V2_S2 p0; RGB8 c1; U1 pad1;
|
||||
V2_S2 p1; RGB8 c2; U1 pad2;
|
||||
V2_S2 p2; RGB8 c3; U1 pad3;
|
||||
V2_S2 p3;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
|
||||
+122
-93
@@ -3,7 +3,8 @@
|
||||
# include "gen/offsets.h"
|
||||
# include "gte.h"
|
||||
# include "gp.h"
|
||||
# include "lottes_tape.h"
|
||||
# include "tape.h"
|
||||
# include "math.atom.h"
|
||||
#endif
|
||||
|
||||
ATOM_FILE_DEBUGGER_LINE_MARKER(gte_atom_c);
|
||||
@@ -63,10 +64,10 @@ FI_ Slice_MipsCode ac_gte_store_f3(AtomBuilder_R ab, U4 r_primitive_cursor) atom
|
||||
})
|
||||
|
||||
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
|
||||
I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
shift_lleft(R_AT, r_v0, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
shift_lleft(R_AT, r_v1, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
||||
shift_lleft(R_AT, r_v2, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
||||
I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, Reg vbase, Reg v0, Reg v1, Reg v2) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
shift_lleft(R_AT, v0, v3s2_byteoff), add_u_self(R_AT, vbase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
shift_lleft(R_AT, v1, v3s2_byteoff), add_u_self(R_AT, vbase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
||||
shift_lleft(R_AT, v2, v3s2_byteoff), add_u_self(R_AT, vbase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
||||
})
|
||||
|
||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
|
||||
@@ -96,11 +97,11 @@ FI_ Slice_MipsCode ac_gte_sqr_v3(AtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4
|
||||
})
|
||||
|
||||
/* ─── SQR FIRE — mtc2 3 GPRs into IR1/IR2/IR3, then fire SQR. ─── */
|
||||
FI_ Slice_MipsCode ac_gte_sqr_v3s4(AtomBuilder_R ab, Reg r_sx, Reg r_sy, Reg r_sz, MipsCode delay_slot)
|
||||
FI_ Slice_MipsCode ac_gte_sqr_v3s4(AtomBuilder_R ab, Reg sx, Reg sy, Reg sz, MipsCode delay_slot)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
gte_mv_to_data_r(r_sx, C2_IR1),
|
||||
gte_mv_to_data_r(r_sy, C2_IR2),
|
||||
gte_mv_to_data_r(r_sz, C2_IR3),
|
||||
gte_mv_to_data_r(sx, C2_IR1),
|
||||
gte_mv_to_data_r(sy, C2_IR2),
|
||||
gte_mv_to_data_r(sz, C2_IR3),
|
||||
delay_slot, gte_cmdw_sqr,
|
||||
})
|
||||
|
||||
@@ -157,16 +158,13 @@ FI_ Slice_MipsCode ac_trans_mt3s3s4(AtomBuilder_R ab
|
||||
*
|
||||
* Note: C2_LZCR (cop2r31) is a fixed read-only C2 data register — the caller must read it via mfc2 from C2_LZCR;
|
||||
* there is no register choice at the hardware level. Only the GPR that holds the result is caller-determined. */
|
||||
FI_ Slice_MipsCode ac_lzcr_round_even_half_shift(AtomBuilder_R ab,
|
||||
U4 r_shift,
|
||||
U4 r_mag_sq,
|
||||
U4 r_mag_sq_copy)
|
||||
FI_ Slice_MipsCode ac_lzcr_round_even_half_shift(AtomBuilder_R ab, Reg shift, Reg mag_sq, Reg mag_sq_copy)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
and_i(r_shift, r_shift, gte_lzcr_even_mask),
|
||||
or_u(r_mag_sq_copy, r_mag_sq, 0),
|
||||
li_s( r_mag_sq, 31),
|
||||
sub_s( r_mag_sq, r_mag_sq, r_shift),
|
||||
shift_aright(r_mag_sq, r_mag_sq, 1),
|
||||
and_i(shift, shift, gte_lzcr_even_mask),
|
||||
or_u(mag_sq_copy, mag_sq, 0),
|
||||
li_s( mag_sq, 31),
|
||||
sub_s( mag_sq, mag_sq, shift),
|
||||
shift_aright(mag_sq, mag_sq, 1),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_gte_general_purpose_interopolation(AtomBuilder_R ab
|
||||
@@ -186,60 +184,95 @@ MipsAtomComp_Proc_(ab, {
|
||||
gte_mv_from_data_r(fr_mac3, C2_MAC3),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_gte_mv_from_data_r_mac123(AtomBuilder_R ab
|
||||
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3)
|
||||
FI_ Slice_MipsCode ac_gte_mv_from_data_r_mac123(AtomBuilder_R ab, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3)
|
||||
MipsAtomComp_Proc_(ab, {
|
||||
gte_mv_from_data_r(fr_mac1, C2_MAC1),
|
||||
gte_mv_from_data_r(fr_mac2, C2_MAC2),
|
||||
gte_mv_from_data_r(fr_mac3, C2_MAC3),
|
||||
})
|
||||
|
||||
|
||||
FI_ Slice_MipsCode ac_gte_mv_from_mac123_v3s4(AtomBuilder_R ab, Reg_(V3_S4) v) MipsAtomComp_ProcMap_(ab, mac_gte_mv_from_data_r_mac123(v.x, v.y, v.z))
|
||||
|
||||
#pragma endregion MACs (Mips Atom Components)
|
||||
|
||||
#pragma region Atom Procs
|
||||
|
||||
/* ─── Local copy of PSYQ's sqrtbl (1/sqrt lookup table for VectorNormal). ───
|
||||
* Source: PSYQ 4.7 libgte sqrtbl at 0x800185B4 in hello_camera.elf.
|
||||
* objdump -s --start-address=0x800185B4 --stop-address=0x800185F4 hello_camera.elf → 192 entries × 16-bit signed, in 1.12 fixed-point (max value 0x1000 = 1.0).
|
||||
*
|
||||
* Data is identical to the libgte original (byte-for-byte verified).
|
||||
/* Normalize V3_S4 using the PSYQ/libgte reciprocal-sqrt method:
|
||||
* |v|² = x² + y² + z²
|
||||
* LZCR determines the exponent of |v|².
|
||||
* Round that exponent even and shift |v|² into [1, 4).
|
||||
* sqrtbl approximates 1/sqrt(mantissa).
|
||||
* GPF multiplies v by that reciprocal-sqrt mantissa.
|
||||
* srav_shift restores the exponent scale.
|
||||
* Effectively: v_normalized = v * (1 / sqrt(|v|²)).
|
||||
*
|
||||
* ─── Per-entry semantics (decoded from libgte msc02 VectorNormal) ───
|
||||
* Each entry is `1/sqrt(x)` in 1.12 fixed point (value / 4096).
|
||||
* The 192 entries span 4 octaves of the input magnitude, with 48 entries per octave:
|
||||
* Octave 0 (entries 0- 47): mantissa in [0x8000, 0x10000) output ~[1.000, 0.707]
|
||||
* Octave 1 (entries 48- 95): mantissa in [0x10000, 0x20000) output ~[0.707, 0.500]
|
||||
* Octave 2 (entries 96-143): mantissa in [0x20000, 0x40000) output ~[0.500, 0.354]
|
||||
* Octave 3 (entries144-191): mantissa in [0x40000, 0x80000) output ~[0.354, 0.251]
|
||||
* Within each octave, 8 sub-entries interpolate over the 8 fractional bits of the mantissa
|
||||
* (the byte `(0x80 | (i mod 8))` for the lower-byte of the aligned value).
|
||||
* Sampling the first value of each octave:
|
||||
* [0] 0x1000 = 1.0000 ; 1 / sqrt(1.0000)
|
||||
* [48] 0x0e4f = 0.8940 ; 1 / sqrt(1.2500)
|
||||
* [96] 0x0d10 = 0.8164 ; 1 / sqrt(1.5000)
|
||||
* [144] 0x0c0a = 0.7520 ; 1 / sqrt(1.7500)
|
||||
* And representative sub-entries within octave 0 (mantissa in [0x8000, 0x8100)):
|
||||
* [0] 0x1000 = 1.0000 ; 1 / sqrt(0x8000)
|
||||
* [1] 0x0fe0 = 0.9922 ; 1 / sqrt(0x8100)
|
||||
* [2] 0x0fc1 = 0.9846 ; 1 / sqrt(0x8200)
|
||||
* [3] 0x0fa3 = 0.9773 ; 1 / sqrt(0x8300)
|
||||
* [4] 0x0f85 = 0.9700 ; 1 / sqrt(0x8400)
|
||||
* [5] 0x0f68 = 0.9629 ; 1 / sqrt(0x8500)
|
||||
* [6] 0x0f4c = 0.9561 ; 1 / sqrt(0x8600)
|
||||
* [7] 0x0f30 = 0.9492 ; 1 / sqrt(0x8700)
|
||||
* ─── Local port of PSYQ's sqrtbl (1/sqrt lookup table for VectorNormal). ───
|
||||
* Source: PSYQ 4.7 libgte sqrtbl at 0x800185B4 in hello_camera.elf.
|
||||
* objdump -s --start-address=0x800185B4 --stop-address=0x800185F4 hello_camera.elf -> 192 entries x 16-bit signed, stored in 1.12 fixed point.
|
||||
* Data is identical to the libgte original (byte-for-byte verified).
|
||||
*
|
||||
* The algorithm's `addi -64 / sll 1 / lh` selects the entry at `(aligned - 64) * 2` for the case where `aligned` has its top bit at bit 24.
|
||||
* After the sllv/srav pair, `aligned` always lands in `[0x80, 0x100)`
|
||||
* (with top bit at bit 24 → after `sub $aligned - 64`, the index sits in `[0x40, 0x80) * 2 = [0x80, 0x100)` bytes = entries [64, 128) within the sqrtbl).
|
||||
* The earlier 64 entries (octave 0) are reached when the magnitude after shifting puts the top bit below bit 24 (the `sllv` branch),
|
||||
* and the load upper_halves of the table bracket the input range.
|
||||
* The later 64 entries (octaves 2-3) are the `srav` branch when the magnitude's top bit is well above bit 24.
|
||||
* ─── Table semantics ───
|
||||
* For table index i in [0, 192):
|
||||
* x = 1 + i / 64
|
||||
* tbl[i] = floor(4096 / sqrt(x))
|
||||
* Thus the table uniformly samples 1/sqrt(x) over:
|
||||
* x in [1.0, 4.0)
|
||||
* at steps of 1/64, with the result represented in 1.12 fixed point (0x1000 = 1.0).
|
||||
*
|
||||
* Reproduced verbatim from libgte (verified against libpsn00b/psxgte/vector.s:100-123 — 24 rows × 8 halfwords, last entry 0x0804).
|
||||
* */
|
||||
* Representative entries:
|
||||
* [ 0] 0x1000 = 1.000000 ; 1 / sqrt(1.000000)
|
||||
* [ 16] 0x0e4f = 0.894287 ; 1 / sqrt(1.250000)
|
||||
* [ 32] 0x0d10 = 0.816406 ; 1 / sqrt(1.500000)
|
||||
* [ 48] 0x0c18 = 0.755859 ; 1 / sqrt(1.750000)
|
||||
* [ 64] 0x0b50 = 0.707031 ; 1 / sqrt(2.000000)
|
||||
* [128] 0x093c = 0.577148 ; 1 / sqrt(3.000000)
|
||||
* [191] 0x0804 = 0.500977 ; 1 / sqrt(3.984375)
|
||||
*
|
||||
* ─── How VectorNormal indexes it ───
|
||||
* Let:
|
||||
* mag_sq = x*x + y*y + z*z
|
||||
* lzcr = leading-zero count of mag_sq
|
||||
* For a non-zero magnitude, libgte first rounds LZCR down to an even number:
|
||||
* lzcr_even = lzcr & ~1
|
||||
*
|
||||
* It then shifts mag_sq so that its significant bits land in one of two
|
||||
* adjacent normalized ranges:
|
||||
* if lzcr_even >= 24:
|
||||
* aligned = mag_sq << (lzcr_even - 24)
|
||||
* else:
|
||||
* aligned = mag_sq >> (24 - lzcr_even)
|
||||
*
|
||||
* Because lzcr_even differs from the true LZCR by at most one bit:
|
||||
* raw LZCR even -> aligned in [0x80, 0x100)
|
||||
* raw LZCR odd -> aligned in [0x40, 0x080)
|
||||
* therefore:
|
||||
* aligned in [0x40, 0x100)
|
||||
*
|
||||
* Dividing this normalized integer by 64 gives exactly the table domain:
|
||||
* x = aligned / 64
|
||||
* x in [1.0, 4.0)
|
||||
*
|
||||
* The lookup is therefore:
|
||||
* index = aligned - 0x40
|
||||
* byte_offset = index * sizeof(S2)
|
||||
* inv_len = sqrtbl[index]
|
||||
* or equivalently, matching the libgte instructions:
|
||||
* addi aligned, -64
|
||||
* sll aligned, 1
|
||||
* lh inv_len, sqrtbl + aligned
|
||||
*
|
||||
* ─── Why the domain spans [1, 4) instead of [1, 2) ───
|
||||
* Square-root scaling depends on the parity of the exponent.
|
||||
* Rounding LZCR to even absorbs exponent changes in pairs of bits, leaving the lookup mantissa normalized over a factor-of-four interval [1, 4).
|
||||
*
|
||||
* The corresponding exponent correction is retained separately as:
|
||||
* srav_shift = (31 - lzcr_even) >> 1
|
||||
*
|
||||
* After GPF multiplies the original vector components by the table's reciprocal-square-root coefficient,
|
||||
* this shift restores the exponent scale and yields the normalized vector.
|
||||
*
|
||||
* Reproduced verbatim from libgte; also matches PSn00bSDK VectorNormalS _norm_table (24 rows x 8 halfwords, final entry 0x0804).
|
||||
**/
|
||||
internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
|
||||
0x1000, 0x0fe0, 0x0fc1, 0x0fa3, 0x0f85, 0x0f68, 0x0f4c, 0x0f30,
|
||||
0x0f15, 0x0efb, 0x0ee1, 0x0ec7, 0x0eae, 0x0e96, 0x0e7e, 0x0e66,
|
||||
@@ -267,28 +300,25 @@ internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
|
||||
0x0820, 0x081c, 0x0818, 0x0814, 0x0810, 0x080c, 0x0808, 0x0804,
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_NormalizeV3S4) {
|
||||
U2 src_offset; /* offset of src V3_S4 within the BIOS scratchpad */
|
||||
U2 dst_offset; /* offset of dst V3_S4 within the BIOS scratchpad */
|
||||
};
|
||||
typedef Struct_(RegUse_build_normalize_v3s4) {
|
||||
typedef Struct_(Binds_normalize_v3s4) { U2 src_offset; U2 dst_offset; };
|
||||
typedef Struct_(RegUse_normalize_v3s4) {
|
||||
union { Reg_(V3_S4) res, src; };
|
||||
union { Reg r0, src_ptr, mac2; };
|
||||
union { Reg r1, dst_ptr; };
|
||||
union { Reg r2, dst_offset, mac1, v_sqr_aligned; };
|
||||
union { Reg r3, src_offset, btarget, shift_count, sqrtbl_index; };
|
||||
union { Reg r4, mac3, v_sqr_sum, half_shift_tmp, inv_len; };
|
||||
union { Reg r5, lzcr, half_shift; };
|
||||
union { Reg r2, dst_offset, mac1, v_sqr_aligned, sqrtbl_byte_offset; };
|
||||
union { Reg r3, src_offset, align_delta, shift_count, sqrtbl_lookup; };
|
||||
union { Reg r4, mac3, v_sqr_sum, srav_shift; };
|
||||
union { Reg r5, lzcr_raw, lzcr_even, inv_len; };
|
||||
};
|
||||
/* ─── Full normalize (all 4 stages inline) ───
|
||||
* Generic 4-stage GTE normalize (SQR → sum+LZCR → align+sqrtbl → GPF+srav). */
|
||||
internal MipsAtom* build_normalize_v3s4(AtomArena_R aa, RegUse_build_normalize_v3s4 r)
|
||||
internal MipsAtom* normalize_v3s4(AtomArena_R aa, RegUse_normalize_v3s4 r)
|
||||
MipsAtom_Proc_(aa, {
|
||||
load_half(r.src_offset, R_TapePtr, O_(Binds_NormalizeV3S4, src_offset)),
|
||||
load_half(r.dst_offset, R_TapePtr, O_(Binds_NormalizeV3S4, dst_offset)),
|
||||
load_half(r.src_offset, R_TapePtr, O_(Binds_normalize_v3s4, src_offset)),
|
||||
load_half(r.dst_offset, R_TapePtr, O_(Binds_normalize_v3s4, dst_offset)),
|
||||
LdSlot_ add_u(r.src_ptr, R_ScratchBase, r.src_offset),
|
||||
LdSlot_ add_u(r.dst_ptr, R_ScratchBase, r.dst_offset),
|
||||
LdSlot_ add_ui_self(R_TapePtr, S_(Binds_NormalizeV3S4)),
|
||||
LdSlot_ add_ui_self(R_TapePtr, S_(Binds_normalize_v3s4)),
|
||||
|
||||
mac_load_v3s4(r.src, r.src_ptr, 0),
|
||||
|
||||
@@ -300,33 +330,32 @@ MipsAtom_Proc_(aa, {
|
||||
add_u_self( r.v_sqr_sum, r.mac1),
|
||||
add_u_self( r.v_sqr_sum, r.mac2),
|
||||
gte_mv_to_data_r( r.v_sqr_sum, C2_LZCS), GteDelay_ nop2,
|
||||
gte_mv_from_data_r(r.lzcr, C2_LZCR), GteDelay_ nop,
|
||||
gte_mv_from_data_r(r.lzcr_raw, C2_LZCR), GteDelay_ nop,
|
||||
|
||||
/* Stage 3: even(LZCR), half-shift, align |v|² to bit 24. */
|
||||
mac_lzcr_round_even_half_shift(r.lzcr, r.v_sqr_sum, r.v_sqr_aligned),
|
||||
add_si( r.btarget, r.lzcr, -24),
|
||||
branch_lt_zero(r.btarget, atom_offset(aligned_done, srav_path)), BdSlot_ nop, /* bltz → srav_path (LZCR < 24 path) */
|
||||
mac_lzcr_round_even_half_shift(r.lzcr_raw, r.v_sqr_sum, r.v_sqr_aligned),
|
||||
add_si( r.align_delta, r.lzcr_even, -24),
|
||||
branch_lt_zero(r.align_delta, atom_offset(aligned_done, srav_path)), BdSlot_ nop, /* bltz → srav_path (LZCR < 24 path) */
|
||||
jump_rel(atom_offset(srav_path, aligned_done)), /* b → aligned_done (LZCR >= 24 path) */
|
||||
BdSlot_ shift_lleft_var(r.v_sqr_aligned, r.v_sqr_aligned, r.btarget),
|
||||
BdSlot_ shift_lleft_var(r.v_sqr_aligned, r.v_sqr_aligned, r.align_delta),
|
||||
atom_label(srav_path)
|
||||
li_s( r.shift_count, 24),
|
||||
sub_s(r.shift_count, r.shift_count, r.lzcr),
|
||||
sub_s(r.shift_count, r.shift_count, r.lzcr_even),
|
||||
shift_aright_var(r.v_sqr_aligned, r.v_sqr_aligned, r.shift_count),
|
||||
atom_label(aligned_done)
|
||||
or_u(r.half_shift, r.half_shift_tmp, 0),
|
||||
add_si( r.v_sqr_aligned, r.v_sqr_aligned, -64),
|
||||
shift_lleft(r.v_sqr_aligned, r.v_sqr_aligned, 1),
|
||||
mac_load_word_imm(r.sqrtbl_index, & gte_normalize_sqr_tbl), add_u_self(r.sqrtbl_index, r.v_sqr_aligned),
|
||||
load_half(r.inv_len, r.sqrtbl_index, 0),
|
||||
add_si( r.sqrtbl_byte_offset, r.v_sqr_aligned, -64),
|
||||
shift_lleft(r.sqrtbl_byte_offset, r.sqrtbl_byte_offset, 1),
|
||||
mac_load_word_imm(r.sqrtbl_lookup, & gte_normalize_sqr_tbl), add_u_self(r.sqrtbl_lookup, r.sqrtbl_byte_offset),
|
||||
load_half(r.inv_len, r.sqrtbl_lookup, 0),
|
||||
LdSlot_ nop,
|
||||
|
||||
mac_gte_general_purpose_interopolation(r.inv_len,
|
||||
r.src.x, r.src.y, r.src.z,
|
||||
r.res.x, r.res.y, r.res.z,
|
||||
GteDelay_ load_word(R_AtomJmp, R_TapePtr, 0), LdSlot_ // ac_yield: word 1
|
||||
GteDelay_ add_ui_self( R_TapePtr, S_(MipsCode)) // ac_yield: word 2
|
||||
GteDelay_ add_ui_self( R_TapePtr, S_(MipsCode)) // ac_yield: word
|
||||
),
|
||||
mac_shift_aright_var_v3s4_self(r.res, r.half_shift),
|
||||
mac_shift_aright_var_v3s4_self(r.res, r.srav_shift),
|
||||
mac_store_v3s4(r.res, r.dst_ptr, 0),
|
||||
|
||||
jump_reg(R_AtomJmp), BdSlot_ nop // ac_yield: word 3-4
|
||||
@@ -340,21 +369,21 @@ typedef Struct_(Binds_gte_cross_v3s4) { V3_S4* src_a; V3_S4* src_b; V3_S4* out;
|
||||
typedef Struct_(RegUse_gte_cross_v3s4) {
|
||||
Reg_(V3_S4) a;
|
||||
Reg_(V3_S4) b;
|
||||
union { Reg out, t0; } x;
|
||||
union { Reg src_a, t1, rt11; } y;
|
||||
union { Reg src_b, t2, rt22; } z;
|
||||
Reg out;
|
||||
Reg src_a;
|
||||
Reg src_b;
|
||||
};
|
||||
internal MipsAtom* gte_cross_v3s4(AtomArena_R aa, RegUse_gte_cross_v3s4 r)
|
||||
atom_info(atom_bind(Binds_gte_cross_v3s4)) MipsAtom_Proc_(aa, {
|
||||
load_word(r.y.src_a, R_TapePtr, O_(Binds_gte_cross_v3s4,src_a)),
|
||||
load_word(r.z.src_b, R_TapePtr, O_(Binds_gte_cross_v3s4,src_b)),
|
||||
load_word(r.x.out, R_TapePtr, O_(Binds_gte_cross_v3s4,out)),
|
||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_gte_cross_v3s4)),
|
||||
load_word(r.src_a, R_TapePtr, O_(Binds_gte_cross_v3s4,src_a)),
|
||||
load_word(r.src_b, R_TapePtr, O_(Binds_gte_cross_v3s4,src_b)),
|
||||
load_word(r.out, R_TapePtr, O_(Binds_gte_cross_v3s4,out)),
|
||||
LdSlot_ add_ui_self(R_TapePtr, S_(Binds_gte_cross_v3s4)),
|
||||
|
||||
mac_load_v3s4(r.a, r.y.src_a, 0), LdSlot_
|
||||
mac_load_v3s4(r.b, r.z.src_b, 0), LdSlot_
|
||||
mac_load_v3s4(r.a, r.src_a, 0), LdSlot_
|
||||
mac_load_v3s4(r.b, r.src_b, 0), LdSlot_
|
||||
mac_gte_op_cross_v3s4(r.a, r.b), /* RT diagonal + IR + OP + MAC read + shift */
|
||||
mac_store_v3s4(r.a, r.x.out, 0),
|
||||
mac_store_v3s4(r.a, r.out, 0),
|
||||
|
||||
mac_yield()
|
||||
})
|
||||
|
||||
+1
-26
@@ -1,21 +1,10 @@
|
||||
/* ============================================================================
|
||||
* duffle DSL Suffix Conventions
|
||||
* ============================================================================
|
||||
*
|
||||
* Every mnemonic in this header follows the same suffix grammar:
|
||||
*
|
||||
* Primitive commands: gp0_cmd_poly_f3 = 0x20 (byte opcode)
|
||||
* Packed 32-bit cmd: gp0_word_poly_f3(r, g, b) (32-bit, shifted)
|
||||
*
|
||||
* Type ordering: domain?_(direction)?_action_target_modifier_type?
|
||||
* Examples: add_ui (add + unsigned + immediate)
|
||||
* add_s (add + signed, R-type implicit)
|
||||
* shift_lleft (shift + logical + left)
|
||||
* shift_aright (shift + arithmetic + right)
|
||||
* call_reg(rs) (call + register, $ra implicit)
|
||||
* gte_mv_to_data_r (gte + mv + to + data + register)
|
||||
* gte_lw_v0_xy(base) (gte + lw + v0 + xy)
|
||||
* load_upper_i (load-upper + immediate, unique verb)
|
||||
* ============================================================================ */
|
||||
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
@@ -29,21 +18,7 @@
|
||||
/* ============================================================================
|
||||
* gte.h — Geometry Transformation Engine (COP2) for the PS1
|
||||
* ============================================================================
|
||||
*
|
||||
* Hand-rolled DSL for emitting GTE/MIPS instruction words from C.
|
||||
* No GCC inline-assembly string syntax in the code body.
|
||||
*
|
||||
* STYLE NOTES
|
||||
* -----------
|
||||
* - Per-field encoders are named `enc_gte_<field>(value)` and each one self-masks its argument before shifting.
|
||||
* Mirrors the `enc_op / enc_rs / enc_rt / ...` family in mips.h.
|
||||
* - The composite `enc_gte_cmdw(sf, mx, v, cv, lm, cmd)` is a flat OR of the per-field encoders, plus the COP2/CO base.
|
||||
* - Pre-baked shortcuts (`gte_cmd_rtpt`, `gte_cmd_rtps`, …) are defined for the common cases so call sites read like assembly source.
|
||||
* - All register/field values are enums (not `#define`s) so they show up in debugger symbol tables and IDE autocomplete.
|
||||
*
|
||||
* SEE ALSO
|
||||
* --------
|
||||
* - mips.h: The MIPS encoder layer this builds on.
|
||||
* DSL for emitting GTE/MIPS instruction words from C.
|
||||
*/
|
||||
|
||||
/* C2 data registers */
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
# include "gen/macs.h"
|
||||
# include "gen/offsets.h"
|
||||
# include "math.h"
|
||||
# include "lottes_tape.h"
|
||||
# include "tape.h"
|
||||
#endif
|
||||
|
||||
ATOM_FILE_DEBUGGER_LINE_MARKER(math_atom_c);
|
||||
|
||||
+16
-17
@@ -5,7 +5,7 @@
|
||||
|
||||
#define MEM_ALIGNMENT_DEFAULT 4
|
||||
|
||||
#define assert_bounds(point, start, end) for(;0;){ \
|
||||
#define assert_bounds(point, start, end) do{ \
|
||||
assert((start) <= (point)); \
|
||||
assert((point) <= (end)); \
|
||||
} while(0)
|
||||
@@ -23,10 +23,10 @@ FI_ void mem_bump(U4 cap, U4*R_ used, U4 amount) {
|
||||
used[0] += amount;
|
||||
}
|
||||
|
||||
FI_ U4 mem_copy (U4 dest, U4 src, U4 len) { return (U4)(__builtin_memcpy ((void*)dest, (void const*)src, len)); }
|
||||
FI_ U4 mem_copy_overlapping(U4 dest, U4 src, U4 len) { return (U4)(__builtin_memmove((void*)dest, (void const*)src, len)); }
|
||||
FI_ U4 mem_fill (U4 dest, U4 value, U4 len) { return (U4)(__builtin_memset ((void*)dest, (int) value, len)); }
|
||||
FI_ B4 mem_zero (U4 dest, U4 len) { if(dest == 0){return false;} mem_fill(dest, 0, len); return true; }
|
||||
FI_ U4 mem_copy (U1_R dest, U1_R src, U4 len) { return (U4)(__builtin_memcpy ((void*)dest, (void const*)src, len)); }
|
||||
FI_ U4 mem_copy_overlapping(U1* dest, U1* src, U4 len) { return (U4)(__builtin_memmove((void*)dest, (void const*)src, len)); }
|
||||
FI_ U4 mem_fill (U1_R dest, U4 value, U4 len) { return (U4)(__builtin_memset ((void*)dest, (int) value, len)); }
|
||||
FI_ B4 mem_zero (U1_R dest, U4 len) { if(dest == 0){return false;} mem_fill(dest, 0, len); return true; }
|
||||
|
||||
#pragma region DAG
|
||||
|
||||
@@ -58,31 +58,30 @@ typedef Struct_(Str8) { UTF8* ptr; U4 len; };
|
||||
typedef Struct_(Slice_Str8) { Str8* ptr; U4 len; };
|
||||
#define slit(string_literal) (Str8){ (UTF8*) string_literal, S_(string_literal) - 1 }
|
||||
|
||||
typedef Struct_(Slice) { B1* ptr; U4 len; }; // Untyped Slice (byte-addressable; .len in elements)
|
||||
FI_ Slice slice_ut_(U4 ptr, U4 len) { return (Slice){(B1*)ptr, len}; }
|
||||
typedef Struct_(Slice) { U1* ptr; U4 len; };
|
||||
FI_ Slice slice_ut_(U1* ptr, U4 len) { return (Slice){ptr, len}; }
|
||||
|
||||
#define Slice_(type) Struct_(tmpl(Slice,type)) { type* ptr; U4 len; }
|
||||
typedef Slice_(B1);
|
||||
#define slice_assert(s) do { assert((s).ptr != 0); assert((s).len > 0); } while(0)
|
||||
#define slice_end(slice) ((slice).ptr + S_slice(slice) / S_(B1)) /* byte-ptr arithmetic; .len is in elements per slice convention */
|
||||
#define slice_end(slice) ((slice).ptr + S_slice(slice))
|
||||
#define S_slice(s) ((s).len * S_((s).ptr[0]))
|
||||
|
||||
#define slice_ut(ptr,len) slice_ut_(u4_(ptr), u4_(len))
|
||||
#define slice_ut_arr(a) slice_ut_(u4_(a), S_(a))
|
||||
#define slice_to_ut(s) slice_ut_(u4_((s).ptr), S_slice(s))
|
||||
#define slice_ut(ptr,len) slice_ut_(C_(U1*,ptr), u4_(len))
|
||||
#define slice_ut_arr(a) slice_ut_(C_(U1*,a), S_(a))
|
||||
#define slice_to_ut(s) slice_ut_(C_(U1*,(s).ptr), S_slice(s))
|
||||
|
||||
#define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter)
|
||||
#define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = Array_decl(type,__VA_ARGS__), .len = Array_len( Array_decl(type,__VA_ARGS__)) }
|
||||
#define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = Array_len(array) }
|
||||
|
||||
FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(u4_(s.ptr), s.len); }
|
||||
FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(s.ptr, s.len); }
|
||||
#define slice_zero(s) slice_zero_(slice_to_ut(s))
|
||||
|
||||
FI_ void slice_copy_(Slice dest, Slice src) {
|
||||
assert(S_slice(dest) >= S_slice(src));
|
||||
slice_assert(dest);
|
||||
slice_assert(src);
|
||||
mem_copy(u4_(dest.ptr), u4_(src.ptr), S_slice(src));
|
||||
mem_copy(dest.ptr, src.ptr, S_slice(src));
|
||||
}
|
||||
#define slice_copy(dest, src) do { \
|
||||
static_assert(T_same(dest, src)); \
|
||||
@@ -95,6 +94,7 @@ FI_ Slice slice_bump(U4_R used, U4 start, U4 len, U4 amount) {
|
||||
return slice_ut(ptr, amount);
|
||||
}
|
||||
|
||||
typedef Slice_(B1);
|
||||
typedef Slice_(U1);
|
||||
typedef Slice_(U4);
|
||||
|
||||
@@ -117,7 +117,7 @@ I_ Slice farena_push(FArena_R arena, U4 amount, Opt_farena o) {
|
||||
U4 to_commit = align_pow2(desired, o.alignment ? o.alignment : MEM_ALIGNMENT_DEFAULT);
|
||||
U4 ptr = arena->start + arena->used;
|
||||
mem_bump(arena->capacity, & arena->used, to_commit);
|
||||
return (Slice){ (B1*)ptr, to_commit };
|
||||
return (Slice){ (U1*)ptr, to_commit };
|
||||
}
|
||||
FI_ void farena_reset (FArena_R arena) { arena->used = 0; }
|
||||
FI_ void farena_rewind(FArena_R arena, U4 save_point) {
|
||||
@@ -134,8 +134,7 @@ FI_ U4 farena_unused_start(FArena arena) { return arena.start + arena.used; }
|
||||
|
||||
#pragma region BIOS Scratchpad
|
||||
/* BIOS scratchpad location. 1 KB at 0x1F800000.
|
||||
* TapeHostFrame occupies the final 44 bytes while tape code executes.
|
||||
* Atom scratch is bounded by the TapeHostFrame_Loc declaration in lottes_tape.h. */
|
||||
* TapeHostFrame occupies the final 44 bytes while tape code executes. */
|
||||
enum {
|
||||
Scratchpad_Loc = 0x1F800000,
|
||||
Scratchpad_Len = 0x400, /* 1 KB */
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
# include "gen/offsets.h"
|
||||
# include "bios.h"
|
||||
# include "mips.h"
|
||||
# include "lottes_tape.h"
|
||||
# include "tape.h"
|
||||
#endif
|
||||
|
||||
ATOM_FILE_DEBUGGER_LINE_MARKER(mips_atom_c);
|
||||
|
||||
+2
-7
@@ -80,14 +80,9 @@ enum {
|
||||
* Every R_* enum below has a parallel R_*_Code `#define` so that the preprocessor can stringify the integer
|
||||
* (e.g. for asm clobber lists and register-variable declarations via `rgcc(R_X)`).
|
||||
* The enum value is bound to the `#define` so the two forms cannot drift apart.
|
||||
*
|
||||
* Only registers that get stringified need a `_Code` form; the rest are plain enum values.
|
||||
* If you need to add a new one, follow the pattern:
|
||||
* #define R_T7_Code 15
|
||||
* R_T7 = R_T7_Code, // in the enum
|
||||
*
|
||||
*
|
||||
* User code should always reference the enum form (`R_T4`) at arithmetic sites and let
|
||||
* `rlit(R_T4_Code)` / `rgcc(R_T4)` handle the stringify cases — never write the bare number `12`.
|
||||
* `rlit(R_T4_Code)` / `rgcc(R_T4)` handle the stringify cases
|
||||
* ============================================================================ */
|
||||
#define R_0_Code 0
|
||||
#define R_AT_Code 1
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
# include "gen/offsets.h"
|
||||
# include "mips.h"
|
||||
# include "dsl.atom.h"
|
||||
# include "lottes_tape.h"
|
||||
# include "tape.h"
|
||||
# include "pad.h"
|
||||
#endif
|
||||
|
||||
|
||||
+2
-11
@@ -6,16 +6,10 @@
|
||||
# include "pad.h"
|
||||
#endif
|
||||
|
||||
/* Uses ONE 8-byte frame allocated via the compiler's standard prologue.
|
||||
/* Uses an 8-byte frame allocated via the compiler's standard prologue.
|
||||
* 4 wasted-arg words for B(12h) InitPAD2 are at [SP+0..15] but are not explicitly allocated.
|
||||
* Compiler handles the MIPS O32 "wasted stack" convention for us by treating the B-call as a 4-arg call.
|
||||
*
|
||||
* The buffer pointers are passed as arguments so the compiler keeps them in callee-saved registers;
|
||||
* The B(12h) asm volatile block does NOT clobber those registers (it clobbers only the volatile GPRs + B-table arg registers explicitly).
|
||||
* The C-level writes after the call re-load the pointers from their callee-saved homes.
|
||||
*
|
||||
* The clobber list for both B-calls names the full BIOS destroy set documented in kernelbios.md:167-174 (R1..R15, R24..R25, R31, HI/LO).
|
||||
* The kernel-ABI "volatile GPRs" subset is clb_mem_drain; the rest of the destroy set is enumerated explicitly here. */
|
||||
*/
|
||||
NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
|
||||
{
|
||||
/* Pin raw0 + raw1 to $a0 + $a1 via rgcc; the B(12h) call uses these directly.
|
||||
@@ -24,9 +18,6 @@ NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
|
||||
register PadBiosRaw* p1 rgcc(R_A1) = raw1;
|
||||
(void)p0; (void)p1;
|
||||
|
||||
// TODO(Ed): Properly annotate the raw values in the inline asm instructions.
|
||||
// Use enums.
|
||||
|
||||
/* B(12h) InitPAD2(raw0, 0x22, raw1, 0x22)
|
||||
* $a0 = raw0 (rgcc-bound; survives the sequence below)
|
||||
* $a1 = raw1 (preserved into $a2 before $a1 is overwritten)
|
||||
|
||||
+2
-9
@@ -4,10 +4,7 @@
|
||||
# include "math.h"
|
||||
#endif
|
||||
|
||||
/* PSX button bit positions — 1:1 with PSX-SPX docs at docs/psx-spx/docs/controllersandmemorycards.md:405-421.
|
||||
* Wire is active-low (0 = pressed).
|
||||
* The decoder atom computes buttons = (~raw_buttons) & 0xFFFF;
|
||||
* active-low-to-active-high inversion is applied bit-by-bit. */
|
||||
// PSX button bit positions: PSX-SPX docs/psx-spx/docs/controllersandmemorycards.md:405-421.
|
||||
typedef Enum_(U2, PadBtns) {
|
||||
Bit_(Pad_Select, 0),
|
||||
Bit_(Pad_L3, 1),
|
||||
@@ -62,11 +59,7 @@ typedef Enum_(U4, PadStatus) {
|
||||
PadStatus_Invalid,
|
||||
};
|
||||
|
||||
/* Distinct from the game-facing PadStatus enum: PadRawStatus_Ok and PadRawStatus_Timeout are raw BIOS values;
|
||||
* PadStatus_* are game-facing post-decode states. PadUnknownId_Sentinel is written by the decoder
|
||||
* when the controller id does not match any known controller type.
|
||||
* PadAxisCentered_Word: Four-byte 0x80 pattern used to clear / center
|
||||
* four byte axes at PadState.left_x through PadState.right_y. */
|
||||
// Distinct from the game-facing PadStatus enum: PadRawStatus_Ok and PadRawStatus_Timeout are raw BIOS values
|
||||
typedef Enum_(U1, PadRawStatus) {
|
||||
PadRawStatus_Ok = 0x00,
|
||||
PadRawStatus_Timeout = 0xFF,
|
||||
|
||||
+1
-1
@@ -104,7 +104,7 @@ void gte_matrix_set_translation(MT3_S2S4* mat) asm("SetTransMatrix");
|
||||
|
||||
// Einheit, Metrication to unit vector. "Normalization", not Orthogonal "Normal, Normalis". Directionalization.
|
||||
// RGA(Lengyel): Normalize the bulk of a zero-weight direction. This is not finite-point unitization (which forces w=1).
|
||||
S4 normalize_v3s4(V3_S4* v0, V3_S4* v1) asm("VectorNormal");
|
||||
S4 psy_normalize_v3s4(V3_S4* v0, V3_S4* v1) asm("VectorNormal");
|
||||
|
||||
// RGA(Lengyel): Apply the matrix expansion of a rigid transformation.
|
||||
// Motor antiproduct is equivalent for unitized points; LA form is what GTE consumes.
|
||||
|
||||
@@ -13,56 +13,32 @@
|
||||
|
||||
#pragma region Tape Drive
|
||||
/* -----------------------------------------------------------------------------------------------------------
|
||||
* TAPE DRIVE ABI
|
||||
* THREADED ATOMS - TAPE EXECUTION & ABI
|
||||
* _________
|
||||
* | ___ |
|
||||
* | o___o | ,-----<-----.
|
||||
* |__/___\__| V ^
|
||||
* \_[Enter]_[A]->[A]->[A]->[A(B)]->[A]->[Exit]
|
||||
* -----------------------------------------------------------------------------------------------------------
|
||||
* Note(Ed): One of the main purposes of this codebase is to help me learn this,
|
||||
* as such the information below may not* be entirely realized or finalized conceptually.
|
||||
* -----------------------------------------------------------------------------------------------------------
|
||||
* This ABI and its associated legos were directly inspired by researching the work of
|
||||
* Timothy Lottes and Onat Türkçüoğlu; along with many others. It's the simplest bootstrap of a
|
||||
* directly executed chain of assemby arrays (Atoms) that terminate with a yield sequence to the next atom.
|
||||
* These eventually lead to a terminal atom for the tape which is defined below as "tape_exit".
|
||||
* This ABI and its associated legos were directly inspired by researching the work of Timothy Lottes and
|
||||
* Onat Türkçüoğlu; Forth, threaded code system, and various other people or programming techniques.
|
||||
*
|
||||
* The setup is simple:
|
||||
* A tape is a linear stream containing addresses of directly executable native-code fragments ("Atoms").
|
||||
* Most atoms terminate in a small yield sequence which loads the next atom address from the tape.
|
||||
* It's a runtime composed of directly executed native machine-code sequences (Atoms) that usually terminate
|
||||
* in a yield sequence to the next atom. These eventually lead to a terminal atom for the tape
|
||||
* which is defined below as "tape_exit". Traditionally referred to as Direct Threaded Execution.
|
||||
*
|
||||
* It behaves as one of the simplest runtime harnesses ontop of a host-enviornment's execution engine
|
||||
* to author and compose programs with. From here various conventions can be further applied.
|
||||
* To make things easier to understand it may be better to focus on what this ABI does not have.
|
||||
* It does not have have any branching within the tape but relative branches within atoms or between atoms.
|
||||
* Branching nearly is always downstream. Automatic stack usage is non-existent.
|
||||
* Push/Pop, FIFO, or Arena/Bump data structures are used by atoms explicitly.
|
||||
* In it's current form with the C11 macro DSL, the user also has fullfill manual register allocation per atom.
|
||||
*
|
||||
* One of the remarkable things about utilizing this ABI is its essentially interopable with CPUs, GPUs, FPGA,
|
||||
* or, basically anything from the 5th generation consoles and onward.
|
||||
* The ABI directly reflects how all computational hardware must be architected in order to execute
|
||||
* digital logic effectively on current era tech.
|
||||
* On the PS1 we don't have access to a few features like multi-threading, speculative execution, or L3 cache;
|
||||
* but, we can set the foundation for legoing whats required for eventually expanding this ABI's paradigm
|
||||
* and core atoms to take those newer hardware features into account. For example, you can easily expand
|
||||
* this to support wave-based execution model on a PS2 or PS3. Not having a stack or
|
||||
* automatic register allocation means the user cannot ignore excessive argument shuffle across workload or
|
||||
* waves and thier phases. Crossing ABI boundaries to other runtimes that do has obviouss penalties.
|
||||
*
|
||||
* Learning data-oriented code becomes a natural progression. Your not fighting a stack-based procedural
|
||||
* paradigm that wants to argument shuffle. There is no ambiguity due to the lack of constraints, for example,
|
||||
* on how the user may "call" a procedure in traditional random dispatch runtimes. The user does have to
|
||||
* hammer down "rules" or patterns for massaging the compiler to dissolve those call frames; just to get
|
||||
* the asesmbly into its desired form. The form is obvious, and once the user gets to author these compoonents
|
||||
* it becomes a game of tetris.
|
||||
*
|
||||
* Another feature is this ABI is very compatible with bootstrapping and developing simple toolchains built off
|
||||
* of bit-packed annotated command streams the user can directly author, maintatain, and immediately execute.
|
||||
* That being like a color forth, or maybe something more familar like an immediate mode library
|
||||
* for various systems such as GUIs. This can make the tetris less of a chore with some helpful policy
|
||||
* generation for allocation of registers, helping to choose resuable components, designing DSL on the fly, etc.
|
||||
* -----------------------------------------------------------------------------------------------------------
|
||||
* TODO(Ed): We need pretty ascii diagrams and proper guides, articles, etc.
|
||||
* -----------------------------------------------------------------------------------------------------------
|
||||
* For now this ideation has just started functioning. I'm abusing C11 & a lua metaprogram to help establish
|
||||
* a hybrid toolchain to ideate on a traditional text-based authoring UX for this paradigm.
|
||||
* If pcsx-redux provides viable hot-reload and persistent data storage beyond save-states
|
||||
* (just copying ram to filesystem), I can author a color forth to mess around with.
|
||||
* With either an editor in-emulator or on the actual machine itself. Assembly is tedius,
|
||||
* but I think this codebase most likely has a pretty ergonomic flavor worst case...
|
||||
* To make things easier to understand it may be better to focus on what this ABI does not have.
|
||||
* The tape itself does not have have any branching behavior.
|
||||
* Branches, loops, skips, or other control-flow policies must be implemented explicitly by atoms.
|
||||
* Push/Pop, FIFO, or Arena/Bump data structures are utilized by atoms explicitly.
|
||||
* There is no implicit call-stack, return stack, or per-atom stack-frame.
|
||||
* The user must also explictly handle register allocation per atom (by default).
|
||||
* However they could procedurally automate it using metaprogramming functionality.
|
||||
* */
|
||||
/* Register Allocation Info */
|
||||
enum {
|
||||
@@ -99,8 +75,8 @@ enum {
|
||||
R_Atom7 = R_T7,
|
||||
R_Atom8 = R_T8,
|
||||
R_Atom9 = R_T9,
|
||||
R_Atom10 = R_V0, // Tend to be used with gte DMAs
|
||||
R_Atom11 = R_V1, // Tend to be used with gte DMAs
|
||||
R_Atom10 = R_V0, // Tend to be used with gte moves
|
||||
R_Atom11 = R_V1, // Tend to be used with gte moves
|
||||
R_Atom12 = R_A0,
|
||||
R_Atom13 = R_A1,
|
||||
R_Atom14 = R_A2,
|
||||
@@ -121,7 +97,7 @@ typedef U2 Reg; // Register parameter used with atom or atom component procedure
|
||||
typedef U4 const MipsCode; // Underlying type to mips asm words.
|
||||
typedef Slice_(MipsCode);
|
||||
|
||||
typedef U4 const MipsAtom; // Underlying type to a mips atom defnition
|
||||
typedef U4 const MipsAtom; // Underlying type to a mips atom definition
|
||||
typedef Slice_(MipsAtom);
|
||||
|
||||
// Sometimes a user will define a bundle of atoms that represent a procedure of work as:
|
||||
@@ -163,6 +139,12 @@ typedef Slice_(MipsAtom);
|
||||
// Used for trivial mappings from one atom component proc to the command of a more baser (meant for type-mapping)
|
||||
#define MipsAtomComp_ProcMap_(ab, base_command) atom_dbg_skip MipsAtomComp_Proc_(ab, {base_command })
|
||||
|
||||
// WIP: Atoms Assocated closely with each other to form a tape procedure. (Maybe also a phase in a procedure/pipeline?)
|
||||
|
||||
#define AtomBundle_(name) Struct_(tmpl(AtomBundle,name))
|
||||
#define AtomBundle_Len(name) S_(tmpl(AtomBundle,name))/S_(MipsAtom*)
|
||||
#define AtomBundleEntry_(bundle,entry) tmpl(bundle,entry)
|
||||
|
||||
/* Line-table anchor: gcc only adds a file to the .debug_line file table when the contains line-numbered content.
|
||||
Files containing only atoms and atom components.
|
||||
Place `ATOM_FILE_LINE_MARKER();` once at file scope in any `.atom.c` that defines atoms.
|
||||
@@ -170,7 +152,7 @@ typedef Slice_(MipsAtom);
|
||||
The constant is in `.rodata` so the linker may eliminate it. */
|
||||
#define ATOM_FILE_DEBUGGER_LINE_MARKER(file_name) internal U4 const tmpl(atom_file_debugger_line_marker,file_name) = 0
|
||||
|
||||
typedef Slice_MipsAtom Tape;
|
||||
typedef Struct_(Tape) { union { MipsAtom* ptr; U4* inlaid_data; }; U4 len; };
|
||||
|
||||
typedef Struct_(TapeHostFrame) {
|
||||
U4 s0;
|
||||
@@ -236,22 +218,27 @@ FI_ void tape_run(Tape tape) { C_(TapeEntryFn*, tape_enter)(tape.ptr); }
|
||||
// Procedural authoring of tapes:
|
||||
typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; };
|
||||
FI_ void tb_init(TapeBuilder* tb, FArena* arena) { tb->ptr = arena->start; tb->used = 0; }
|
||||
FI_ TapeBuilder tb_make_old( FArena* arena) { return (TapeBuilder){ arena->start, 0 }; }
|
||||
FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ u4_(mem.ptr), mem.len, 0 }; } /* capacity in elements (matches used units) */
|
||||
|
||||
FI_ void tb_emit(TapeBuilder* tb, MipsAtom* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
|
||||
FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; }
|
||||
#define tb_emit_(atom) tb_emit(& tb, atom)
|
||||
#define tb_data_(field, data) tb_data(& tb, u4_(data))
|
||||
|
||||
FI_ void tb_emit_bundle(TapeBuilder_R tb, Slice_MipsAtom atoms) { mem_copy(u4_(tb->ptr), u4_(atoms.ptr), S_slice(atoms)); tb->used += atoms.len; }
|
||||
FI_ void tb_bind(TapeBuilder* tb, Slice data) { mem_copy(b1_r(tb->ptr + tb->used * S_(MipsCode)), data.ptr, data.len); tb->used += data.len / S_(MipsCode); }
|
||||
#define tb_bind_(tb,type,...) tb_bind(tb, (Slice){ (U1*)(& (type){__VA_ARGS__}), S_(type) }); static_assert(S_(type) % S_(MipsCode) == 0)
|
||||
|
||||
FI_ Tape tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Tape){ C_(U4*,tb->ptr), tb->used }; }
|
||||
FI_ Tape tb_slice(TapeBuilder tb) { return (Tape){ C_(U4*,tb.ptr), tb.used }; }
|
||||
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit))
|
||||
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit))
|
||||
|
||||
FI_ void tb_scope_run_end(TapeBuilder* tb) { tb_emit(tb,tape_exit); tape_run(tb_slice(tb[0])); }
|
||||
#define tb_scope_run(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_scope_run_end(tb))
|
||||
|
||||
|
||||
// NOTE(Ed): Wip still ideating convention. Possibly will never use a composite.
|
||||
#define tb_emit_wbind_(tb,atom,...) tb_emit(tb,atom); tb_bind_(tb,tmpl(Binds,atom),__VA_ARGS__)
|
||||
#define tb_emit_wbind2_(tb,atom,type,...) tb_emit(tb,atom); tb_bind_(tb,type,__VA_ARGS__)
|
||||
|
||||
#pragma endregion Tape Drive
|
||||
|
||||
#pragma region Macro Mips Atom Components
|
||||
@@ -260,7 +247,8 @@ FI_ void tb_scope_run_end(TapeBuilder* tb) { tb_emit(tb,tape_exit); tape_run(tb_
|
||||
* These do NOT yield. They are expanded inline inside Tape Atoms.
|
||||
* ---------------------------------------------------------------------------*/
|
||||
|
||||
// The 'Yield' sequence for Tape Atoms (mac_yield).
|
||||
// The 'Yield' sequence for Tape Atoms (mac_yield).
|
||||
// In Forth this is considered the "NEXT" mechanism.
|
||||
|
||||
atom_dbg_skip MipsAtomComp_(ac_yield) {
|
||||
load_word(R_AtomJmp, R_TapePtr, 0), LdSlot_
|
||||
@@ -289,7 +277,7 @@ typedef Relative_(FArena) Struct_(AtomBuilder) { U4 start; U4 capacity; U4 used;
|
||||
FI_ void atombuilder_push(AtomBuilder_R ab, Slice_MipsCode code) {
|
||||
assert(ab->capacity - ab->used - code.len);
|
||||
U4 dest = ab->start + ab->used * S_(MipsCode); U4 size = S_slice(code);
|
||||
mem_copy(dest, u4_(code.ptr), size); ab->used += size;
|
||||
mem_copy(b1_r(dest), b1_r(code.ptr), size); ab->used += size;
|
||||
}
|
||||
#define atombuilder_push_mac(ab, mac) atombuilder_push(ab, slice_arg_from_array(Slice_MipsCode, mac))
|
||||
|
||||
@@ -313,7 +301,7 @@ FI_ AtomArena atomarena_make(Slice mem) { AtomArena a; atomarena_init(& a, mem);
|
||||
FI_ MipsAtom* atomarena_push(AtomArena_R aa, Slice_MipsCode code) {
|
||||
assert(aa->capacity - aa->used - code.len);
|
||||
U4 dest = atomarena_unused_start(aa[0]); U4 size = S_slice(code);
|
||||
mem_copy(dest, u4_(code.ptr), size); aa->used += size;
|
||||
mem_copy(b1_r(dest), b1_r(code.ptr), size); aa->used += size;
|
||||
return C_(MipsAtom*, dest);
|
||||
}
|
||||
FI_ void atomarena_reset(AtomArena_R aa) { aa->used = 0; }
|
||||
@@ -344,24 +332,14 @@ internal Reg const regfile_alloc_order[] = {
|
||||
R_T8, R_T9,
|
||||
};
|
||||
|
||||
typedef Struct_(RegFile) {
|
||||
A2_U2 GPR;
|
||||
A2_U2 GTE;
|
||||
};
|
||||
typedef Struct_(RegFile) { A2_U2 GPR; };
|
||||
#define regfile(pin_mask) {.GPR={u4_lo(pin_mask), u4_hi(pin_mask)} }
|
||||
FI_ void regfile_init(RegFile_R rf) {
|
||||
/* pack the 32-bit ABI mask into the two U2s */
|
||||
rf->GPR[0] = u4_lo(regfile_abi_mask);
|
||||
rf->GPR[1] = u4_hi(regfile_abi_mask);
|
||||
rf->GTE[0] = rf->GTE[1] = 0;
|
||||
rf->GPR[0] = u4_lo(regfile_abi_mask); rf->GPR[1] = u4_hi(regfile_abi_mask);
|
||||
}
|
||||
FI_ RegFile regfile_make(void) { RegFile rf; regfile_init(& rf); return rf; }
|
||||
|
||||
typedef Struct_(RegFile_RInfo) {
|
||||
U2_R section;
|
||||
U2 mask;
|
||||
B2 occupied;
|
||||
};
|
||||
typedef Struct_(RegFile_RInfo) { U2_R section; U2 mask; B2 occupied; };
|
||||
FI_ RegFile_RInfo regfile_rinfo(A2_U2 file, Reg r_id) {
|
||||
U2 s_id = r_id >> 4;
|
||||
U2_R section = & file[s_id];
|
||||
@@ -371,19 +349,11 @@ FI_ RegFile_RInfo regfile_rinfo(A2_U2 file, Reg r_id) {
|
||||
}
|
||||
FI_ Reg regfile__alloc_helper(A2_U2 file, Reg r_id) {
|
||||
Reg result = 0; RegFile_RInfo info = regfile_rinfo(file, r_id);
|
||||
if (info.occupied == false) {
|
||||
info.section[0] |= info.mask;
|
||||
result = r_id;
|
||||
}
|
||||
if (info.occupied == false) { info.section[0] |= info.mask; result = r_id; }
|
||||
return result;
|
||||
}
|
||||
/* regfile_alloc picks the next free GPR from regfile_alloc_order.
|
||||
* The table is the first-fit allocation order: T0..T7, V0..V1, A0..A3,
|
||||
* S0..S7, T8..T9. The 24 entries leave room for the tape program to use
|
||||
* any of them while R0, R1, R26-R31 remain reserved. */
|
||||
I_ Reg regfile_alloc(RegFile_R rf) {
|
||||
Reg allocated = 0;
|
||||
for index_iter(U4, r_id, R_V0, <, R_T9) {
|
||||
I_ Reg regfile_alloc(RegFile_R rf) {
|
||||
Reg allocated = 0; for index_iter(U4, r_id, R_V0, <, R_T9) {
|
||||
allocated = regfile__alloc_helper(rf->GPR, r_id);
|
||||
Jmp_nZero_(allocated,resolved);
|
||||
}
|
||||
@@ -392,13 +362,12 @@ resolved: return allocated;
|
||||
}
|
||||
FI_ Reg regfile_pin(RegFile_R rf, Reg r_id) {
|
||||
RegFile_RInfo info = regfile_rinfo(rf->GPR, r_id);
|
||||
assert(info.occupied == false);
|
||||
assert(info.occupied == false);
|
||||
info.section[0] |= info.mask;
|
||||
return r_id;
|
||||
}
|
||||
FI_ void regfile_pin_mask(RegFile_R rf, U4 mask) {
|
||||
B4 occupied = u4_r(rf->GPR)[0] & mask;
|
||||
assert(occupied == false);
|
||||
B4 occupied = u4_r(rf->GPR)[0] & mask; assert(occupied == false);
|
||||
u4_r(rf->GPR)[0] |= mask;
|
||||
}
|
||||
FI_ void regfile_free_mask(RegFile_R rf, U4 mask) {
|
||||
@@ -406,21 +375,23 @@ FI_ void regfile_free_mask(RegFile_R rf, U4 mask) {
|
||||
u4_r(rf->GPR)[0] &= ~mask;
|
||||
}
|
||||
FI_ void regfile_free_reg(RegFile_R rf, Reg r_id) {
|
||||
/* never free the ABI set */
|
||||
if (regfile_abi_mask & (1u << r_id)) return;
|
||||
if (regfile_abi_mask & (1u << r_id)) return; // never free the ABI set
|
||||
RegFile_RInfo info = regfile_rinfo(rf->GPR, r_id);
|
||||
info.section[0] &= ~info.mask;
|
||||
}
|
||||
FI_ void regfile_reset(RegFile_R rf) {
|
||||
rf->GPR[0] = u4_lo(regfile_abi_mask);
|
||||
rf->GPR[1] = u4_hi(regfile_abi_mask);
|
||||
}
|
||||
FI_ void regfile_reset_to_mask(RegFile_R rf, U4 mask) {
|
||||
rf->GPR[0] = u4_lo(mask);
|
||||
rf->GPR[1] = u4_hi(mask);
|
||||
}
|
||||
FI_ void regfile_reset (RegFile_R rf) { rf->GPR[0] = u4_lo(regfile_abi_mask); rf->GPR[1] = u4_hi(regfile_abi_mask); }
|
||||
FI_ void regfile_reset_to_mask(RegFile_R rf, U4 mask) { rf->GPR[0] = u4_lo(mask); rf->GPR[1] = u4_hi(mask); }
|
||||
#pragma endregion RegFileArena (Register File Allocator)
|
||||
|
||||
#pragma region Mips Atom Components (Procedures)
|
||||
|
||||
// For doing direct-chaining of "atoms or fragments".
|
||||
FI_ Slice_MipsCode ac_yield_to(AtomBuilder_R ab, Reg code_ptr) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
jump_reg(code_ptr), BdSlot_ nop,
|
||||
})
|
||||
|
||||
#pragma endregion Mips Atom Components (Procedures)
|
||||
|
||||
#pragma region Mips Atom Procs
|
||||
/* RegUse structs are a convention to organize register allocations for a mips atom procedure.
|
||||
Unlike the usual enum-based declarations, they provide a namespaced scope and have view types via union declarations. */
|
||||
@@ -441,7 +412,6 @@ FI_ void regfile_reset_to_mask(RegFile_R rf, U4 mask) {
|
||||
add_si(r.t1.view_3, r.usual_modifiable, 10),
|
||||
mac_yield(),
|
||||
})
|
||||
|
||||
#pragma endregion Mips Atom Procs
|
||||
|
||||
#pragma region Baked Mips Atoms
|
||||
@@ -3,7 +3,7 @@
|
||||
# include "duffle/gen/macs.h"
|
||||
# include "duffle/gen/offsets.h"
|
||||
# include "duffle/dsl.atom.h"
|
||||
# include "duffle/lottes_tape.h"
|
||||
# include "duffle/tape.h"
|
||||
# include "duffle/mips.h"
|
||||
# include "duffle/gte.h"
|
||||
# include "duffle/gp.h"
|
||||
@@ -11,8 +11,8 @@
|
||||
# include "duffle/word_count.metadata.h"
|
||||
# include "duffle/psyq.h"
|
||||
# include "duffle/math.atom.h"
|
||||
# include "duffle/mips.atom.c"
|
||||
# include "duffle/gte.atom.c"
|
||||
# include "duffle/mips.atom.c"
|
||||
# include "duffle/gp.atom.c"
|
||||
# include "duffle/psyq.atom.c"
|
||||
# include "gen/offsets.h"
|
||||
@@ -21,7 +21,7 @@
|
||||
# include "hello_camera.h"
|
||||
#endif
|
||||
|
||||
ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c);
|
||||
ATOM_FILE_DEBUGGER_LINE_MARKER(hello_camera_atom_c);
|
||||
|
||||
#pragma region MACs (Mips Atom components)
|
||||
|
||||
@@ -136,7 +136,7 @@ atom_info(atom_bind(Binds_ResolveLookAtSub)) MipsAtom_Proc_(aa, {
|
||||
load_word(r.target_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,target)),
|
||||
load_word(r.eye_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)),
|
||||
load_word(r.up_in_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)),
|
||||
LdSlot_ add_ui_self(R_TapePtr, S_(Binds_ResolveLookAtSub)),
|
||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtSub)),
|
||||
|
||||
/* Stage up_in.x/y/z into the scratchpad. R_ScratchBase = R_SP = 0x1F800000. */
|
||||
mac_load_v3s4( r.up_in, r.up_in_ptr, 0), LdSlot_
|
||||
@@ -155,7 +155,7 @@ atom_info(atom_bind(Binds_ResolveLookAtSub)) MipsAtom_Proc_(aa, {
|
||||
})
|
||||
|
||||
typedef Struct_(Binds_ResolveLookAt_PopulateMT3S4S2) {
|
||||
U4 look_at; /* MT3_S2S4* — destination matrix address */
|
||||
MT3_S2S4* look_at; /* MT3_S2S4* — destination matrix address */
|
||||
};
|
||||
typedef Struct_(RegUse_resolve_look_at_populate_mt3s4s2) {
|
||||
Reg look_at;
|
||||
@@ -182,7 +182,7 @@ internal MipsAtom* AtomBundleEntry_(resolve_look_at,populate_mt3s4s2)(AtomArena_
|
||||
atom_info(atom_bind(Binds_ResolveLookAt_PopulateMT3S4S2)) MipsAtom_Proc_(aa, {
|
||||
/* --- Tape pop: look_at pointer --- */
|
||||
load_word(r.look_at, R_TapePtr, O_(Binds_ResolveLookAt_PopulateMT3S4S2,look_at)),
|
||||
LdSlot_ add_ui_self(R_TapePtr, S_(Binds_ResolveLookAt_PopulateMT3S4S2)),
|
||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_ResolveLookAt_PopulateMT3S4S2)),
|
||||
|
||||
add_si(r.ux, R_ScratchBase, O_(ResolveLookAtScratch, ux)), LdSlot_
|
||||
add_si(r.uy, R_ScratchBase, O_(ResolveLookAtScratch, uy)),
|
||||
@@ -237,8 +237,8 @@ enum {
|
||||
};
|
||||
//screen_env_init. Mirrors the libpsyx's SetDefDispEnv + SetDefDrawEnv + the manual enable_auto_clear / initial_bg_color writes.
|
||||
internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
|
||||
, atom_reads(R_T0, R_ScreenX, R_ScreenY, R_ScreenBuf)
|
||||
, atom_writes(R_T0, R_ScreenX, R_ScreenY)
|
||||
, atom_reads(R_ScreenBuf)
|
||||
, atom_writes(R_ScreenBuf)
|
||||
) {
|
||||
/* display[0] = (0, 0, 320, 240); rest of struct zeroed. */
|
||||
add_ui(R_ScreenX, R_0, ScreenRes_X), add_ui(R_ScreenY, R_0, ScreenRes_Y),
|
||||
@@ -288,23 +288,6 @@ internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
|
||||
mac_yield(),
|
||||
};
|
||||
|
||||
/* gp_screen_init's GPR setup. Tests the mixed user-pinning + auto-reg pattern:
|
||||
* - R_IO_BaseAddr = R_T4 (user-pinned via atom_reg; pre-existing)
|
||||
* - R_GP1_Offset = R_T2 (user-pinned via atom_reg; NEW -- for GPIO_PORT1_OFFSET)
|
||||
* - R_ScreenX = R_T5 (user-pinned via atom_reg; used as a transfer and GTE setup reg)
|
||||
* - R_GpTmp = auto-allocated by the lua pass and used for several GPU transfers;
|
||||
* the C preprocessor resolves it to the chosen free pool GPR.
|
||||
*
|
||||
* For gp_screen_init, the auto-reg pool exclusions are:
|
||||
* user_pinned (from the corpus register_alias_registry) : R_T0..R_T7 (all 8 user-pinned across hello_camera.atom.c)
|
||||
* body-parsed physical registers : aliases resolve through the registry;
|
||||
* the body uses R_ScreenX, not raw R_T5
|
||||
* source_pool after both subtractions : {R_V0, R_V1} only
|
||||
* R_GpTmp gets R_V0 (the first-fit choice). Its repeated GPU-transfer use proves that the
|
||||
* auto-reg allocation is active while the R_ScreenX references prove the pinned alias is used.
|
||||
* R_TapePtr (R_T9), R_AtomJmp (R_T8), R_AT are excluded from the POOL by construction in
|
||||
* passes/auto_reg.lua -- see the "obvious exclusions" comment block at the top of that file.
|
||||
*/
|
||||
enum {
|
||||
R_IO_BaseAddr = R_T4 atom_reg, /* Caller-pinned: IO_BASE_ADDR = 0x1F800000 */
|
||||
R_GP1_Offset = R_T2 atom_reg, /* Caller-pinned: GPIO_PORT1_OFFSET = 0x10 */
|
||||
@@ -458,7 +441,7 @@ typedef Struct_(Binds_PadInputCam) {
|
||||
Camera* cam;
|
||||
};
|
||||
internal MipsAtom_(pad_input_cam) atom_info(atom_bind(Binds_PadInputCam)
|
||||
, atom_reads( R_Cam, R_CamPadState, R_TapePtr)
|
||||
, atom_reads( R_Cam, R_CamPadState)
|
||||
, atom_writes(R_Cam)
|
||||
) {
|
||||
/* Bind pop: state → R_CamPadState (R_T5), cam → R_Cam (R_T4), advance R_TapePtr by 8. */
|
||||
@@ -516,20 +499,17 @@ enum {
|
||||
#define R_OtBase_Code R_T6_Code
|
||||
};
|
||||
typedef Struct_(Binds_CubeTri) {
|
||||
U4 PrimCursor;
|
||||
V4_S2* FaceCursor;
|
||||
V3_S2* VertBase;
|
||||
U4* OtBase;
|
||||
U1* prim_cursor;
|
||||
V4_S2* face_cursor;
|
||||
V3_S2* vert_base;
|
||||
U4* ot_base;
|
||||
};
|
||||
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)
|
||||
, atom_reads(R_TapePtr)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
|
||||
){
|
||||
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)){
|
||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,prim_cursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,face_cursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,vert_base)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,ot_base)),
|
||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
||||
mac_yield()
|
||||
};
|
||||
@@ -545,11 +525,13 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
||||
// load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
||||
|
||||
LdSlot_ mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2), GteDelay_ load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)), LdSlot_
|
||||
GteDelay_ nop, gte_cmdw_rotate_translate_perspective_triple,
|
||||
LdSlot_ mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
||||
GteDelay_ load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)), LdSlot_
|
||||
GteDelay_ load_word(R_AtomJmp, R_TapePtr, 0), LdSlot_ //ac_yield: word 1,
|
||||
gte_cmdw_rotate_translate_perspective_triple,
|
||||
gte_cmdw_nclip,
|
||||
|
||||
gte_mv_from_data_r(R_T0, C2_MAC0), GteDelay_ load_word(R_AtomJmp, R_TapePtr, 0), // ac_yield: word 1
|
||||
gte_mv_from_data_r(R_T0, C2_MAC0), GteDelay_ add_ui_self(R_TapePtr, S_(MipsCode)), // ac_yield: word 2
|
||||
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
|
||||
/* BD-slot: Write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
|
||||
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
|
||||
@@ -568,7 +550,7 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), BdSlot_ add_ui_self(R_TapePtr, S_(MipsCode)), // ac_yield: word 2
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), BdSlot_ nop,
|
||||
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_G4)),
|
||||
mac_format_g4_color(R_PrimCursor,
|
||||
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
||||
@@ -579,16 +561,16 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||
// end: branch(cull)
|
||||
|
||||
atom_label(cube_g4_face_exit)
|
||||
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
|
||||
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
|
||||
jump_reg(R_TapePtr), BdSlot_ nop // ac_yield: word 3-4
|
||||
add_ui_self(R_PrimCursor, S_(Poly_G4)), // 9 words = Poly_G4
|
||||
add_ui_self(R_FaceCursor, S_(S2) * 4), // 4 × S2 = 8 bytes
|
||||
jump_reg(R_AtomJmp), BdSlot_ nop // ac_yield: word 3-4
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_FloorTri) {
|
||||
U4 PrimCursor;
|
||||
V3_S2* FaceCursor;
|
||||
V3_S2* VertBase;
|
||||
U4* OtBase;
|
||||
U1* prim_cursor;
|
||||
V3_S2* face_cursor;
|
||||
V3_S2* vert_base;
|
||||
U4* ot_base;
|
||||
};
|
||||
internal
|
||||
MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(floor_f3)
|
||||
@@ -596,10 +578,10 @@ MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(f
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
|
||||
){
|
||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,prim_cursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,face_cursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,vert_base)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,ot_base)),
|
||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
||||
mac_yield()
|
||||
};
|
||||
@@ -638,13 +620,12 @@ MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
||||
/* Advance Input Cursor & Yield (Both branch targets land here) */
|
||||
atom_label(floor_f3_face_exit)
|
||||
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
|
||||
jump_reg(R_TapePtr), BdSlot_ nop // ac_yield: word 3-4
|
||||
jump_reg(R_AtomJmp), BdSlot_ nop // ac_yield: word 3-4
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
|
||||
typedef Struct_(Binds_SyncPrimitiveArena) { U4* used; U1* cursor; };
|
||||
internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitiveArena)
|
||||
, atom_reads( R_TapePtr, R_PrimCursor)
|
||||
, atom_writes(R_TapePtr)
|
||||
, atom_reads(R_PrimCursor), atom_writes(R_AT)
|
||||
){
|
||||
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
|
||||
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)), LdSlot_
|
||||
|
||||
+100
-126
@@ -24,7 +24,7 @@
|
||||
#include "duffle/pad.h"
|
||||
|
||||
#include "duffle/dsl.atom.h"
|
||||
#include "duffle/lottes_tape.h"
|
||||
#include "duffle/tape.h"
|
||||
|
||||
#include "duffle/bios.h"
|
||||
#include "duffle/psyq.h"
|
||||
@@ -80,9 +80,6 @@ typedef Struct_(SMemory) {
|
||||
PadBiosRaw pad_raw[2];
|
||||
PadState pad[2];
|
||||
|
||||
// TODO(Ed): We don't need this we can just cast at any point an address to a desired view of scratchpad, we have the address.
|
||||
U4_V scratchpad; // d-cache
|
||||
|
||||
U1 ct_init_atom_mem[CT_InitAtomMem_Size];
|
||||
MipsAtom* normalize_v3s4;
|
||||
MipsAtom* gte_cross_v3s4;
|
||||
@@ -96,11 +93,11 @@ extern SMemory smem;
|
||||
#define pad0_btn_(btn) btn & smem.pad[0].buttons
|
||||
#define pad1_btn_(btn) btn & smem.pad[1].buttons
|
||||
|
||||
I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
|
||||
I_ U1* prim__alloc(U4 type_width, Str8 type_name) {
|
||||
gknown PrimitiveArena* pa = & smem.primitives;
|
||||
gknown B1* buf = (B1*) r_(smem.primitives.buf)[smem.active_buf_id];
|
||||
gknown U1* buf = (U1*) r_(smem.primitives.buf)[smem.active_buf_id];
|
||||
assert(pa->used + type_width < PrimitiveBuff_Len);
|
||||
B1* next = buf + pa->used;
|
||||
U1* next = buf + pa->used;
|
||||
pa->used += type_width;
|
||||
return next;
|
||||
}
|
||||
@@ -114,10 +111,10 @@ I_ void resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4*
|
||||
V3_S4 pos, off;
|
||||
|
||||
forward = target[0]; sub_v3s4(& forward, eye[0]); // RGA(Lengyel): Affine point - point = zero-weight direction.
|
||||
normalize_v3s4(& forward, & uz); // RGA(Lengyel): Normalize the direction bulk. Not finite-point unitization.
|
||||
psy_normalize_v3s4(& forward, & uz); // RGA(Lengyel): Normalize the direction bulk. Not finite-point unitization.
|
||||
|
||||
cross_v3s4(& uz, up_in, & right); normalize_v3s4(& right, & ux); // RGA(Lengyel): Complement(Wedge(forward, up_in)) -> right axis.
|
||||
cross_v3s4(& uz, & ux, & up); normalize_v3s4(& up, & uy); // RGA(Lengyel): Complement(Wedge(forward, right)) -> up axis.
|
||||
cross_v3s4(& uz, up_in, & right); psy_normalize_v3s4(& right, & ux); // RGA(Lengyel): Complement(Wedge(forward, up_in)) -> right axis.
|
||||
cross_v3s4(& uz, & ux, & up); psy_normalize_v3s4(& up, & uy); // RGA(Lengyel): Complement(Wedge(forward, right)) -> up axis.
|
||||
|
||||
// RGA(Lengyel): matrix expansion of the world-to-camera rotation (basis rows).
|
||||
look_at->m[0][0] = ux.x; look_at->m[0][1] = ux.y; look_at->m[0][2] = ux.z;
|
||||
@@ -141,16 +138,16 @@ internal void compile_init_atoms(void) {
|
||||
|
||||
smem.gte_cross_v3s4 = gte_cross_v3s4(& ab,
|
||||
RegUse_(gte_cross_v3s4) {
|
||||
.a = ralloc_v3(),
|
||||
.b = ralloc_v3(),
|
||||
.x = ralloc(),
|
||||
.y = ralloc(),
|
||||
.z = ralloc(),
|
||||
.a = ralloc_v3(),
|
||||
.b = ralloc_v3(),
|
||||
.out = ralloc(),
|
||||
.src_a = ralloc(),
|
||||
.src_b = ralloc(),
|
||||
});
|
||||
regfile_reset(& rf);
|
||||
|
||||
smem.normalize_v3s4 = build_normalize_v3s4(& ab,
|
||||
RegUse_(build_normalize_v3s4) {
|
||||
smem.normalize_v3s4 = normalize_v3s4(& ab,
|
||||
RegUse_(normalize_v3s4) {
|
||||
.res = ralloc_v3(),
|
||||
.r0 = ralloc(),
|
||||
.r1 = ralloc(),
|
||||
@@ -167,15 +164,11 @@ internal void compile_init_atoms(void) {
|
||||
}
|
||||
|
||||
internal void compile_resolve_look_at(void) {
|
||||
AtomArena ab = atomarena_make(slice_ut_arr(smem.resolve_look_at_mem));
|
||||
AtomBundle_resolve_look_at_R bundle = C_(void*, smem.resolve_look_at_bundle);
|
||||
|
||||
/* R_ScratchBase (= R_SP) is a tape carrier preserved across atoms; no carrier
|
||||
* pin is needed in the regfile. The standard 24-register pool is sufficient. */
|
||||
RegFile rf = regfile(regfile_abi_mask);
|
||||
AtomArena ab = atomarena_make(slice_ut_arr(smem.resolve_look_at_mem));
|
||||
RegFile rf = regfile(regfile_abi_mask);
|
||||
#define ralloc() regfile_alloc(& rf)
|
||||
#define ralloc_v3() { ralloc(), ralloc(), ralloc() }
|
||||
|
||||
bundle->input_and_sub = AtomBundleEntry_(resolve_look_at, input_and_sub)(& ab,
|
||||
RegUse_(resolve_look_at_input_and_sub) {
|
||||
.target_ptr = ralloc(),
|
||||
@@ -203,78 +196,63 @@ internal void compile_resolve_look_at(void) {
|
||||
.r2 = ralloc(),
|
||||
});
|
||||
|
||||
/* Sanity check: arena didn't overflow. */
|
||||
assert(ab.used <= ResolveLookAtArena_Size);
|
||||
assert(ab.used <= ResolveLookAtArena_Size); // Sanity check: arena didn't overflow.
|
||||
#undef ralloc
|
||||
}
|
||||
|
||||
/* Emit the resolve_look_at bundle into the tape. Called once per frame from update(). */
|
||||
I_ void resolve_look_at(TapeBuilder_R tb
|
||||
, MT3_S2S4* look_at
|
||||
, P3_S4* eye
|
||||
, P3_S4* target
|
||||
, V3_S4* up_in
|
||||
){
|
||||
// Emit the resolve_look_at bundle into the tape. Called once per frame from update().
|
||||
I_ void resolve_look_at(TapeBuilder_R tb, MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4* up_in) {
|
||||
/* Typed view of the scratchpad for field-address arithmetic. */
|
||||
ResolveLookAtScratch* sp = C_scratch(ResolveLookAtScratch*);
|
||||
ResolveLookAtScratch* sp = C_scratch(ResolveLookAtScratch*);
|
||||
AtomBundle_resolve_look_at_R bundle = C_(void*, smem.resolve_look_at_bundle);
|
||||
|
||||
tb_emit(tb, bundle->input_and_sub); {
|
||||
tb_data(tb, u4_(target));
|
||||
tb_data(tb, u4_(eye));
|
||||
tb_data(tb, u4_(up_in));
|
||||
}
|
||||
tb_emit(tb, bundle->normalize_fwd_uz); {
|
||||
tb_data(tb, u4_(O_(ResolveLookAtScratch, fwd) | (O_(ResolveLookAtScratch, uz) << 16)));
|
||||
}
|
||||
tb_emit(tb, bundle->cross_to_right); {
|
||||
tb_data(tb, u4_(& sp->uz));
|
||||
tb_data(tb, u4_(& sp->up_in));
|
||||
tb_data(tb, u4_(& sp->right));
|
||||
}
|
||||
tb_emit(tb, bundle->normalize_right_ux); {
|
||||
tb_data(tb, u4_(O_(ResolveLookAtScratch, right) | (O_(ResolveLookAtScratch, ux) << 16)));
|
||||
}
|
||||
tb_emit(tb, bundle->cross_to_up); {
|
||||
tb_data(tb, u4_(& sp->uz));
|
||||
tb_data(tb, u4_(& sp->ux));
|
||||
tb_data(tb, u4_(& sp->up));
|
||||
}
|
||||
tb_emit(tb, bundle->normalize_up_uy); {
|
||||
tb_data(tb, u4_(O_(ResolveLookAtScratch, up) | (O_(ResolveLookAtScratch, uy) << 16)));
|
||||
}
|
||||
tb_emit(tb, bundle->populate_mt3s4s2); {
|
||||
tb_data(tb, u4_(look_at));
|
||||
}
|
||||
tb_emit(tb, bundle->input_and_sub); tb_bind_(tb, Binds_ResolveLookAtSub,
|
||||
.target = target,
|
||||
.eye = eye,
|
||||
.up_in = up_in,
|
||||
);
|
||||
tb_emit(tb, bundle->normalize_fwd_uz); tb_bind_(tb, Binds_normalize_v3s4,
|
||||
.src_offset = O_(ResolveLookAtScratch,fwd),
|
||||
.dst_offset = O_(ResolveLookAtScratch,uz),
|
||||
);
|
||||
tb_emit(tb, bundle->cross_to_right); tb_bind_(tb, Binds_gte_cross_v3s4,
|
||||
.src_a = & sp->uz,
|
||||
.src_b = & sp->up_in,
|
||||
.out = & sp->right,
|
||||
);
|
||||
tb_emit(tb, bundle->normalize_right_ux); tb_bind_(tb, Binds_normalize_v3s4,
|
||||
.src_offset = O_(ResolveLookAtScratch,right),
|
||||
.dst_offset = O_(ResolveLookAtScratch,ux),
|
||||
);
|
||||
tb_emit(tb, bundle->cross_to_up); tb_bind_(tb, Binds_gte_cross_v3s4,
|
||||
.src_a = & sp->uz,
|
||||
.src_b = & sp->ux,
|
||||
.out = & sp->up,
|
||||
);
|
||||
tb_emit(tb, bundle->normalize_up_uy); tb_bind_(tb, Binds_normalize_v3s4,
|
||||
.src_offset = O_(ResolveLookAtScratch,up),
|
||||
.dst_offset = O_(ResolveLookAtScratch,uy),
|
||||
);
|
||||
tb_emit(tb, bundle->populate_mt3s4s2); tb_bind_(tb, Binds_ResolveLookAt_PopulateMT3S4S2,
|
||||
.look_at = look_at,
|
||||
);
|
||||
}
|
||||
FI_ void camera_look_at(TapeBuilder_R tb, Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at(tb, & c->look_at, & c->pos, target, up_in); }
|
||||
|
||||
GCC_OPTIMIZATION_DISABLE
|
||||
void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
{
|
||||
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
|
||||
|
||||
// Pad Input
|
||||
{
|
||||
/*Pad Input*/ {
|
||||
tb.used = 0; tb_scope_run(& tb) {
|
||||
// Grab latest state from bios.
|
||||
tb_emit_(pad_bios_snapshot);
|
||||
tb_data_(raw, & smem.pad_raw[0]);
|
||||
tb_data_(state, & smem.pad[0]);
|
||||
// tb_emit_(pad_bios_snapshot);
|
||||
// tb_data_(raw, & smem.pad_raw[1]);
|
||||
// tb_data_(state, & smem.pad[1]);
|
||||
|
||||
tb_data(& tb, u4_(& smem.pad_raw[0]));
|
||||
tb_data(& tb, u4_(& smem.pad[0]));
|
||||
tb_emit_(pad_input_cam);
|
||||
tb_data_(state, & smem.pad[0]);
|
||||
tb_data_(cam, & smem.cam);
|
||||
|
||||
// tb_emit_(pad_input_cube_rotation);
|
||||
// tb_data_(state, & smem.pad[0]);
|
||||
// tb_data_(cube_rot, & smem.cube.rot);
|
||||
// tb_data_(floor_rot, & smem.floor.rot);
|
||||
tb_data(& tb, u4_(& smem.pad[0]));
|
||||
tb_data(& tb, u4_(& smem.cam));
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
orderingtbl_clear_reverse(ordering_buf, OrderingTbl_Len);
|
||||
|
||||
// Update the position based on acceleration and velocity
|
||||
@@ -299,7 +277,7 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
if (use_c11_path == false)
|
||||
{
|
||||
tb.used = 0; tb_scope_run(& tb) {
|
||||
resolve_look_at(& tb, & smem.cam.look_at, & smem.cam.pos, & smem.cube.pos, & v3s4(0, -fp_one, 0));
|
||||
camera_look_at(& tb, & smem.cam, & smem.cube.pos, & v3s4(0, -fp_one, 0));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -309,33 +287,30 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
mt3s2s4_rotation (& smem.cube.rot, & smem.tform_world);
|
||||
mt3s2s4_translation(& smem.tform_world, & smem.cube.pos);
|
||||
mt3s2s4_scale (& smem.tform_world, & smem.cube.scale);
|
||||
|
||||
// Combine world and look_at matrix.
|
||||
gte_comp_coord_m3s2(& smem.cam.look_at, & smem.tform_world, & smem.tform_view);
|
||||
gte_matrix_set_rotation (& smem.tform_view);
|
||||
gte_matrix_set_translation(& smem.tform_view);
|
||||
|
||||
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||
U4 prim_cursor = prim_base + pa->used;
|
||||
|
||||
tb.used = 0; tb_scope(& tb) {
|
||||
tb_emit(& tb, rbind_cube_g4_face);
|
||||
tb_data(& tb, prim_cursor);
|
||||
tb_data(& tb, u4_(smem.cube.faces));
|
||||
tb_data(& tb, u4_(smem.cube.verts));
|
||||
tb_data(& tb, u4_(ordering_buf));
|
||||
|
||||
// TODO(Ed): We should do a bounds check beforehand to confirm pa can hold all tris?
|
||||
// The tape atoms in-flight should not need to care.
|
||||
U1* prim_base = u1_r(pa->buf[smem.active_buf_id]);
|
||||
U1* prim_cursor = prim_base + pa->used;
|
||||
tb.used = 0; tb_scope_run(& tb) {
|
||||
tb_emit(& tb, rbind_cube_g4_face); tb_bind_(& tb, Binds_CubeTri,
|
||||
.prim_cursor = prim_cursor,
|
||||
.face_cursor = smem.cube.faces,
|
||||
.vert_base = smem.cube.verts,
|
||||
.ot_base = ordering_buf,
|
||||
);
|
||||
for (U4 i = 0; i < Cube_num_faces; i++) {
|
||||
// Two triangles per quad face: (x,y,z) and (x,z,w)
|
||||
tb_emit(& tb, cube_g4_face);
|
||||
tb_emit(& tb, cube_g4_face); // Two triangles per quad face: (x,y,z) and (x,z,w)
|
||||
}
|
||||
|
||||
tb_emit(& tb, sync_primitive_arena);
|
||||
tb_data(& tb, u4_(& pa->used));
|
||||
tb_data(& tb, prim_base);
|
||||
tb_emit(& tb, sync_primitive_arena); tb_bind_(& tb, Binds_SyncPrimitiveArena,
|
||||
.used = & pa->used,
|
||||
.cursor = prim_base,
|
||||
);
|
||||
}
|
||||
tape_run(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
|
||||
|
||||
// smem.cube.rot.y += 30;
|
||||
}
|
||||
// Draw floor
|
||||
@@ -344,45 +319,35 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
mt3s2s4_rotation (& smem.floor.rot, & smem.tform_world);
|
||||
mt3s2s4_translation(& smem.tform_world, & smem.floor.pos);
|
||||
mt3s2s4_scale (& smem.tform_world, & smem.floor.scale);
|
||||
|
||||
// Combine world and look_at matrix.
|
||||
gte_comp_coord_m3s2(& smem.cam.look_at, & smem.tform_world, & smem.tform_view);
|
||||
|
||||
gte_matrix_set_rotation (& smem.tform_view);
|
||||
gte_matrix_set_translation(& smem.tform_view);
|
||||
|
||||
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||
U4 prim_cursor = prim_base + pa->used;
|
||||
|
||||
// TODO(Ed): We should do a bounds check beforehand to confirm pa can hold all tris?
|
||||
// The tape atoms in-flight should not need to care.
|
||||
|
||||
// Prepare the tape. (Push protocol to tape)
|
||||
tb.used = 0; tb_scope(& tb) {
|
||||
// tb_emit(& tb, set_gte_mt3s2s4);
|
||||
// tb_data(& tb, u4_(& smem.tform_view));
|
||||
|
||||
tb_emit(& tb, rbind_floor_f3_face);
|
||||
// TODO(Ed): Just use a single context struct ref?
|
||||
tb_data(& tb, prim_cursor);
|
||||
tb_data(& tb, u4_(smem.floor.faces));
|
||||
tb_data(& tb, u4_(smem.floor.verts));
|
||||
tb_data(& tb, u4_(ordering_buf));
|
||||
U1_R prim_base = u1_r(pa->buf[smem.active_buf_id]);
|
||||
U1_R prim_cursor = prim_base + pa->used;
|
||||
tb.used = 0; tb_scope_run(& tb) { // Prepare the tape. (Push protocol to tape)
|
||||
tb_emit(& tb, rbind_floor_f3_face); tb_bind_(& tb, Binds_FloorTri,
|
||||
.prim_cursor = prim_cursor,
|
||||
.face_cursor = smem.floor.faces,
|
||||
.vert_base = smem.floor.verts,
|
||||
.ot_base = ordering_buf,
|
||||
);
|
||||
for (U4 i = 0; i < Floor_num_faces; i++) {
|
||||
tb_emit(& tb, floor_f3_face);
|
||||
}
|
||||
// After floor_f3_face iterations complete, the primitive arena's used counter needs updating.
|
||||
tb_emit(& tb, sync_primitive_arena);
|
||||
tb_data(& tb, u4_(& pa->used));
|
||||
tb_data(& tb, prim_base);
|
||||
tb_emit(& tb, sync_primitive_arena); tb_bind_(& tb, Binds_SyncPrimitiveArena,
|
||||
.used = & pa->used,
|
||||
.cursor = prim_base,
|
||||
);
|
||||
}
|
||||
tape_run(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
|
||||
|
||||
// C-side state (pa->used) has already been updated by the tape!
|
||||
// smem.floor.rot.y += 5;
|
||||
}
|
||||
}
|
||||
GCC_OPTIMIZATION_ENABLE
|
||||
|
||||
void render(void) {
|
||||
}
|
||||
@@ -399,12 +364,22 @@ void gp_display_frame(DoubleBuffer* screen_buf, S4* active_buf_id, U4* ordering_
|
||||
active_buf_id[0] = ! active_buf_id[0]; // Swap current buffer
|
||||
}
|
||||
|
||||
GCC_OPTIMIZATION_DISABLE
|
||||
int main(void)
|
||||
{
|
||||
smem = (SMemory){0};
|
||||
// TODO(Ed): remove this field we don't need it in smem.
|
||||
smem.scratchpad = C_(U4_V, Scratchpad_Loc);
|
||||
|
||||
B4 basic_sample = false; if (basic_sample) {
|
||||
// We will be defining the tape here along with its atom, then running the tape after.
|
||||
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
|
||||
MipsCode add_one_to_R_T1[] = {
|
||||
add_ui_self(R_T1, 1),
|
||||
mac_yield(),
|
||||
};
|
||||
tb_emit(& tb, C_(MipsAtom*, add_one_to_R_T1));
|
||||
Tape tape = tb_end(& tb);
|
||||
tape_run(tape);
|
||||
}
|
||||
|
||||
// smem.primitives.used = 0;
|
||||
// smem.active_buf_id = 0;
|
||||
smem.cam.pos = v3s4(500, -1000, -1500);
|
||||
@@ -449,4 +424,3 @@ int main(void)
|
||||
};
|
||||
return 0;
|
||||
}
|
||||
GCC_OPTIMIZATION_ENABLE
|
||||
|
||||
@@ -24,7 +24,7 @@ enum {
|
||||
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
||||
typedef Array_(OrderingTable_Buffer, 2);
|
||||
|
||||
typedef B1 PrimitiveBuffer[PrimitiveBuff_Len];
|
||||
typedef U1 PrimitiveBuffer[PrimitiveBuff_Len];
|
||||
typedef Array_(PrimitiveBuffer, 2);
|
||||
typedef Struct_(PrimitiveArena) {
|
||||
A2_PrimitiveBuffer buf;
|
||||
@@ -54,14 +54,14 @@ I_ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
||||
{ 2, 3, 6, 7 },
|
||||
{ 3, 0, 7, 4 },
|
||||
};
|
||||
mem_copy(u4_(verts), u4_(& baked_verts), S_(A8_V3_S2) );
|
||||
mem_copy(u4_(faces), u4_(& baked_faces), S_(A6_V4_S2) );
|
||||
mem_copy(b1_r(verts), b1_r(& baked_verts), S_(A8_V3_S2) );
|
||||
mem_copy(b1_r(faces), b1_r(& baked_faces), S_(A6_V4_S2) );
|
||||
return;
|
||||
}
|
||||
typedef Struct_(Ent_Cube) {
|
||||
V3_S4 accel;
|
||||
V3_S4 vel;
|
||||
V3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
|
||||
V3_S4 pos;
|
||||
V3_S4 scale;
|
||||
V3_S2 rot;
|
||||
A8_V3_S2 verts;
|
||||
@@ -83,12 +83,12 @@ I_ void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
|
||||
{ 0, 1, 2 },
|
||||
{ 1, 3, 2 },
|
||||
};
|
||||
mem_copy(u4_(verts), u4_(& baked_verts), S_(A4_V3_S2));
|
||||
mem_copy(u4_(faces), u4_(& baked_faces), S_(A2_V3_S2));
|
||||
mem_copy(b1_r(verts), b1_r(& baked_verts), S_(A4_V3_S2));
|
||||
mem_copy(b1_r(faces), b1_r(& baked_faces), S_(A2_V3_S2));
|
||||
};
|
||||
typedef Struct_(Ent_Floor) {
|
||||
V3_S4 accel;
|
||||
V3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
|
||||
V3_S4 pos;
|
||||
V3_S4 scale;
|
||||
V3_S2 rot;
|
||||
A4_V3_S2 verts;
|
||||
@@ -96,7 +96,7 @@ typedef Struct_(Ent_Floor) {
|
||||
};
|
||||
|
||||
typedef Struct_(Camera) {
|
||||
P3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
|
||||
P3_S4 pos;
|
||||
V3_S2 rot;
|
||||
MT3_S2S4 look_at;
|
||||
};
|
||||
|
||||
@@ -1041,6 +1041,16 @@ function M.find_atom_proc_decl_for(source, before_pos, mips_atom_ptr_len)
|
||||
if source:sub(next_pos, next_pos) == "(" then
|
||||
local inner, after_paren = M.read_parens(source, next_pos) ---@type string|nil, integer
|
||||
if inner then
|
||||
if ident == "AtomBundleEntry_" then
|
||||
local tmpl = M.split_top_level_commas(inner) ---@type string[]
|
||||
if #tmpl ~= 2 then return nil end
|
||||
local name = M.trim(tmpl[1]) .. "_" .. M.trim(tmpl[2]) ---@type string
|
||||
local formals_pos = M.skip_ws_and_cmt(source, after_paren) ---@type integer
|
||||
if source:sub(formals_pos, formals_pos) ~= "(" then return nil end
|
||||
local real_inner, after_real = M.read_parens(source, formals_pos) ---@type string|nil, integer
|
||||
if not real_inner then return nil end
|
||||
return name, real_inner, name, after_real
|
||||
end
|
||||
return ident, inner, ident, after_paren
|
||||
end
|
||||
end
|
||||
|
||||
@@ -544,6 +544,50 @@ local function render_section_binds(add, view)
|
||||
if not wrote then add("_(none)_"); add("") end
|
||||
end
|
||||
|
||||
--- @param add fun(s: string): nil
|
||||
--- @param view ModuleView
|
||||
--- @return nil
|
||||
local function render_section_atom_bundles(add, view)
|
||||
local bundles = (view.corpus and view.corpus.atom_bundles) or {} ---@type table<string, AtomBundle>
|
||||
local names = {} ---@type string[]
|
||||
for name, bundle in pairs(bundles) do ---@type string, AtomBundle
|
||||
if path_in_module(bundle.path, view) then
|
||||
names[#names + 1] = name
|
||||
end
|
||||
end
|
||||
if #names == 0 then add("_(none)_"); add(""); return end
|
||||
table.sort(names)
|
||||
for _, name in ipairs(names) do ---@type integer, string
|
||||
local bundle = bundles[name] ---@type AtomBundle
|
||||
local entries = bundle.entries or {} ---@type table<string, string>
|
||||
add(string.format("### %s", name))
|
||||
for _, slot in ipairs(bundle.slots or {}) do ---@type integer, string
|
||||
local ident = entries[slot] ---@type string|nil
|
||||
if ident then
|
||||
add(string.format("- `%s` `%s`", slot, ident))
|
||||
else
|
||||
add(string.format("- `%s`", slot))
|
||||
end
|
||||
end
|
||||
add("")
|
||||
end
|
||||
end
|
||||
|
||||
--- @param add fun(s: string): nil
|
||||
--- @param view ModuleView
|
||||
--- @return nil
|
||||
local function render_section_tape_emits(add, view)
|
||||
local wrote = false ---@type boolean
|
||||
for _, src in ipairs(view.sources or {}) do ---@type integer, SourceFile
|
||||
for _, emit in ipairs((src.scan and src.scan.tape_emits) or {}) do ---@type integer, TapeEmit
|
||||
wrote = true
|
||||
add(string.format("- `%s` `%s`", emit.name or "?", emit.binds or "—"))
|
||||
end
|
||||
end
|
||||
if not wrote then add("_(none)_"); add(""); return end
|
||||
add("")
|
||||
end
|
||||
|
||||
--- @param add fun(s: string): nil
|
||||
--- @param view ModuleView
|
||||
--- @return nil
|
||||
@@ -895,6 +939,8 @@ local SECTION_RENDERERS = { ---@type SectionRenderer[]
|
||||
{ header = "## Annotations", render = render_section_annotations },
|
||||
{ header = "## Component annotations", render = render_section_component_annotations },
|
||||
{ header = "## Binds_* structs", render = render_section_binds },
|
||||
{ header = "## Atom bundles", render = render_section_atom_bundles },
|
||||
{ header = "## Tape emits", render = render_section_tape_emits },
|
||||
{ header = "## Phases / views / ctx", render = render_section_phases },
|
||||
{ header = "## Register aliases", render = render_section_aliases },
|
||||
{ header = "## Auto-reg", render = render_section_autoreg },
|
||||
|
||||
@@ -127,6 +127,13 @@ local parse_enum_int_literal ---@type fun(text: string, start: integer): (intege
|
||||
--- @field body string
|
||||
--- @field bytes integer|nil
|
||||
|
||||
--- @class AtomBundle
|
||||
--- @field name string
|
||||
--- @field slots string[] -- typedef order
|
||||
--- @field line integer
|
||||
--- @field path string
|
||||
--- @field entries table<string, string>|nil -- slot → B_E when an AtomBundleEntry_ proc exists
|
||||
|
||||
--- @class AliasEntry
|
||||
--- @field name string
|
||||
--- @field code integer
|
||||
@@ -182,6 +189,14 @@ local parse_enum_int_literal ---@type fun(text: string, start: integer): (intege
|
||||
--- @class TapeChain
|
||||
--- @field [integer] string -- ordered atom names in one tb_emit chain
|
||||
|
||||
--- @class TapeEmit
|
||||
--- @field name string
|
||||
--- @field binds string|nil
|
||||
--- @field line integer
|
||||
--- @field path string
|
||||
--- @field slot string|nil
|
||||
--- @field data_words integer|nil
|
||||
|
||||
--- @class RegUseView
|
||||
--- @field names string[]
|
||||
--- @field lanes boolean
|
||||
@@ -208,6 +223,8 @@ local parse_enum_int_literal ---@type fun(text: string, start: integer): (intege
|
||||
--- @field reg_use_schemas table<string, RegUseSchema>
|
||||
--- @field reg_use_errors RegUseError[]
|
||||
--- @field tape_chains TapeChain[]
|
||||
--- @field tape_emits TapeEmit[]
|
||||
--- @field atom_bundles table<string, AtomBundle>
|
||||
--- @field _source_file string|nil
|
||||
--- @field _code_macros table<string, integer>|nil -- bag
|
||||
--- @field _code_macro_bodies table<string, string>|nil -- bag
|
||||
@@ -2439,11 +2456,76 @@ local function parse_typedef_array(source, pos, id2_end, line_of, out, after_typ
|
||||
return semi and (semi + 1) or after_paren
|
||||
end
|
||||
|
||||
--- Parse `MipsAtom *slot, …;` from an AtomBundle_ typedef body.
|
||||
--- Ignores the MipsAtom type token. Slot name is the ident after `*`.
|
||||
--- @param body string
|
||||
--- @return string[]
|
||||
local function parse_atom_bundle_slots(body)
|
||||
local slots = {} ---@type string[]
|
||||
local pos = 1 ---@type integer
|
||||
while pos <= #body do
|
||||
pos = duffle.skip_ws_and_cmt(body, pos)
|
||||
if pos > #body then break end
|
||||
local b = body:byte(pos) ---@type integer
|
||||
if b == BYTE_SEMI then
|
||||
break
|
||||
elseif b == BYTE_COMMA then
|
||||
pos = pos + 1
|
||||
elseif b == BYTE_STAR then
|
||||
local after_star = duffle.skip_ws_and_cmt(body, pos + 1) ---@type integer
|
||||
local ident, ident_end = duffle.read_ident(body, after_star) ---@type string|nil, integer
|
||||
if ident then
|
||||
slots[#slots + 1] = ident
|
||||
pos = ident_end
|
||||
else
|
||||
pos = pos + 1
|
||||
end
|
||||
else
|
||||
local ident, ident_end = duffle.read_ident(body, pos) ---@type string|nil, integer
|
||||
if ident then
|
||||
pos = ident_end
|
||||
else
|
||||
pos = pos + 1
|
||||
end
|
||||
end
|
||||
end
|
||||
return slots
|
||||
end
|
||||
|
||||
-- Shape 5: `typedef AtomBundle_(<name>) { MipsAtom *slot, … };`
|
||||
--- @param source string
|
||||
--- @param pos integer
|
||||
--- @param id2_end integer
|
||||
--- @param line_of fun(pos: integer): integer
|
||||
--- @param out SourceScan
|
||||
--- @param after_typedef integer
|
||||
--- @return integer
|
||||
local function parse_typedef_atom_bundle(source, pos, id2_end, line_of, out, after_typedef)
|
||||
local inner, after_paren, open_paren = read_parens_after(source, id2_end, id2_end) ---@type string|nil, integer, integer
|
||||
if not inner then return id2_end end
|
||||
local name = duffle.trim(inner) ---@type string
|
||||
|
||||
local body, after_brace = find_body_braces(source, after_paren, open_paren + 1) ---@type string|nil, integer
|
||||
if not body then return after_brace end
|
||||
if name ~= "" and out.atom_bundles[name] == nil then
|
||||
local bundle = { ---@type AtomBundle
|
||||
name = name,
|
||||
slots = parse_atom_bundle_slots(body),
|
||||
line = line_of(pos),
|
||||
path = out._source_file or "",
|
||||
}
|
||||
out.atom_bundles[name] = bundle
|
||||
end
|
||||
attach_debug_skip_marker(out, "unrelated")
|
||||
return after_brace
|
||||
end
|
||||
|
||||
local TYPE_FORMS = { ---@type table<string, fun(source: string, pos: integer, id2_end: integer, line_of: fun(pos: integer): integer, out: SourceScan, after_typedef: integer): integer>
|
||||
Struct_ = parse_typedef_struct,
|
||||
Enum_ = parse_typedef_enum,
|
||||
TSet_ = parse_typedef_tset,
|
||||
Array_ = parse_typedef_array,
|
||||
Struct_ = parse_typedef_struct,
|
||||
Enum_ = parse_typedef_enum,
|
||||
TSet_ = parse_typedef_tset,
|
||||
Array_ = parse_typedef_array,
|
||||
AtomBundle_ = parse_typedef_atom_bundle,
|
||||
}
|
||||
|
||||
--- Parse: `typedef` declarations.
|
||||
@@ -2827,6 +2909,23 @@ local function parse_addrs_assign(source, pos, ident_end, line_of, out)
|
||||
return rhs
|
||||
end
|
||||
|
||||
--- @param out SourceScan
|
||||
--- @param name string
|
||||
--- @param last string
|
||||
--- @param line integer
|
||||
--- @return nil
|
||||
local function push_tape_emit(out, name, last, line)
|
||||
out.tape_emits = out.tape_emits or {}
|
||||
out.tape_emits[#out.tape_emits + 1] = { ---@type TapeEmit
|
||||
name = name,
|
||||
binds = nil,
|
||||
line = line,
|
||||
path = out._source_file or "",
|
||||
slot = last:find("->", 1, true) and name or nil,
|
||||
data_words = 0,
|
||||
}
|
||||
end
|
||||
|
||||
--- @param source string
|
||||
--- @param pos integer
|
||||
--- @param ident_end integer
|
||||
@@ -2837,10 +2936,12 @@ local function parse_tb_emit_(source, pos, ident_end, line_of, out)
|
||||
local after = duffle.skip_ws_and_cmt(source, ident_end) ---@type integer
|
||||
if source:sub(after, after) ~= "(" then return ident_end end
|
||||
local inner, after_p = duffle.read_parens(source, after) ---@type string|nil, integer
|
||||
local name = duffle.trim(inner or ""):match("^([%w_]+)") ---@type string
|
||||
local last = duffle.trim(inner or "") ---@type string
|
||||
local name = last:match("^([%w_]+)") ---@type string
|
||||
if name then
|
||||
out._chain = out._chain or {}
|
||||
out._chain[#out._chain + 1] = name
|
||||
push_tape_emit(out, name, last, line_of(pos))
|
||||
end
|
||||
return after_p or (after + 1)
|
||||
end
|
||||
@@ -2865,6 +2966,49 @@ local function parse_tb_emit(source, pos, ident_end, line_of, out)
|
||||
if name then
|
||||
out._chain = out._chain or {}
|
||||
out._chain[#out._chain + 1] = name
|
||||
push_tape_emit(out, name, last, line_of(pos))
|
||||
end
|
||||
return after_p or (after + 1)
|
||||
end
|
||||
|
||||
--- @param source string
|
||||
--- @param pos integer
|
||||
--- @param ident_end integer
|
||||
--- @param line_of fun(pos: integer): integer
|
||||
--- @param out SourceScan
|
||||
--- @return integer
|
||||
local function parse_tb_bind_(source, pos, ident_end, line_of, out)
|
||||
local after = duffle.skip_ws_and_cmt(source, ident_end) ---@type integer
|
||||
if source:sub(after, after) ~= "(" then return ident_end end
|
||||
local inner, after_p = duffle.read_parens(source, after) ---@type string|nil, integer
|
||||
local args = duffle.split_top_level_commas(inner or "") ---@type string[]
|
||||
local typ = duffle.trim(args[2] or "") ---@type string
|
||||
if typ ~= "" then
|
||||
local emits = out.tape_emits or {} ---@type TapeEmit[]
|
||||
for i = #emits, 1, -1 do ---@type integer
|
||||
if emits[i].binds == nil then
|
||||
emits[i].binds = typ
|
||||
break
|
||||
end
|
||||
end
|
||||
end
|
||||
return after_p or (after + 1)
|
||||
end
|
||||
|
||||
--- @param source string
|
||||
--- @param pos integer
|
||||
--- @param ident_end integer
|
||||
--- @param line_of fun(pos: integer): integer
|
||||
--- @param out SourceScan
|
||||
--- @return integer
|
||||
local function parse_tb_data(source, pos, ident_end, line_of, out)
|
||||
local after = duffle.skip_ws_and_cmt(source, ident_end) ---@type integer
|
||||
if source:sub(after, after) ~= "(" then return ident_end end
|
||||
local _, after_p = duffle.read_parens(source, after) ---@type string|nil, integer
|
||||
local emits = out.tape_emits or {} ---@type TapeEmit[]
|
||||
local last = emits[#emits] ---@type TapeEmit|nil
|
||||
if last then
|
||||
last.data_words = (last.data_words or 0) + 1
|
||||
end
|
||||
return after_p or (after + 1)
|
||||
end
|
||||
@@ -2872,6 +3016,8 @@ end
|
||||
local C_STMT_PARSERS = { ---@type table<string, fun(source: string, pos: integer, ident_end: integer, line_of: fun(pos: integer): integer, out: SourceScan): integer>
|
||||
tb_emit_ = parse_tb_emit_,
|
||||
tb_emit = parse_tb_emit,
|
||||
tb_bind_ = parse_tb_bind_,
|
||||
tb_data = parse_tb_data,
|
||||
addrs = parse_addrs_assign,
|
||||
}
|
||||
|
||||
@@ -2924,6 +3070,8 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
|
||||
atoms = {},
|
||||
raw_atoms = {},
|
||||
binds = {},
|
||||
atom_bundles = {},
|
||||
tape_emits = {},
|
||||
atom_infos = {},
|
||||
component_atom_infos = {},
|
||||
macros = {},
|
||||
@@ -3238,6 +3386,8 @@ local function merge_corpus_registries(corpus)
|
||||
corpus.reg_use_schemas = corpus.reg_use_schemas or {}
|
||||
corpus.reg_use_errors = corpus.reg_use_errors or {}
|
||||
corpus.tape_chains = corpus.tape_chains or {}
|
||||
corpus.atom_bundles = corpus.atom_bundles or {}
|
||||
corpus.tape_emits = corpus.tape_emits or {}
|
||||
|
||||
-- Replace the existing corpus collections with empty tables so a re-run on the same corpus produces identical state (deterministic merge).
|
||||
-- This is safe because M.run is the only writer to these tables within a single orchestrator invocation.
|
||||
@@ -3255,6 +3405,8 @@ local function merge_corpus_registries(corpus)
|
||||
"reg_use_schemas",
|
||||
"reg_use_errors",
|
||||
"tape_chains",
|
||||
"atom_bundles",
|
||||
"tape_emits",
|
||||
}) do
|
||||
corpus[key] = {}
|
||||
end
|
||||
@@ -3362,6 +3514,55 @@ local function merge_corpus_registries(corpus)
|
||||
for _, chain in ipairs(scan.tape_chains or {}) do ---@type integer, TapeChain
|
||||
corpus.tape_chains[#corpus.tape_chains + 1] = chain
|
||||
end
|
||||
for _, emit in ipairs(scan.tape_emits or {}) do ---@type integer, TapeEmit
|
||||
corpus.tape_emits[#corpus.tape_emits + 1] = emit
|
||||
end
|
||||
|
||||
-- atom_bundles: keyed by typedef name. First-wins per name (like components).
|
||||
for name, bundle in pairs(scan.atom_bundles or {}) do ---@type string, AtomBundle
|
||||
if corpus.atom_bundles[name] == nil then
|
||||
corpus.atom_bundles[name] = bundle
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
-- Join slot roles to AtomBundleEntry_ identities after catalogs and atoms exist.
|
||||
-- Do not invent a catalog from an entry that has no typedef.
|
||||
for _, bundle in pairs(corpus.atom_bundles) do ---@type string, AtomBundle
|
||||
local entries = {} ---@type table<string, string>
|
||||
for _, slot in ipairs(bundle.slots or {}) do ---@type integer, string
|
||||
local atom_name = bundle.name .. "_" .. slot ---@type string
|
||||
if corpus.atoms_by_name[atom_name] then
|
||||
entries[slot] = atom_name
|
||||
end
|
||||
end
|
||||
if next(entries) then
|
||||
bundle.entries = entries
|
||||
end
|
||||
end
|
||||
|
||||
-- Resolve tb_emit names: atom first, else unique catalog slot → entries[slot].
|
||||
-- Ambiguous slot or missing entry: leave the raw ident (no A/B later).
|
||||
for _, emit in ipairs(corpus.tape_emits) do ---@type integer, TapeEmit
|
||||
if corpus.atoms_by_name[emit.name] == nil then
|
||||
local slot = emit.slot or emit.name ---@type string
|
||||
local hit = nil ---@type AtomBundle|nil
|
||||
local n = 0 ---@type integer
|
||||
for _, bundle in pairs(corpus.atom_bundles) do ---@type string, AtomBundle
|
||||
for _, s in ipairs(bundle.slots or {}) do ---@type integer, string
|
||||
if s == slot then
|
||||
n = n + 1
|
||||
hit = bundle
|
||||
break
|
||||
end
|
||||
end
|
||||
end
|
||||
local ident = hit and hit.entries and hit.entries[slot] ---@type string|nil
|
||||
if n == 1 and ident then
|
||||
emit.name = ident
|
||||
emit.slot = slot
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
@@ -4016,6 +4016,66 @@ end
|
||||
-- Each check is one table row and one `check_*` function.
|
||||
-- This is the plex pattern: the iteration is in ONE place (validate), the variation is in DATA (this table).
|
||||
|
||||
--- @param src SourceFile
|
||||
--- @param pipe_ctx PassScratch
|
||||
--- @param findings Finding[]
|
||||
--- @return nil
|
||||
local function check_tb_bind_type_match(src, pipe_ctx, findings)
|
||||
local atoms_by_name = pipe_ctx.atoms_by_name or {} ---@type table<string, AtomEntry>
|
||||
local info_by = {} ---@type table<string, AtomInfoEntry>
|
||||
for _, info in ipairs(pipe_ctx.atom_infos_all or {}) do ---@type integer, AtomInfoEntry
|
||||
info_by[info.atom_name] = info
|
||||
end
|
||||
for name, info in pairs(pipe_ctx.info_by_atom or {}) do ---@type string, AtomInfoEntry
|
||||
info_by[name] = info
|
||||
end
|
||||
for _, emit in ipairs((src.scan and src.scan.tape_emits) or {}) do ---@type integer, TapeEmit
|
||||
if atoms_by_name[emit.name] then
|
||||
local atom_binds = info_by[emit.name] and info_by[emit.name].binds ---@type string|nil
|
||||
if emit.binds and atom_binds and emit.binds ~= atom_binds then
|
||||
findings[#findings + 1] = {
|
||||
atom = emit.name,
|
||||
line = emit.line or 0,
|
||||
check = "tb_bind_type_match",
|
||||
kind = "error",
|
||||
msg = string.format("tb_emit '%s' binds %s but atom_bind is %s"
|
||||
, emit.name, emit.binds, atom_binds),
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
--- @param src SourceFile
|
||||
--- @param pipe_ctx PassScratch
|
||||
--- @param findings Finding[]
|
||||
--- @return nil
|
||||
local function check_tb_bind_required(src, pipe_ctx, findings)
|
||||
local atoms_by_name = pipe_ctx.atoms_by_name or {} ---@type table<string, AtomEntry>
|
||||
local info_by = {} ---@type table<string, AtomInfoEntry>
|
||||
for _, info in ipairs(pipe_ctx.atom_infos_all or {}) do ---@type integer, AtomInfoEntry
|
||||
info_by[info.atom_name] = info
|
||||
end
|
||||
for name, info in pairs(pipe_ctx.info_by_atom or {}) do ---@type string, AtomInfoEntry
|
||||
info_by[name] = info
|
||||
end
|
||||
for _, emit in ipairs((src.scan and src.scan.tape_emits) or {}) do ---@type integer, TapeEmit
|
||||
if atoms_by_name[emit.name] then
|
||||
local atom_binds = info_by[emit.name] and info_by[emit.name].binds ---@type string|nil
|
||||
if atom_binds and emit.binds == nil and (emit.data_words or 0) == 0 then
|
||||
findings[#findings + 1] = {
|
||||
atom = emit.name,
|
||||
line = emit.line or 0,
|
||||
check = "tb_bind_required",
|
||||
kind = "error",
|
||||
msg = string.format("tb_emit '%s' has atom_bind(%s) but no tb_bind_ or tb_data"
|
||||
, emit.name, atom_binds),
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
local CHECK_RULES = { ---@type CheckRule[]
|
||||
{ name = "transfer_hazards", per_atom = check_transfer_hazards },
|
||||
{ name = "gte_input_latch", per_atom = check_gte_input_latch },
|
||||
@@ -4040,6 +4100,8 @@ local CHECK_RULES = { ---@type CheckRule[]
|
||||
{ name = "binds_no_substruct_deref", per_source = check_binds_no_substruct_deref },
|
||||
{ name = "component_self_consistency", per_source = check_component_self_consistency },
|
||||
{ name = "atom_calls_inferred_traffic", per_atom = check_atom_calls_inferred_traffic },
|
||||
{ name = "tb_bind_type_match", per_source = check_tb_bind_type_match },
|
||||
{ name = "tb_bind_required", per_source = check_tb_bind_required },
|
||||
}
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
@@ -77,6 +77,8 @@ local PASS_FLAG_DISPATCH_KEY = "__pass__" ---@type string
|
||||
--- @field atom_phases table<string, AtomPhaseGroup>
|
||||
--- @field word_counts WordCounts
|
||||
--- @field components table<string, Component>
|
||||
--- @field atom_bundles table<string, AtomBundle>|nil
|
||||
--- @field tape_emits TapeEmit[]|nil
|
||||
--- @field collisions CorpusCollision[]
|
||||
--- @field resolver SourceResolver
|
||||
--- @field component_atom_infos AtomInfoEntry[]|nil
|
||||
@@ -373,8 +375,7 @@ COMMON_FLAGS:
|
||||
--help Show this help and exit
|
||||
|
||||
EXIT CODES:
|
||||
0 All requested passes succeeded
|
||||
1 Validation errors found
|
||||
0 Ran. Findings print on stderr and in the report; they do not fail the process.
|
||||
2 Metaprogram internal error
|
||||
|
||||
EXAMPLES:
|
||||
@@ -665,6 +666,8 @@ local function build_ctx(args)
|
||||
atom_phases = {},
|
||||
word_counts = {},
|
||||
components = {},
|
||||
atom_bundles = {},
|
||||
tape_emits = {},
|
||||
collisions = {},
|
||||
resolver = resolution.resolver,
|
||||
}
|
||||
@@ -816,8 +819,7 @@ local function main(argv)
|
||||
local requested = args.requested_set ---@type string[]
|
||||
local closed = topo_sort(PASSES, requested) ---@type string[]
|
||||
|
||||
local had_errors = dispatch_passes(ctx, closed) ---@type boolean
|
||||
if had_errors then os.exit(EXIT_VALIDATION_ERRORS) end
|
||||
dispatch_passes(ctx, closed)
|
||||
end)
|
||||
|
||||
if not ok then
|
||||
|
||||
Reference in New Issue
Block a user