mirror of
https://github.com/Ed94/pikuma_ps1.git
synced 2026-08-07 08:08:49 +00:00
Compare commits
11
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
97d2f66c5a | ||
|
|
d9406553b3 | ||
|
|
e662d175ab | ||
|
|
5387a07b84 | ||
|
|
65d805e3ba | ||
|
|
987f4dee1e | ||
|
|
df723c691d | ||
|
|
45ac85c038 | ||
|
|
072231c46b | ||
|
|
2b00956862 | ||
|
|
1ffad6cf98 |
@@ -15,3 +15,5 @@ toolchain/PSn00bSDK
|
||||
*.a
|
||||
.sentry-native
|
||||
.vscode/settings.json
|
||||
toolchain/lfs
|
||||
toolchain/lpeg
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
#pragma once
|
||||
#endif
|
||||
// Auto-generated by tape_atom_annotation_pass.lua — DO NOT EDIT
|
||||
// Auto-generated by ps1_meta.lua — DO NOT EDIT
|
||||
// Source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
||||
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
|
||||
|
||||
|
||||
+2
-2
@@ -419,7 +419,7 @@ typedef Struct_(Poly_F4) {
|
||||
};
|
||||
};
|
||||
|
||||
/* ---------- Poly_G3 (Gouraud Triangle; 6 words) ---------- */
|
||||
/* ---------- Poly_G3 (Gouraud Triangle; 7 words) ---------- */
|
||||
typedef Struct_(Poly_G3) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
@@ -427,7 +427,7 @@ typedef Struct_(Poly_G3) {
|
||||
V2_S2 p2;
|
||||
};
|
||||
|
||||
/* ---------- Poly_G4 (Gouraud Quad; 5 words in the demo's interleaved layout) ---------- */
|
||||
/* ---------- Poly_G4 (Gouraud Quad; 9 words) ---------- */
|
||||
typedef Struct_(Poly_G4) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
|
||||
@@ -123,7 +123,6 @@ MipsAtom_(floor_f3_face) atom_info(
|
||||
nop,
|
||||
branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop,
|
||||
/* Format Primitive */
|
||||
// mac_format_f3_color(0x20FF, 0xFFFF), // works
|
||||
mac_format_f3_color(0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
|
||||
mac_gte_store_f3_post_rtpt(),
|
||||
|
||||
|
||||
+176
-179
@@ -18,12 +18,12 @@
|
||||
local M = {}
|
||||
|
||||
local BLOCK_OPEN = {
|
||||
["do"] = true,
|
||||
["function"] = true,
|
||||
["if"] = true,
|
||||
["for"] = true,
|
||||
["while"] = true,
|
||||
["repeat"] = true,
|
||||
["do"] = true,
|
||||
["function"] = true,
|
||||
["if"] = true,
|
||||
["for"] = true,
|
||||
["while"] = true,
|
||||
["repeat"] = true,
|
||||
}
|
||||
|
||||
local function is_block_close(token) return token == "end" or token == "until" end
|
||||
@@ -31,121 +31,118 @@ local function is_block_close(token) return token == "end" or token == "until" e
|
||||
-- (internal) Walk one source file and return a list of
|
||||
-- {line, depth, token} entries where depth > max_nesting.
|
||||
local function audit_file(path, max_nesting)
|
||||
local f = io.open(path, "r")
|
||||
if not f then error("Cannot open " .. path) end
|
||||
local content = f:read("*a")
|
||||
f:close()
|
||||
local f = io.open(path, "r")
|
||||
if not f then error("Cannot open " .. path) end
|
||||
local content = f:read("*a")
|
||||
f:close()
|
||||
|
||||
local violations = {}
|
||||
local depth = 0
|
||||
local line = 1
|
||||
local pos = 1
|
||||
local src_len = #content
|
||||
local token_idx = 0
|
||||
local violations = {}
|
||||
local depth = 0
|
||||
local line = 1
|
||||
local pos = 1
|
||||
local src_len = #content
|
||||
local token_idx = 0
|
||||
|
||||
local function read_ident_at(start_pos)
|
||||
local ident_start = start_pos
|
||||
if ident_start > src_len then return nil end
|
||||
local first_ch = content:sub(ident_start, ident_start)
|
||||
if not (first_ch:match("[%a_]")) then return nil end
|
||||
local scan = start_pos + 1
|
||||
while scan <= src_len do
|
||||
local ch = content:sub(scan, scan)
|
||||
if not (ch:match("[%w_]")) then break end
|
||||
scan = scan + 1
|
||||
end
|
||||
return content:sub(ident_start, scan - 1), scan
|
||||
end
|
||||
local function read_ident_at(start_pos)
|
||||
local ident_start = start_pos
|
||||
if ident_start > src_len then return nil end
|
||||
local first_ch = content:sub(ident_start, ident_start)
|
||||
if not (first_ch:match("[%a_]")) then return nil end
|
||||
local scan = start_pos + 1
|
||||
while scan <= src_len do
|
||||
local ch = content:sub(scan, scan)
|
||||
if not (ch:match("[%w_]")) then break end
|
||||
scan = scan + 1
|
||||
end
|
||||
return content:sub(ident_start, scan - 1), scan
|
||||
end
|
||||
|
||||
-- Skip past a string literal or comment starting at `start_pos`.
|
||||
-- Returns the position just past the construct, or nil if `start_pos`
|
||||
-- is not the start of a string/comment.
|
||||
local function skip_string_or_comment(start_pos)
|
||||
local ch = content:sub(start_pos, start_pos)
|
||||
if ch == '"' or ch == "'" then
|
||||
local scan = start_pos + 1
|
||||
while scan <= src_len do
|
||||
local c = content:sub(scan, scan)
|
||||
if c == "\\" then
|
||||
scan = scan + 2
|
||||
elseif c == ch then
|
||||
return scan + 1
|
||||
else
|
||||
scan = scan + 1
|
||||
end
|
||||
end
|
||||
return src_len + 1
|
||||
elseif ch == "-" and content:sub(start_pos + 1, start_pos + 1) == "-" then
|
||||
local scan = start_pos + 2
|
||||
if content:sub(scan, scan + 1) == "[[" and content:sub(scan + 2, scan + 3) == "[" then
|
||||
-- Long bracket comment [==[ ... ]==]
|
||||
scan = scan + 2
|
||||
local eq = ""
|
||||
while content:sub(scan, scan) == "=" do
|
||||
eq = eq .. "="
|
||||
scan = scan + 1
|
||||
end
|
||||
local close_marker = "]" .. eq .. "]"
|
||||
local close_pos = content:find(close_marker, scan, true)
|
||||
if close_pos then
|
||||
return close_pos + #close_marker
|
||||
else
|
||||
return src_len + 1
|
||||
end
|
||||
else
|
||||
while scan <= src_len and content:sub(scan, scan) ~= "\n" do scan = scan + 1 end
|
||||
return scan + 1
|
||||
end
|
||||
elseif ch == "[" and content:sub(start_pos + 1, start_pos + 1) == "[" then
|
||||
local scan = start_pos + 2
|
||||
local eq = ""
|
||||
while content:sub(scan, scan) == "=" do
|
||||
eq = eq .. "="
|
||||
scan = scan + 1
|
||||
end
|
||||
local close_marker = "]" .. eq .. "]"
|
||||
local close_pos = content:find(close_marker, scan, true)
|
||||
if close_pos then
|
||||
return close_pos + #close_marker
|
||||
else
|
||||
return src_len + 1
|
||||
end
|
||||
end
|
||||
return nil
|
||||
end
|
||||
-- Skip past a string literal or comment starting at `start_pos`.
|
||||
-- Returns the position just past the construct, or nil if `start_pos`
|
||||
-- is not the start of a string/comment.
|
||||
local function skip_string_or_comment(start_pos)
|
||||
local ch = content:sub(start_pos, start_pos)
|
||||
if ch == '"' or ch == "'" then
|
||||
local scan = start_pos + 1
|
||||
while scan <= src_len do
|
||||
local c = content:sub(scan, scan)
|
||||
if c == "\\" then scan = scan + 2
|
||||
elseif c == ch then return scan + 1
|
||||
else scan = scan + 1
|
||||
end
|
||||
end
|
||||
return src_len + 1
|
||||
elseif ch == "-" and content:sub(start_pos + 1, start_pos + 1) == "-" then
|
||||
local scan = start_pos + 2
|
||||
if content:sub(scan, scan + 1) == "[[" and content:sub(scan + 2, scan + 3) == "[" then
|
||||
-- Long bracket comment [==[ ... ]==]
|
||||
scan = scan + 2
|
||||
local eq = ""
|
||||
while content:sub(scan, scan) == "=" do
|
||||
eq = eq .. "="
|
||||
scan = scan + 1
|
||||
end
|
||||
local close_marker = "]" .. eq .. "]"
|
||||
local close_pos = content:find(close_marker, scan, true)
|
||||
if close_pos then
|
||||
return close_pos + #close_marker
|
||||
else
|
||||
return src_len + 1
|
||||
end
|
||||
else
|
||||
while scan <= src_len and content:sub(scan, scan) ~= "\n" do scan = scan + 1 end
|
||||
return scan + 1
|
||||
end
|
||||
elseif ch == "[" and content:sub(start_pos + 1, start_pos + 1) == "[" then
|
||||
local scan = start_pos + 2
|
||||
local eq = ""
|
||||
while content:sub(scan, scan) == "=" do
|
||||
eq = eq .. "="
|
||||
scan = scan + 1
|
||||
end
|
||||
local close_marker = "]" .. eq .. "]"
|
||||
local close_pos = content:find(close_marker, scan, true)
|
||||
if close_pos then
|
||||
return close_pos + #close_marker
|
||||
else
|
||||
return src_len + 1
|
||||
end
|
||||
end
|
||||
return nil
|
||||
end
|
||||
|
||||
while pos <= src_len do
|
||||
local ch = content:sub(pos, pos)
|
||||
if ch == "\n" then line = line + 1 end
|
||||
while pos <= src_len do
|
||||
local ch = content:sub(pos, pos)
|
||||
if ch == "\n" then line = line + 1 end
|
||||
|
||||
local skip_to = skip_string_or_comment(pos)
|
||||
if skip_to then
|
||||
for scan = pos, skip_to - 1 do
|
||||
if content:sub(scan, scan) == "\n" then line = line + 1 end
|
||||
end
|
||||
pos = skip_to
|
||||
elseif ch:match("[%a_]") then
|
||||
local tok, next_pos = read_ident_at(pos)
|
||||
token_idx = token_idx + 1
|
||||
if BLOCK_OPEN[tok] then
|
||||
depth = depth + 1
|
||||
if depth > max_nesting then
|
||||
violations[#violations + 1] = {
|
||||
line = line,
|
||||
depth = depth,
|
||||
token = tok,
|
||||
}
|
||||
end
|
||||
elseif is_block_close(tok) then
|
||||
depth = depth - 1
|
||||
end
|
||||
pos = next_pos
|
||||
else
|
||||
pos = pos + 1
|
||||
end
|
||||
end
|
||||
local skip_to = skip_string_or_comment(pos)
|
||||
if skip_to then
|
||||
for scan = pos, skip_to - 1 do
|
||||
if content:sub(scan, scan) == "\n" then line = line + 1 end
|
||||
end
|
||||
pos = skip_to
|
||||
elseif ch:match("[%a_]") then
|
||||
local tok, next_pos = read_ident_at(pos)
|
||||
token_idx = token_idx + 1
|
||||
if BLOCK_OPEN[tok] then
|
||||
depth = depth + 1
|
||||
if depth > max_nesting then
|
||||
violations[#violations + 1] = {
|
||||
line = line,
|
||||
depth = depth,
|
||||
token = tok,
|
||||
}
|
||||
end
|
||||
elseif is_block_close(tok) then
|
||||
depth = depth - 1
|
||||
end
|
||||
pos = next_pos
|
||||
else
|
||||
pos = pos + 1
|
||||
end
|
||||
end
|
||||
|
||||
return violations
|
||||
return violations
|
||||
end
|
||||
|
||||
--- Audit one file. Returns nil if clean, else a list of violations.
|
||||
@@ -153,78 +150,78 @@ end
|
||||
--- @param max_nesting integer -- default 5
|
||||
--- @return table|nil
|
||||
function M.audit(path, max_nesting)
|
||||
local violations = audit_file(path, max_nesting or 5)
|
||||
if #violations == 0 then return nil end
|
||||
return violations
|
||||
local violations = audit_file(path, max_nesting or 5)
|
||||
if #violations == 0 then return nil end
|
||||
return violations
|
||||
end
|
||||
|
||||
-- Module CLI.
|
||||
if arg and arg[1] then
|
||||
local max_nesting = 5
|
||||
local files = {}
|
||||
for arg_idx = 1, #arg do
|
||||
if arg[arg_idx] == "--max" and arg[arg_idx + 1] then
|
||||
max_nesting = tonumber(arg[arg_idx + 1]) or 5
|
||||
else
|
||||
files[#files + 1] = arg[arg_idx]
|
||||
end
|
||||
end
|
||||
local max_nesting = 5
|
||||
local files = {}
|
||||
for arg_idx = 1, #arg do
|
||||
if arg[arg_idx] == "--max" and arg[arg_idx + 1] then
|
||||
max_nesting = tonumber(arg[arg_idx + 1]) or 5
|
||||
else
|
||||
files[#files + 1] = arg[arg_idx]
|
||||
end
|
||||
end
|
||||
|
||||
-- Accept either a directory or a file path. Directory args are
|
||||
-- expanded via `dir /b *.lua` (Windows) or `ls *.lua` (Unix).
|
||||
local function is_dir(p)
|
||||
local f = io.open(p, "r")
|
||||
if f then f:close() return false end
|
||||
return true
|
||||
end
|
||||
local function list_lua(dir)
|
||||
local out = {}
|
||||
local cmd
|
||||
if package.config:sub(1, 1) == "\\" then
|
||||
cmd = 'dir /b "' .. dir .. '\\*.lua" 2>nul'
|
||||
else
|
||||
cmd = 'ls -1 "' .. dir .. '"/*.lua 2>/dev/null'
|
||||
end
|
||||
local p = io.popen(cmd)
|
||||
if p then
|
||||
for line in p:lines() do
|
||||
if line:match("%.lua$") then
|
||||
out[#out + 1] = dir .. "/" .. line
|
||||
end
|
||||
end
|
||||
p:close()
|
||||
end
|
||||
return out
|
||||
end
|
||||
-- Accept either a directory or a file path. Directory args are
|
||||
-- expanded via `dir /b *.lua` (Windows) or `ls *.lua` (Unix).
|
||||
local function is_dir(p)
|
||||
local f = io.open(p, "r")
|
||||
if f then f:close() return false end
|
||||
return true
|
||||
end
|
||||
local function list_lua(dir)
|
||||
local out = {}
|
||||
local cmd
|
||||
if package.config:sub(1, 1) == "\\" then
|
||||
cmd = 'dir /b "' .. dir .. '\\*.lua" 2>nul'
|
||||
else
|
||||
cmd = 'ls -1 "' .. dir .. '"/*.lua 2>/dev/null'
|
||||
end
|
||||
local p = io.popen(cmd)
|
||||
if p then
|
||||
for line in p:lines() do
|
||||
if line:match("%.lua$") then
|
||||
out[#out + 1] = dir .. "/" .. line
|
||||
end
|
||||
end
|
||||
p:close()
|
||||
end
|
||||
return out
|
||||
end
|
||||
|
||||
local to_check = {}
|
||||
for _, f in ipairs(files) do
|
||||
if is_dir(f) then
|
||||
for _, sub in ipairs(list_lua(f)) do to_check[#to_check + 1] = sub end
|
||||
else
|
||||
to_check[#to_check + 1] = f
|
||||
end
|
||||
end
|
||||
local to_check = {}
|
||||
for _, f in ipairs(files) do
|
||||
if is_dir(f) then
|
||||
for _, sub in ipairs(list_lua(f)) do to_check[#to_check + 1] = sub end
|
||||
else
|
||||
to_check[#to_check + 1] = f
|
||||
end
|
||||
end
|
||||
|
||||
local total_violations = 0
|
||||
for _, f in ipairs(to_check) do
|
||||
local v = M.audit(f, max_nesting)
|
||||
if v then
|
||||
io.write(string.format("\n%s\n", f))
|
||||
for _, x in ipairs(v) do
|
||||
io.write(string.format(" line %d: depth %d (after '%s')\n", x.line, x.depth, x.token))
|
||||
end
|
||||
total_violations = total_violations + #v
|
||||
end
|
||||
end
|
||||
local total_violations = 0
|
||||
for _, f in ipairs(to_check) do
|
||||
local v = M.audit(f, max_nesting)
|
||||
if v then
|
||||
io.write(string.format("\n%s\n", f))
|
||||
for _, x in ipairs(v) do
|
||||
io.write(string.format(" line %d: depth %d (after '%s')\n", x.line, x.depth, x.token))
|
||||
end
|
||||
total_violations = total_violations + #v
|
||||
end
|
||||
end
|
||||
|
||||
if total_violations == 0 then
|
||||
io.write("OK: no files exceed max nesting of " .. max_nesting .. "\n")
|
||||
os.exit(0)
|
||||
else
|
||||
io.write(string.format("\n%d nesting violation(s) found.\n", total_violations))
|
||||
os.exit(1)
|
||||
end
|
||||
if total_violations == 0 then
|
||||
io.write("OK: no files exceed max nesting of " .. max_nesting .. "\n")
|
||||
os.exit(0)
|
||||
else
|
||||
io.write(string.format("\n%d nesting violation(s) found.\n", total_violations))
|
||||
os.exit(1)
|
||||
end
|
||||
end
|
||||
|
||||
return M
|
||||
|
||||
+326
-106
@@ -15,13 +15,17 @@
|
||||
--- Lua 5.3 compatible; no `<close>`/`<toclose>`, no `continue`, no
|
||||
--- 5.4 string.dump improvements. LuaJIT 5.1+extensions model is the primary target.
|
||||
---
|
||||
--- **No `:match` / `:gmatch` regex use anywhere**; all delimiter-
|
||||
--- splitting is hand-rolled or via LPeg (the regex-free PEG library).
|
||||
--- The hot lexer primitives are LPeg-backed where it pays off;
|
||||
--- hand-rolled variants remain for callers that need a fallback.
|
||||
--- **No `:match` / `:gmatch` regex use anywhere**;
|
||||
--- all delimiter-splitting is hand-rolled or via LPeg (the regex-free PEG library).
|
||||
|
||||
local M = {}
|
||||
|
||||
-- Optional native extension: lfs (LuaFileSystem). When present, ensure_dir uses
|
||||
-- lfs.attributes + lfs.mkdir instead of spawning `cmd.exe mkdir` — saves ~55ms per
|
||||
-- unique directory on Windows. Built by `update_deps.ps1` to `toolchain/lfs/lfs.dll`
|
||||
-- and wired into package.cpath by `scripts/duffle_paths.lua`.
|
||||
local lfs = pcall(require, "lfs") and require("lfs") or nil
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Cross-file type aliases
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
@@ -297,6 +301,52 @@ function M.write_file(path, content)
|
||||
f:write(content); f:close()
|
||||
end
|
||||
|
||||
-- Write content to disk in binary mode so LF line endings are preserved on Windows
|
||||
-- (text mode would convert LF -> CRLF, breaking byte-identical diffs against git-tracked gen/*.h files which are stored as LF).
|
||||
-- @param path string
|
||||
-- @param content string
|
||||
function M.write_file_lf(path, content)
|
||||
local f = io.open(path, "wb")
|
||||
if not f then error("Cannot write " .. path) end
|
||||
f:write(content); f:close()
|
||||
end
|
||||
|
||||
-- Convert a (possibly relative) path to an absolute path, using CWD if needed.
|
||||
-- Normalizes forward slashes to backslashes on Windows.
|
||||
-- Used for byte-identical emit: the // Source: comment line uses the absolute path.
|
||||
--
|
||||
-- The CWD is memoized (one `io.popen("cd")` per process — ~50ms on Windows).
|
||||
-- Without the cache, calling this per-source in the components pass added ~1.5s to a 30-source build.
|
||||
-- @param path string
|
||||
-- @return string
|
||||
local _absolute_path_cache = {}
|
||||
|
||||
function M.to_absolute_path(path)
|
||||
if _absolute_path_cache[path] then return _absolute_path_cache[path] end
|
||||
if #path >= 2 and path:sub(2, 2) == ":" then
|
||||
-- Already absolute; normalize slashes for consistency.
|
||||
local result = (path:gsub("/", "\\"))
|
||||
_absolute_path_cache[path] = result
|
||||
return result
|
||||
end
|
||||
-- Native: lfs.currentdir() is ~0ms vs io.popen("cd") at ~50ms per call.
|
||||
local cwd
|
||||
if lfs then
|
||||
cwd = lfs.currentdir()
|
||||
else
|
||||
local p = io.popen("cd")
|
||||
if not p then _absolute_path_cache[path] = path; return path end
|
||||
cwd = p:read("*l")
|
||||
p:close()
|
||||
end
|
||||
if not cwd then _absolute_path_cache[path] = path; return path end
|
||||
cwd = cwd:gsub("/", "\\")
|
||||
local tail = (path:gsub("/", "\\"))
|
||||
local result = cwd .. "\\" .. tail
|
||||
_absolute_path_cache[path] = result
|
||||
return result
|
||||
end
|
||||
|
||||
-- Cache of directories already verified to exist in this process.
|
||||
-- Each ensure_dir() call may otherwise spawn a `cmd.exe mkdir` (50-100ms per call on Windows) — calling it inside per-source loops added 1.5+
|
||||
-- seconds to the report pass. Cache makes ensure_dir idempotent within the process lifetime.
|
||||
@@ -306,35 +356,52 @@ local _ensured_dirs = {}
|
||||
function M.ensure_dir(path)
|
||||
if _ensured_dirs[path] then return end
|
||||
_ensured_dirs[path] = true
|
||||
local is_win = package.config:sub(1, 1) == "\\"
|
||||
os.execute(is_win and ('if not exist "' .. path .. '" mkdir "' .. path .. '"') or ('mkdir -p "' .. path .. '" 2>/dev/null'))
|
||||
if lfs then
|
||||
-- Native: ~0ms when dir exists (the common case). lfs.mkdir on a new dir is ~2ms (no shell spawn).
|
||||
-- Falls through silently if lfs.mkdir fails (e.g. permission denied); the subsequent write_file will surface the error.
|
||||
if lfs.attributes(path, "mode") ~= "directory" then lfs.mkdir(path) end
|
||||
else
|
||||
-- Fallback: shell mkdir. Slow (~55ms per call on Windows due to cmd.exe spawn) but works without lfs.
|
||||
local is_win = package.config:sub(1, 1) == "\\"
|
||||
os.execute(is_win and ('if not exist "' .. path .. '" mkdir "' .. path .. '"') or ('mkdir -p "' .. path .. '" 2>/dev/null'))
|
||||
end
|
||||
end
|
||||
|
||||
-- Test helper: clear the cache (used by tests + between process runs).
|
||||
-- Not normally needed since Lua state is per-process.
|
||||
function M._reset_ensured_dirs() _ensured_dirs = {} end
|
||||
|
||||
-- Group a list of `SourceFile`-shaped records by their `dir` field.
|
||||
-- Used by the annotation / static-analysis / report passes to partition sources into per-DIRECTORY (per-module) buckets
|
||||
-- before emitting per-module reports. Insertion order preserved within each bucket (matches source order in `ctx.sources`).
|
||||
-- @param sources table[] -- list of source records (each having a `dir` string field)
|
||||
-- @return table<string, table[]> -- map of `dir` -> sources in that dir
|
||||
function M.group_sources_by_dir(sources)
|
||||
local by_dir = {}
|
||||
for _, src in ipairs(sources) do
|
||||
by_dir[src.dir] = by_dir[src.dir] or {}
|
||||
table.insert(by_dir[src.dir], src)
|
||||
end
|
||||
return by_dir
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Section 4: C-language scanner primitives
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Skip a string or C-style comment starting at position `pos`.
|
||||
-- Returns the position just past the construct, or `pos` unchanged if no string/comment starts there. LPeg-backed.
|
||||
function M.skip_str_or_cmt(s, pos)
|
||||
return lpeg.match(lpeg_str_or_cmt_pat, s, pos) or pos
|
||||
end
|
||||
function M.skip_str_or_cmt(s, pos) return lpeg.match(lpeg_str_or_cmt_pat, s, pos) or pos end
|
||||
|
||||
-- Skip whitespace AND C-style comments starting at position `pos`.
|
||||
-- LPeg-backed; ~5-10x faster than a hand-rolled byte-by-byte walker.
|
||||
function M.skip_ws_and_cmt(s, pos)
|
||||
return lpeg.match(lpeg_ws_and_cmt_pat, s, pos) or pos
|
||||
end
|
||||
function M.skip_ws_and_cmt(s, pos) return lpeg.match(lpeg_ws_and_cmt_pat, s, pos) or pos end
|
||||
|
||||
-- Read a C-style identifier (alpha followed by zero+ alnum) starting at position `pos`.
|
||||
-- Returns the identifier string + the position just past it, or nil + pos if no identifier starts here. LPeg-backed.
|
||||
function M.read_ident(s, pos)
|
||||
local result = lpeg.match(lpeg_ident_pat, s, pos)
|
||||
if result then return result, pos + #result end
|
||||
if result then return result, pos + #result end
|
||||
return nil, pos
|
||||
end
|
||||
|
||||
@@ -344,7 +411,9 @@ end
|
||||
function M.read_balanced(s, open_char, close_char, pos)
|
||||
local open_byte = open_char:byte()
|
||||
if s:byte(pos) ~= open_byte then return nil, pos end
|
||||
-- scan: <open_char>
|
||||
pos = pos + 1
|
||||
-- scan: <open_char> <inner...>
|
||||
local len = #s
|
||||
local depth = 1
|
||||
local a = pos
|
||||
@@ -353,15 +422,23 @@ function M.read_balanced(s, open_char, close_char, pos)
|
||||
if c == open_byte then
|
||||
depth = depth + 1
|
||||
pos = pos + 1
|
||||
-- scan: <open_char> <inner...> <open_char> (depth=depth)
|
||||
elseif c == close_char:byte() then
|
||||
depth = depth - 1
|
||||
if depth == 0 then break end
|
||||
pos = pos + 1
|
||||
-- scan: <open_char> <inner...> <close_char> (depth=depth)
|
||||
else
|
||||
local nx = M.skip_str_or_cmt(s, pos)
|
||||
if nx > pos then pos = nx else pos = pos + 1 end
|
||||
if nx > pos then
|
||||
-- scan: <open_char> <inner...> <str|cmt>
|
||||
pos = nx
|
||||
else
|
||||
pos = pos + 1
|
||||
end
|
||||
end
|
||||
end
|
||||
-- scan: <open_char> <inner> <close_char>
|
||||
return s:sub(a, pos - 1), pos + 1
|
||||
end
|
||||
|
||||
@@ -379,17 +456,35 @@ function M.scan_to_char(s, target, start)
|
||||
while pos <= #s do
|
||||
local c = s:byte(pos)
|
||||
if c == target_byte then return pos end
|
||||
-- scan: ... <target found> | <skipping to target>
|
||||
if c == BYTE_OPEN_PAREN then local _, a = M.read_balanced(s, "(", ")", pos); pos = a
|
||||
-- scan: ... ( <balanced> ) ...
|
||||
elseif c == BYTE_OPEN_BRACE then local _, a = M.read_balanced(s, "{", "}", pos); pos = a
|
||||
-- scan: ... { <balanced> } ...
|
||||
elseif c == BYTE_OPEN_BRACK then local _, a = M.read_balanced(s, "[", "]", pos); pos = a
|
||||
-- scan: ... [ <balanced> ] ...
|
||||
else
|
||||
local nx = M.skip_str_or_cmt(s, pos)
|
||||
pos = (nx > pos) and nx or (pos + 1)
|
||||
-- scan: ... <str|cmt skipped> ...
|
||||
end
|
||||
end
|
||||
return nil
|
||||
end
|
||||
|
||||
-- If `s[pos]` is `#`, skip to the end of the preprocessor directive line (past the newline).
|
||||
-- Returns the position past the newline, or nil if `s[pos]` is not `#`.
|
||||
-- scan: #<directive>\n -> past the newline
|
||||
function M.skip_preprocessor_line(s, pos)
|
||||
if s:byte(pos) ~= 35 then return nil end -- '#'
|
||||
local scan = pos
|
||||
local len = #s
|
||||
while scan <= len and s:byte(scan) ~= BYTE_NEWLINE do
|
||||
scan = scan + 1
|
||||
end
|
||||
return scan + 1
|
||||
end
|
||||
|
||||
-- Split a brace-body into top-level comma-separated tokens. Honors nested
|
||||
-- parens/braces/brackets and skips strings/comments.
|
||||
--
|
||||
@@ -447,28 +542,29 @@ function M.split_top_level_commas(body)
|
||||
end
|
||||
|
||||
while pos <= body_len do
|
||||
local c = body:byte(pos)
|
||||
if c == BYTE_OPEN_PAREN then -- '('
|
||||
local _, a = M.read_parens(body, pos); pos = a
|
||||
elseif c == BYTE_OPEN_BRACE then -- '{'
|
||||
local _, a = M.read_braces(body, pos); pos = a
|
||||
elseif c == BYTE_OPEN_BRACK then -- '['
|
||||
local _, a = M.read_brackets(body, pos); pos = a
|
||||
elseif c == BYTE_COMMA then -- ','
|
||||
local c = body:byte(pos)
|
||||
if c == BYTE_OPEN_PAREN then local _, a = M.read_parens(body, pos); pos = a -- scan: ... ( <balanced> ...
|
||||
elseif c == BYTE_OPEN_BRACE then local _, a = M.read_braces(body, pos); pos = a -- scan: ... { <balanced> ...
|
||||
elseif c == BYTE_OPEN_BRACK then local _, a = M.read_brackets(body, pos); pos = a -- scan: ... ( <balanced> ...
|
||||
elseif c == BYTE_COMMA then
|
||||
-- scan: ... <token> , <next> ...
|
||||
emit(pos - 1)
|
||||
pos = pos + 1
|
||||
token_start = pos
|
||||
elseif c == BYTE_SEMI then -- ';'
|
||||
elseif c == BYTE_SEMI then
|
||||
-- scan: ... <token> ; <next> ...
|
||||
emit(pos - 1)
|
||||
pos = pos + 1
|
||||
token_start = pos
|
||||
elseif c == BYTE_NEWLINE then -- '\n'
|
||||
elseif c == BYTE_NEWLINE then
|
||||
-- scan: ... <token> \n <next> ...
|
||||
emit(pos - 1)
|
||||
pos = pos + 1
|
||||
token_start = pos
|
||||
else
|
||||
local nx = M.skip_str_or_cmt(body, pos)
|
||||
if nx > pos then
|
||||
-- scan: ... <str|cmt> ...
|
||||
-- Skipped a comment or string at top level: emit token break.
|
||||
pos = nx
|
||||
emit(pos - 1)
|
||||
@@ -477,10 +573,111 @@ function M.split_top_level_commas(body)
|
||||
end
|
||||
end
|
||||
end
|
||||
-- scan: <token> , <token> , ... <token>
|
||||
emit(body_len)
|
||||
return tokens
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Section 4b: tokenize_body + build_body_line_index (shared, memoized)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Moved here from passes/static_analysis.lua so all passes can share the memoized
|
||||
-- per-body tokenization. The memoization key is the body string (immutable per pass).
|
||||
|
||||
local _tokenize_body_cache = {}
|
||||
local _body_line_index_cache = {}
|
||||
|
||||
--- Tokenize the body inner-text into a flat list of `{tok, rel}` pairs.
|
||||
--- `tok` is the trimmed token string; `rel` is the byte offset within `body`.
|
||||
--- Memoized on the body string — first call pays O(body_len), subsequent calls return cached.
|
||||
--- @param body string
|
||||
--- @return table[] -- {{tok=string, rel=integer}, ...}
|
||||
function M.tokenize_body(body)
|
||||
if _tokenize_body_cache[body] ~= nil then return _tokenize_body_cache[body] end
|
||||
local out = {}
|
||||
local len = #body
|
||||
local rel = 1
|
||||
while rel <= len do
|
||||
local ws_end = M.skip_ws_and_cmt(body, rel)
|
||||
if ws_end > rel then rel = ws_end end
|
||||
if rel > len then break end
|
||||
|
||||
local scan = rel
|
||||
while scan <= len do
|
||||
local c = body:byte(scan)
|
||||
if c == 44 then break end -- ','
|
||||
if c == 10 then break end -- '\n'
|
||||
if c == 59 then break end -- ';'
|
||||
if c == 40 then local _, a = M.read_parens (body, scan); scan = a -- '('
|
||||
elseif c == 123 then local _, a = M.read_braces (body, scan); scan = a -- '{'
|
||||
elseif c == 91 then local _, a = M.read_brackets (body, scan); scan = a -- '['
|
||||
elseif c == 34 or c == 39 then scan = M.skip_str_or_cmt(body, scan) + 1 -- '"' or '\''
|
||||
else
|
||||
scan = scan + 1
|
||||
end
|
||||
end
|
||||
local tok = M.trim(body:sub(rel, scan - 1))
|
||||
if tok ~= "" then out[#out + 1] = { tok = tok, rel = rel } end
|
||||
if scan <= len then
|
||||
scan = scan + 1
|
||||
local w = M.skip_ws_and_cmt(body, scan)
|
||||
if w > scan then scan = w end
|
||||
end
|
||||
rel = scan
|
||||
end
|
||||
_tokenize_body_cache[body] = out
|
||||
return out
|
||||
end
|
||||
|
||||
--- Tokenize the body into a flat list of trimmed string tokens (preserves comments).
|
||||
--- Uses `split_top_level_commas` (which appends trailing comments to the previous token)
|
||||
--- so the components pass can emit `/* Words: ... */` comments in the .macs.h output.
|
||||
--- @param body string
|
||||
--- @return string[]
|
||||
function M.tokenize_body_simple(body)
|
||||
local tokens = M.split_top_level_commas(body)
|
||||
local out = {}
|
||||
for i = 1, #tokens do out[i] = M.trim(tokens[i]) end
|
||||
return out
|
||||
end
|
||||
|
||||
--- Build a line-index: count `\n` chars from offset 1 up to the offset; that count + 1 is the line number (1-based).
|
||||
--- Memoized on the body string.
|
||||
--- @param body string
|
||||
--- @return table -- index[pos] = line_number
|
||||
function M.build_body_line_index(body)
|
||||
if _body_line_index_cache[body] ~= nil then return _body_line_index_cache[body] end
|
||||
local index = {}
|
||||
local len = #body
|
||||
local newline_count = 0
|
||||
for pos = 1, len do
|
||||
if pos > 1 then
|
||||
index[pos] = newline_count + 1
|
||||
end
|
||||
if body:byte(pos) == 10 then
|
||||
newline_count = newline_count + 1
|
||||
end
|
||||
end
|
||||
index[len + 1] = newline_count + 1
|
||||
_body_line_index_cache[body] = index
|
||||
return index
|
||||
end
|
||||
|
||||
--- Find the end of a marker call (`atom_label(...)` or `atom_offset(...)`).
|
||||
--- Returns the position past the closing `)`, or nil if the token isn't a marker call.
|
||||
--- @param tok string
|
||||
--- @return integer|nil
|
||||
function M.find_marker_call_end(tok)
|
||||
local ident, after = M.read_ident(tok, 1)
|
||||
if not ident then return nil end
|
||||
if ident ~= "atom_label" and ident ~= "atom_offset" then return nil end
|
||||
local paren_pos = M.skip_ws_and_cmt(tok, after)
|
||||
if tok:sub(paren_pos, paren_pos) ~= "(" then return nil end
|
||||
local _, close = M.read_parens(tok, paren_pos)
|
||||
return close
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Section 5: load_word_counts
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
@@ -495,6 +692,7 @@ function M.load_word_counts(metadata_path)
|
||||
local nl = M.find_byte(content, BYTE_NEWLINE, pos)
|
||||
local line_end = nl or (len + 1)
|
||||
local line = content:sub(pos, line_end - 1)
|
||||
-- scan: WORD_COUNT(<name>, <N>)
|
||||
local trimmed = M.trim(line)
|
||||
if trimmed:sub(1, #prefix) == prefix and trimmed:sub(-1) == ")" then
|
||||
local inner = trimmed:sub(#prefix + 1, #trimmed - 1)
|
||||
@@ -509,9 +707,11 @@ function M.load_word_counts(metadata_path)
|
||||
return counts
|
||||
end
|
||||
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- ══════════════════════════════════════════════════
|
||||
-- Section 6: LineIndex (perf fix — replaces the per-call rescan line_of)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- ══════════════════════════════════════════════════
|
||||
|
||||
function M.LineIndex(source)
|
||||
local positions = {}
|
||||
@@ -522,23 +722,19 @@ function M.LineIndex(source)
|
||||
positions[n] = pos
|
||||
end
|
||||
end
|
||||
-- (internal) Binary-search for the line number containing `query_pos`.
|
||||
-- (internal) Binary-search for the line number containing query_pos.
|
||||
local function line_of(query_pos)
|
||||
local lo, hi = 1, n
|
||||
while lo <= hi do
|
||||
local mid = math.floor((lo + hi) / 2)
|
||||
if positions[mid] <= query_pos then
|
||||
lo = mid + 1
|
||||
else
|
||||
hi = mid - 1
|
||||
end
|
||||
if positions[mid] <= query_pos then lo = mid + 1
|
||||
else hi = mid - 1 end
|
||||
end
|
||||
return hi + 1
|
||||
end
|
||||
return line_of
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Section 7: domain tables
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
@@ -559,53 +755,59 @@ M.TAPE_ATOM_MACROS = {
|
||||
["atom_info"] = { kind = "info", binds = false },
|
||||
}
|
||||
|
||||
-- GTE pipeline-fill latency table (static-analysis Phase 1).
|
||||
-- GTE pipeline-fill latency table.
|
||||
--
|
||||
-- For each `gte_cmdw_*` macro in code/duffle/gte.h, the minimum number of consecutive COP2 "nop" words that MUST appear
|
||||
-- before any other COP2 read or non-nop instruction (so the GTE pipeline latency is fully retired).
|
||||
-- Latencies are sourced from the doxygen comments in gte.h
|
||||
-- (e.g. `* @brief Rotate, Translate and Perspective Triple (23 cycles)` with body `Two nop words fill the COP2 pipeline latency`).
|
||||
-- For each `gte_cmdw_*` macro in code/duffle/gte.h, the minimum number of consecutive COP2 "nop" words that MUST appear
|
||||
-- before the command issues so that any preceding `lwc2`/`swc2`/C2 state writes have retired before the GTE starts
|
||||
-- reading its input registers.
|
||||
--
|
||||
-- The check (`scripts/passes/static_analysis.lua :: check_gte_pipeline_fill`) walks each atom body,
|
||||
-- counts the consecutive nop words after every `gte_cmdw_*` invocation, and reports a finding if the count is below this minimum.
|
||||
-- Aliases are dereferenced before lookup (gté_cmdw_rtps_alias -> gte_cmdw_rtps -> 2).
|
||||
--
|
||||
-- Values verified against PSX-SPX gte.txt (rtpt 23cy / 8cy per divide => 2 nops; nclip 8cy => 2 nops; avsz3/avsz4 14cy => 2 nops;
|
||||
-- op single-cycle atomic => 0 nops; mvmva 8cy matrix-vector => 2 nops).
|
||||
-- The check (`scripts/passes/static_analysis.lua :: check_gte_pipeline_fill`) walks each atom body,
|
||||
-- counts the consecutive nop words before every `gte_cmdw_*` invocation, and reports a finding if the count is below this minimum.
|
||||
--
|
||||
-- PRE-FILL vs POST-FILL: this table models PRE-cmdw nops (retiring preceding C2 writes), NOT the post-cmdw input-latch
|
||||
-- window. The PSX-SPX pipeline timings doc (`docs/psx-spx/docs/gtepipelinetimings.md`) measures a DIFFERENT number:
|
||||
-- the smallest N nops between `cop2` and `mtc2` to a specific input register at which the write no longer affects
|
||||
-- the output. For nearly all instructions, inputs latch in the first 0-4 cycles — the GTE snapshots its input
|
||||
-- register file early and works from internal pipeline storage afterward. The documented total cycle count is
|
||||
-- NOT the "do not touch inputs" window; the actual read window is much shorter.
|
||||
--
|
||||
-- The `gte_rtpt()` / `gte_nclip()` wrapper macros in gte.h emit the pre-cmd nops internally (asm_words(nop, nop, ...)),
|
||||
-- but THOSE WRAPPERS ARE NOT USED INSIDE ATOM BODIES in this codebase.
|
||||
-- Every MipsAtom_(name) body uses raw `nop2, gte_cmdw_<X>, ...` form instead — that `nop2,` is the pre-fill this check validates.
|
||||
-- So values here reflect the source-level convention, NOT the wrapper-internal pre-fill.
|
||||
--
|
||||
-- Cycle counts from PSX-SPX `docs/psx-spx/docs/geometrytransformationenginegte.md`:
|
||||
-- cmd PSX-SPX cycles min pre-nops rationale
|
||||
-- rtps 15 2 8c per perspective divide + 6c for IR1..4 + mac write
|
||||
-- rtpt 23 2 3x rtps worth of pipeline depth (per-vertex pipeline fill)
|
||||
-- nclip 8 2 MAC0 write + 5c for sign computation
|
||||
-- avsz3 5 2 5c to compute average + write OTZ (all inputs latch at N=0)
|
||||
-- avsz4 6 2 avsz3 + 1c extra for 4th vertex
|
||||
-- mvmva 8 2 IR1..4 write + matrix work (8c regardless of mx/v/cv selection)
|
||||
-- op 6 0 cross product; output to IR1..3 only (atomic 6c calc, no pre-fill needed)
|
||||
--
|
||||
-- The pre-nop values (2 for most commands) are conservative: PSX-SPX pipeline timings show most inputs latch at N=0-1
|
||||
-- relative to a preceding mtc2, but 2 nops is the gte.h convention for retiring preceding lwc2/swc2 + C2 state.
|
||||
-- OP is set to 0 because it's a short atomic op with no input that needs a long retire window.
|
||||
--
|
||||
-- Aliases are listed separately because source code may use either the alias or the canonical name.
|
||||
M.GTE_PIPELINE_LATENCY = {
|
||||
-- Minimum number of consecutive `nop` words that must appear IMMEDIATELY BEFORE a `gte_cmdw_<X>` invocation
|
||||
-- to retire any preceding `lwc2` / `swc2` / pre-existing C2 state writes before the GTE pipeline starts reading
|
||||
-- Minimum number of consecutive `nop` words that must appear IMMEDIATELY BEFORE a `gte_cmdw_<X>` invocation
|
||||
-- to retire any preceding `lwc2` / `swc2` / pre-existing C2 state writes before the GTE pipeline starts reading
|
||||
-- from V0/V1/V2 or MAC0..3 / OTZ / IR0..3 at the command's issue cycle.
|
||||
--
|
||||
-- Values are from the doxygen comments in code/duffle/gte.h and cross-checked against PSX-SPX `geometrytransformationenginegte.md`:
|
||||
-- cmd cycles min pre-nops rationale
|
||||
-- rtps 14 2 8c per perspective divide + 6c for IR1..4 + mac write
|
||||
-- rptt 22 2 3x rtps worth of pipeline depth
|
||||
-- nclip 7 2 MAC0 write + 5c for sign
|
||||
-- avsz3 14 2 14c to compute average + write OTZ
|
||||
-- avsz4 16 2 avsz3 + 2c extra for avg over 4
|
||||
-- mvmva 8 2 IR1..4 write + matrix work
|
||||
-- op 5 0 output to MAC0 only (atomic 5c calc)
|
||||
--
|
||||
-- The `gte_rtpt()` / `gte_nclip()` / `gte_avsz3()` wrapper macros in gte.h emit the pre-cmd nops internally (asm_words(nop, nop, ...)),
|
||||
-- but THOSE WRAPPERS ARE NOT USED INSIDE ATOM BODIES in this codebase.
|
||||
-- Every MipsAtom_(name) body uses raw `nop2, gte_cmdw_<X>, ...` form instead -- that `nop2,` is the pre-fill this check validates.
|
||||
-- So values here must reflect the source-level convention, NOT the wrapper-internal pre-fill (which is invisible at the source level).
|
||||
--
|
||||
-- Existing clean-atom bodies (cube_g4_face, floor_f3_face, diag_gte) all emit `nop2,` before every `gte_cmdw_<X>` (which matches values >= 2).
|
||||
-- The check passes them all.
|
||||
--
|
||||
-- Aliases are listed separately because source code may use either the alias or the canonical name.
|
||||
-- The check looks up the EXACT macro text, so both forms must be in the table.
|
||||
-- Values are from the doxygen comments in code/duffle/gte.h and cross-checked against
|
||||
-- PSX-SPX `docs/psx-spx/docs/geometrytransformationenginegte.md` (cycle counts) and
|
||||
-- `docs/psx-spx/docs/gtepipelinetimings.md` (input-latch boundaries).
|
||||
|
||||
-- Canonical macros (from code/duffle/gte.h)
|
||||
["gte_cmdw_rtps"] = 2,
|
||||
["gte_cmdw_rtpt"] = 2,
|
||||
["gte_cmdw_nclip"] = 2,
|
||||
["gte_cmdw_op"] = 0,
|
||||
["gte_cmdw_mvmva"] = 2,
|
||||
["gte_cmdw_avsz3"] = 2,
|
||||
["gte_cmdw_avsz4"] = 2,
|
||||
["gte_cmdw_rtps"] = 2, -- RTPS: 15 cycles (PSX-SPX)
|
||||
["gte_cmdw_rtpt"] = 2, -- RTPT: 23 cycles (PSX-SPX)
|
||||
["gte_cmdw_nclip"] = 2, -- NCLIP: 8 cycles (PSX-SPX)
|
||||
["gte_cmdw_op"] = 0, -- OP: 6 cycles, atomic (PSX-SPX)
|
||||
["gte_cmdw_mvmva"] = 2, -- MVMVA: 8 cycles (PSX-SPX)
|
||||
["gte_cmdw_avsz3"] = 2, -- AVSZ3: 5 cycles (PSX-SPX)
|
||||
["gte_cmdw_avsz4"] = 2, -- AVSZ4: 6 cycles (PSX-SPX)
|
||||
|
||||
-- Aliases (must have the same value as their canonical target)
|
||||
["gte_cmdw_rotate_translate_perspective_single"] = 2,
|
||||
@@ -621,7 +823,18 @@ M.GTE_PIPELINE_LATENCY = {
|
||||
}
|
||||
|
||||
-- GP0 packet sizes (total words including the 1-word tag) per GP0 cmd byte.
|
||||
-- Verified against code/duffle/gp.h struct sizes + the set_poly_* macros
|
||||
-- Per PSX-SPX `docs/psx-spx/docs/graphicsprocessingunitgpu.md` §"GPU Render Polygon Commands":
|
||||
-- Each polygon command's word count = 1 (tag/cmd) + per-vertex (vertex + optional color + optional UV).
|
||||
-- F3: cmd + 3 vertices = 4 words; +1 tag = 5
|
||||
-- F4: cmd + 4 vertices = 5 words; +1 tag = 6
|
||||
-- G3: cmd + 3×(color + vertex) = 6 words; +1 tag = 7
|
||||
-- G4: cmd + 4×(color + vertex) = 8 words; +1 tag = 9
|
||||
-- FT3: cmd + tpage + clut + 3×(vertex + UV) = 7 words; +1 tag = 8
|
||||
-- FT4: cmd + tpage + clut + 4×(vertex + UV) = 9 words; +1 tag = 10
|
||||
-- GT3: cmd + tpage + clut + 3×(color + vertex + UV) = 9 words; +1 tag = 10
|
||||
-- GT4: cmd + tpage + clut + 4×(color + vertex + UV) = 12 words; +1 tag = 13
|
||||
--
|
||||
-- Cross-checked against code/duffle/gp.h struct sizes + the set_poly_* macros
|
||||
-- (which encode "len" = "words after tag"):
|
||||
-- set_poly_f3(p) -> set_len(p, 4) -> 5 total GP0 0x20
|
||||
-- set_poly_ft3(p) -> set_len(p, 7) -> 8 total GP0 0x24
|
||||
@@ -666,28 +879,35 @@ M.GP0_MACRO_CONTRIB = {
|
||||
["mac_insert_ot_tag_g4"] = 1,
|
||||
}
|
||||
|
||||
-- Per-macro cycle cost (best-case, no stalls). Used by the static-analysis `count_atom_cycles` pass (Phase 3) to emit per-atom cycle budgets.
|
||||
-- Per-macro cycle cost (best-case, no stalls). Used by the static-analysis pass to emit per-atom cycle budgets.
|
||||
-- The counts cover the EXPANDED instruction sequence the macro emits (NOT just the token it appears as in source).
|
||||
-- For example:
|
||||
-- For example:
|
||||
-- mac_pack_color_word(off, cmd, r, g, b) emits:
|
||||
-- load_upper_i(R_AT, (cmd << 8) | b) -- 1 cycle
|
||||
-- or_i_self(R_AT, (g << 8) | r) -- 1 cycle
|
||||
-- store_word(R_AT, R_PrimCursor, off) -- 1 cycle
|
||||
-- = 3 cycles total
|
||||
--
|
||||
-- mac_yield emits a control-transfer sequence (load_word, add_ui_self, jump_reg, nop)
|
||||
-- which "yields control" the atom body's cycle budget doesn't include the yield's cost (we model it as 0;
|
||||
-- mac_yield emits a control-transfer sequence (load_word, add_ui_self, jump_reg, nop)
|
||||
-- which "yields control" the atom body's cycle budget doesn't include the yield's cost (we model it as 0;
|
||||
-- runtime cost becomes part of the NEXT atom's prologue).
|
||||
--
|
||||
-- GTE command values are the GTE instruction's intrinsic cycles (the latency AFTER any pre-cmd `nop2` has retired).
|
||||
-- When the source emits `nop2, gte_cmdw_X` the nops' cycles are added separately (1+1) plus the gte_cmdw_X value here:
|
||||
-- rtpt = 21 + 2 nops = 23 total cycles (matches PSX-SPX)
|
||||
-- rtps = 12 + 2 nops = 14 total
|
||||
-- nclip = 6 + 2 nops = 8 total
|
||||
-- avsz3 = 12 + 2 nops = 14 total
|
||||
-- avsz4 = 14 + 2 nops = 16 total
|
||||
-- mvmva = 6 + 2 nops = 8 total
|
||||
-- op = 5 (no pre-cmd nops required; single-cycle atomic)
|
||||
-- rtpt = 23 + 2 nops = 25 total cycles (PSX-SPX says 23 cycles for the cmd itself; the nops are pre-fill)
|
||||
-- rtps = 15 + 2 nops = 17 total
|
||||
-- nclip = 8 + 2 nops = 10 total
|
||||
-- avsz3 = 5 + 2 nops = 7 total
|
||||
-- avsz4 = 6 + 2 nops = 8 total
|
||||
-- mvmva = 8 + 2 nops = 10 total
|
||||
-- op = 6 (no pre-cmd nops required; atomic)
|
||||
--
|
||||
-- Note: the "total" above is the pre-fill nops + the GTE intrinsic cycles. PSX-SPX documents the GTE
|
||||
-- intrinsic cycles as the total execution time of the command itself (rtpt=23, rtps=15, nclip=8, etc.).
|
||||
-- The pre-fill nops are a codebase convention for retiring preceding C2 writes, not part of the GTE's
|
||||
-- own execution time. See `docs/psx-spx/docs/geometrytransformationenginegte.md` for the canonical
|
||||
-- per-command cycle counts and `docs/psx-spx/docs/gtepipelinetimings.md` for the hardware-verified
|
||||
-- input-latch boundaries (which show most inputs are safe to clobber after just 0-4 cycles).
|
||||
M.INSTRUCTION_LATENCY = {
|
||||
-- CPU ALU (single-cycle R3000A ops)
|
||||
["nop"] = 1,
|
||||
@@ -750,30 +970,30 @@ M.INSTRUCTION_LATENCY = {
|
||||
["gte_mv_from_ctrl_r"] = 1,
|
||||
["gte_lw"] = 1, ["gte_lwc2"] = 1,
|
||||
["gte_sw"] = 1, ["gte_swc2"] = 1,
|
||||
-- COP2 commands (intrinsic cycles, EXCLUDING the 2 pre-cmd nops that
|
||||
-- the source typically emits as `nop2, gte_cmdw_X`; those nops are
|
||||
-- counted separately via the `nop2` entry above)
|
||||
["gte_cmdw_rtpt"] = 21,
|
||||
["gte_cmdw_rtps"] = 12,
|
||||
["gte_cmdw_nclip"] = 6,
|
||||
["gte_cmdw_avsz3"] = 12,
|
||||
["gte_cmdw_avsz4"] = 14,
|
||||
["gte_cmdw_mvmva"] = 6,
|
||||
["gte_cmdw_op"] = 5,
|
||||
["gte_cmdw_outer_product"] = 5,
|
||||
["gte_cmdw_wedge"] = 5,
|
||||
-- COP2 commands (intrinsic cycles per PSX-SPX, EXCLUDING the 2 pre-cmd nops that
|
||||
-- the source typically emits as `nop2, gte_cmdw_X`; those nops are counted
|
||||
-- separately via the `nop2` entry above)
|
||||
["gte_cmdw_rtpt"] = 23, -- RTPT: 23 cycles (PSX-SPX)
|
||||
["gte_cmdw_rtps"] = 15, -- RTPS: 15 cycles (PSX-SPX)
|
||||
["gte_cmdw_nclip"] = 8, -- NCLIP: 8 cycles (PSX-SPX)
|
||||
["gte_cmdw_avsz3"] = 5, -- AVSZ3: 5 cycles (PSX-SPX)
|
||||
["gte_cmdw_avsz4"] = 6, -- AVSZ4: 6 cycles (PSX-SPX)
|
||||
["gte_cmdw_mvmva"] = 8, -- MVMVA: 8 cycles (PSX-SPX)
|
||||
["gte_cmdw_op"] = 6, -- OP: 6 cycles (PSX-SPX)
|
||||
["gte_cmdw_outer_product"] = 6, -- alias for OP
|
||||
["gte_cmdw_wedge"] = 6, -- alias for OP
|
||||
-- Long-form aliases (same cost as canonical)
|
||||
["gte_cmdw_rotate_translate_perspective_single"] = 12, -- alias for rtps
|
||||
["gte_cmdw_rotate_translate_perspective_triple"] = 21, -- alias for rtpt
|
||||
["gte_cmdw_avg_sort_z4"] = 14, -- alias for avsz4
|
||||
["gte_cmdw_rotate_translate_perspective_single"] = 15, -- alias for rtps
|
||||
["gte_cmdw_rotate_translate_perspective_triple"] = 23, -- alias for rtpt
|
||||
["gte_cmdw_avg_sort_z4"] = 6, -- alias for avsz4
|
||||
-- Non-cmdw aliases from gte.h (these are `#define gte_X gte_cmdw_Y`):
|
||||
["gte_avg_sort_z3"] = 12, -- alias for avsz3
|
||||
["gte_avg_sort_z4"] = 14, -- alias for avsz4
|
||||
["gte_rtps"] = 12, -- alias for rtps
|
||||
["gte_rtpt"] = 21, -- alias for rtpt
|
||||
["gte_nclip"] = 6, -- alias for nclip
|
||||
["gte_avsz3"] = 12,
|
||||
["gte_avsz4"] = 14,
|
||||
["gte_avg_sort_z3"] = 5, -- alias for avsz3
|
||||
["gte_avg_sort_z4"] = 6, -- alias for avsz4
|
||||
["gte_rtps"] = 15, -- alias for rtps
|
||||
["gte_rtpt"] = 23, -- alias for rtpt
|
||||
["gte_nclip"] = 8, -- alias for nclip
|
||||
["gte_avsz3"] = 5,
|
||||
["gte_avsz4"] = 6,
|
||||
-- Legacy single-cycle store helpers (gte_stotz, gte_stsxy3 are 1 cycle)
|
||||
["gte_stotz"] = 1,
|
||||
["gte_stsxy3"] = 1,
|
||||
|
||||
+67
-20
@@ -1,34 +1,76 @@
|
||||
--- duffle_paths.lua — Single-line bootstrap helper for the tape-atom Lua scripts.
|
||||
---
|
||||
--- Each entry script (ps1_meta.lua, word_count_eval.lua, and the 5 passes/*.lua files) starts with:
|
||||
--- Each entry script (ps1_meta.lua + the 7 passes/*.lua files) starts with one of:
|
||||
--- ```lua
|
||||
--- -- Entry script (ps1_meta.lua — `arg[0]` is set):
|
||||
--- local duffle = dofile((arg[0]:match("(.*[/\\])") or "./") .. "duffle_paths.lua")
|
||||
---
|
||||
--- -- Pass module (debug.getinfo path resolution; works both standalone and when require'd):
|
||||
--- local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||
--- local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
--- ```
|
||||
--- That single line: (a) locates this helper via `arg[0]`,
|
||||
--- (b) loads it (which sets `package.path` + `package.cpath` via `git rev-parse`),
|
||||
--- (c) returns the `M` table (a wrapper around the setup function).
|
||||
--- After this line, `require("duffle")` and `require("passes.X")` both resolve normally.
|
||||
---
|
||||
--- That small bootstrap: (a) locates this helper via `arg[0]` / `debug.getinfo`,
|
||||
--- (b) loads it (which sets `package.path` + `package.cpath` via cached `git rev-parse`),
|
||||
--- (c) at the bottom calls `require("duffle")` (now resolvable since `package.path` was just set) and returns the duffle M.
|
||||
--- Net effect: the caller gets the duffle module in one statement; no separate `dofile(...)` + `require("duffle")` dance.
|
||||
---
|
||||
--- Replaces the prior 2-line (entry) or 4-line (pass) pattern that had the call site do its own path resolution + duplicated setup.
|
||||
|
||||
local M = {}
|
||||
|
||||
-- Cache key for the repo root. Stored in `package.loaded` (process-global) so all 8 entry scripts + passes scripts share one git call.
|
||||
-- Without this cache, `git rev-parse --show-toplevel` runs once per script load.
|
||||
-- Cache key for the repo root. Stored in `package.loaded` (process-global) so all 8 entry scripts + passes scripts share one resolution.
|
||||
local CACHE_KEY = "__duffle_repo_root__"
|
||||
|
||||
--- Resolve the repo root via git (cached after first call).
|
||||
--- Returns a normalized path with a trailing forward-slash, or nil if not in a git repo.
|
||||
--- Resolve the repo root from this script's own path. Zero shell spawn.
|
||||
--- `duffle_paths.lua` always lives at `<repo>/scripts/duffle_paths.lua`, so the repo root is the
|
||||
--- parent of the directory containing this script. We derive it directly from `debug.getinfo(1, "S").source`
|
||||
--- (returns `@<path>` for the currently-running chunk).
|
||||
---
|
||||
--- Replaces the prior `io.popen("git rev-parse --show-toplevel")` approach, which cost ~100-180ms per
|
||||
--- LuaJIT process on Windows due to git's CLI startup. The path-derive approach costs <1ms.
|
||||
---
|
||||
--- If this script's path can't be parsed (shouldn't happen — dofile/debug.getinfo always populates source),
|
||||
--- fall back to a defensive walk: starting from this script's directory, walk UP until we find a parent that
|
||||
--- contains a `scripts/` directory. The first match is the repo root.
|
||||
--- @return string|nil
|
||||
local function find_repo_root()
|
||||
if package.loaded[CACHE_KEY] then return package.loaded[CACHE_KEY] end
|
||||
local p = io.popen("git rev-parse --show-toplevel 2>nul")
|
||||
local root
|
||||
if p then root = p:read("*l"); p:close() end
|
||||
if not root or root == "" then return nil end
|
||||
-- Normalize to forward slashes (Windows accepts both, but mixed `\` + `/` confuses LuaJIT's file APIs).
|
||||
root = root:gsub("\\", "/")
|
||||
if not root:match("/$") then root = root .. "/" end
|
||||
package.loaded[CACHE_KEY] = root
|
||||
return root
|
||||
|
||||
local source = debug.getinfo(1, "S").source
|
||||
-- Strip the leading `@` (Lua's dofile marker) and the trailing `/duffle_paths.lua` filename.
|
||||
-- What remains is the directory containing this script, i.e. `<repo>/scripts/` (with trailing slash or not).
|
||||
local scripts_dir = source and source:match("^@?(.*)[/\\]duffle_paths%.lua$")
|
||||
if scripts_dir then
|
||||
-- The repo root is the parent of `scripts/`. Strip the trailing `scripts/` (with or without trailing slash).
|
||||
local root = scripts_dir:gsub("scripts[\\/]?$", "")
|
||||
root = root:gsub("\\", "/")
|
||||
if root == "" then root = "./" end
|
||||
if not root:match("/$") then root = root .. "/" end
|
||||
package.loaded[CACHE_KEY] = root
|
||||
return root
|
||||
end
|
||||
|
||||
-- Defensive fallback: walk UP from this script's directory until we find a parent that contains `scripts/`.
|
||||
-- In practice this branch never fires — debug.getinfo always returns a source for dofile()'d chunks.
|
||||
local lfs = pcall(require, "lfs") and require("lfs") or nil
|
||||
if lfs then
|
||||
local dir = source and source:match("^@?(.*[/\\])") or "./"
|
||||
dir = dir:gsub("\\", "/")
|
||||
while dir and dir ~= "" do
|
||||
local candidate_scripts = dir .. "scripts"
|
||||
if lfs.attributes(candidate_scripts, "mode") == "directory" then
|
||||
dir = dir:gsub("/$", "")
|
||||
package.loaded[CACHE_KEY] = dir .. "/"
|
||||
return dir .. "/"
|
||||
end
|
||||
local parent = dir:match("^(.*)/[^/]+/$")
|
||||
if not parent then break end
|
||||
dir = parent .. "/"
|
||||
end
|
||||
end
|
||||
|
||||
return nil
|
||||
end
|
||||
|
||||
--- Set `package.path` (for `require("duffle")` + `require("passes.X")`) and
|
||||
@@ -54,13 +96,18 @@ function M.setup()
|
||||
.. package.path
|
||||
|
||||
-- lpeg: built by `update_deps.ps1` to `toolchain/lpeg/lpeg.dll`.
|
||||
-- Wire its directory into cpath so `require("lpeg")` resolves.
|
||||
-- lfs: compiled from pcsx-redux's vendored luafilesystem source to `toolchain/lfs/lfs.dll`.
|
||||
-- Wire both directories into cpath so `require("lpeg")` and `require("lfs")` resolve.
|
||||
local lpeg_dir = repo_root .. "toolchain/lpeg/"
|
||||
local lfs_dir = repo_root .. "toolchain/lfs/"
|
||||
package.cpath = lpeg_dir .. "?.dll;"
|
||||
.. lfs_dir .. "?.dll;"
|
||||
.. package.cpath
|
||||
end
|
||||
|
||||
-- Run the setup as a side effect.
|
||||
M.setup()
|
||||
|
||||
return M
|
||||
-- Now that package.path includes scripts/, `require("duffle")` resolves. Return the duffle module
|
||||
-- so callers can do `local duffle = dofile(...duffle_paths.lua)` in one line.
|
||||
return require("duffle")
|
||||
|
||||
+64
-687
@@ -3,6 +3,10 @@
|
||||
--- Validates `MipsAtom_(name) atom_info(atom_bind(Binds_X), atom_reads(...), atom_writes(...)) { ... }` declarations in source files.
|
||||
--- Also reads: `Binds_*` struct declarations (`typedef Struct_(Binds_X) { ... };`)
|
||||
---
|
||||
--- Source scanning: done ONCE upstream by `duffle.scan_source()` (ps1_meta.lua pre-scans each
|
||||
--- source and stashes the result in `src.scan`). This pass is pure: read from the scan, run
|
||||
--- checks, emit findings. No source re-walking.
|
||||
---
|
||||
--- Writes:
|
||||
--- - `<ctx.out_root>/<dir_basename>.errors.h` — one per module, with `#error` directives on findings (the C compile will surface the error)
|
||||
--- - The annotations.txt report is rendered by `passes/report.lua` from the per-module results stashed in `ctx.flags._annot_results`
|
||||
@@ -14,27 +18,12 @@
|
||||
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
||||
-- Uses `debug.getinfo` to find this file's own directory, so it works
|
||||
-- both standalone and when require'd from the orchestrator.
|
||||
local _src = debug.getinfo(1, "S").source:sub(2)
|
||||
local _dir = _src:match("(.*[/\\])") or "./"
|
||||
dofile(_dir .. "../duffle_paths.lua")
|
||||
local duffle = require("duffle")
|
||||
local is_space = duffle.is_space
|
||||
local is_alpha = duffle.is_alpha
|
||||
local is_alnum = duffle.is_alnum
|
||||
local trim = duffle.trim
|
||||
local find_byte = duffle.find_byte
|
||||
local read_file = duffle.read_file
|
||||
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
|
||||
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
local write_file = duffle.write_file
|
||||
local ensure_dir = duffle.ensure_dir
|
||||
local dirname = duffle.dirname
|
||||
local basename_no_ext = duffle.basename_no_ext
|
||||
local skip_str_or_cmt = duffle.skip_str_or_cmt
|
||||
local skip_ws_and_cmt = duffle.skip_ws_and_cmt
|
||||
local read_ident = duffle.read_ident
|
||||
local read_parens = duffle.read_parens
|
||||
local read_braces = duffle.read_braces
|
||||
local scan_to_char = duffle.scan_to_char
|
||||
local split_top_level_commas = duffle.split_top_level_commas
|
||||
|
||||
-- Domain tables (single source of truth in duffle.lua).
|
||||
local WAVE_CONTEXT_REGS = duffle.WAVE_CONTEXT_REGS
|
||||
@@ -42,43 +31,6 @@ local TAPE_ATOM_MACROS = duffle.TAPE_ATOM_MACROS
|
||||
|
||||
local function is_wave_context_reg(n) return WAVE_CONTEXT_REGS[n] ~= nil end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Constants
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Atom declaration + annotation identifiers.
|
||||
local ATOM_DECL = "MipsAtom_"
|
||||
local ATOM_INFO = "atom_info"
|
||||
local STRUCT_TYPE = "Struct_"
|
||||
local PRAGMA_IDENT = "pragma"
|
||||
local PRAGMA_OPERATOR = "_Pragma"
|
||||
|
||||
-- Struct-name prefix + byte size of U4 fields.
|
||||
local BINDS_PREFIX = "Binds_"
|
||||
local BINDS_PREFIX_LEN = 6 -- = #BINDS_PREFIX
|
||||
local U4_TYPE = "U4"
|
||||
local U4_BYTES = 4 -- sizeof(U4)
|
||||
local BINDS_FIELD_PREFIX = "R_" -- wave-context register name prefix
|
||||
|
||||
-- TAPE_WORDS pragma keys (the third token after #pragma).
|
||||
local WORDS_KEY = "words"
|
||||
local WORDS_KEY_PREFIX = "words=" -- the per-macro `words=N` form
|
||||
local WORDS_KEY_PREFIX_LEN = 6 -- = #WORDS_KEY_PREFIX
|
||||
local TAPE_ATOM_WORDS_KEY = "tape_atom words" -- the _Pragma form
|
||||
|
||||
-- ASCII byte values used in tokenization.
|
||||
local BYTE_NEWLINE = 10
|
||||
local BYTE_SPACE = 32
|
||||
local BYTE_DQUOTE = 34
|
||||
local BYTE_EQUALS = 61
|
||||
local BYTE_OPEN_PAREN = 40
|
||||
local BYTE_OPEN_BRACE = 123
|
||||
local BYTE_OPEN_BRACK = 91
|
||||
local BYTE_CLOSE_PAREN = 41
|
||||
local BYTE_CLOSE_BRACE = 125
|
||||
local BYTE_CLOSE_BRACK = 93
|
||||
local BYTE_COMMA = 44
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Type declarations
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
@@ -88,6 +40,7 @@ local BYTE_COMMA = 44
|
||||
--- @field text string -- the full source text
|
||||
--- @field dir string -- the directory containing the source
|
||||
--- @field basename string -- filename without extension
|
||||
--- @field scan table -- pre-scanned SourceScan payload (from duffle.scan_source)
|
||||
|
||||
--- @class PassCtx
|
||||
--- @field sources SourceFile[]
|
||||
@@ -107,10 +60,6 @@ local BYTE_COMMA = 44
|
||||
--- @field errors table[]
|
||||
--- @field warnings table[]
|
||||
|
||||
--- @class Atom
|
||||
--- @field line integer -- source line of the MipsAtom_ declaration
|
||||
--- @field name string -- atom name (e.g. "cube_g4_face")
|
||||
|
||||
--- @class AtomAnnotation
|
||||
--- @field line integer -- source line of the atom_info call
|
||||
--- @field macro string -- the macro name (always "atom_info" in the new shape)
|
||||
@@ -119,616 +68,64 @@ local BYTE_COMMA = 44
|
||||
--- @field binds string|nil -- Binds_X name if any
|
||||
--- @field reads string[] -- R_* names (read targets)
|
||||
--- @field writes string[] -- R_* names (write targets)
|
||||
--- @field error string|nil -- error message if annotation was malformed
|
||||
--- @field errors string[] -- nested errors from per-arg validation
|
||||
|
||||
--- @class BindsField
|
||||
--- @field name string -- field name
|
||||
--- @field offset integer -- byte offset within the Binds_X struct
|
||||
|
||||
--- @class BindsStruct
|
||||
--- @field name string -- struct name (e.g. "Binds_Floor")
|
||||
--- @field line integer -- source line of the typedef
|
||||
--- @field bytes integer -- total byte size
|
||||
--- @field fields BindsField[] -- the field list
|
||||
|
||||
--- @class MacroEntry
|
||||
--- @field name string -- macro name (e.g. "mac_format_f3_color")
|
||||
--- @field line integer -- source line of the TAPE_WORDS pragma
|
||||
--- @field words integer -- declared word count
|
||||
|
||||
--- @class Finding
|
||||
--- @field line integer -- source line (or 0 for pass-level)
|
||||
--- @field msg string -- finding message
|
||||
|
||||
--- @class AnnotatedResult
|
||||
--- @field atoms Atom[]
|
||||
--- @field atoms AtomEntry[]
|
||||
--- @field annots AtomAnnotation[]
|
||||
--- @field macros MacroEntry[]
|
||||
--- @field binds BindsStruct[]
|
||||
--- @field binds BindsEntry[]
|
||||
--- @field errors Finding[]
|
||||
--- @field warnings Finding[]
|
||||
--- @field info Finding[]
|
||||
--- @field pragmas table -- reserved (currently always nil; legacy compat)
|
||||
|
||||
--- ════════════════════════════════════════════════════════════════════════════
|
||||
-- split helpers
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Split a string at top-level commas. Used inside TAPE_ATOM_* macro
|
||||
--- bodies where nested parens/braces/brackets are possible.
|
||||
--- @param s string
|
||||
--- @return string[]
|
||||
local function split_csv_top(s)
|
||||
local tokens = {}
|
||||
local pos = 1
|
||||
local chunk_a = 1
|
||||
local depth = 0
|
||||
local str_len = #s
|
||||
while pos <= str_len do
|
||||
local ch = s:byte(pos)
|
||||
if ch == BYTE_OPEN_PAREN or ch == BYTE_OPEN_BRACE or ch == BYTE_OPEN_BRACK then
|
||||
depth = depth + 1
|
||||
pos = pos + 1
|
||||
elseif ch == BYTE_CLOSE_PAREN or ch == BYTE_CLOSE_BRACE or ch == BYTE_CLOSE_BRACK then
|
||||
depth = depth - 1
|
||||
pos = pos + 1
|
||||
elseif ch == BYTE_COMMA and depth == 0 then
|
||||
tokens[#tokens + 1] = s:sub(chunk_a, pos - 1)
|
||||
pos = pos + 1
|
||||
chunk_a = pos
|
||||
else
|
||||
pos = pos + 1
|
||||
end
|
||||
end
|
||||
local last = s:sub(chunk_a)
|
||||
if trim(last) ~= "" then tokens[#tokens + 1] = last end
|
||||
return tokens
|
||||
end
|
||||
|
||||
--- Split a string into whitespace-separated tokens.
|
||||
--- @param s string
|
||||
--- @return string[]
|
||||
local function split_ws(s)
|
||||
local tokens = {}
|
||||
local pos = 1
|
||||
local n = 1
|
||||
local len = #s
|
||||
while pos <= len do
|
||||
-- Skip whitespace.
|
||||
while pos <= len and is_space(s:sub(pos, pos)) do pos = pos + 1 end
|
||||
if pos > len then break end
|
||||
local chunk_a = pos
|
||||
-- Take non-whitespace run.
|
||||
while pos <= len and not is_space(s:sub(pos, pos)) do pos = pos + 1 end
|
||||
tokens[n] = s:sub(chunk_a, pos - 1)
|
||||
n = n + 1
|
||||
end
|
||||
return tokens
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Parse TAPE_ATOM_ANNOT(...) calls
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Recognize a `atom_bind(...)`, `atom_reads(...)`, or `atom_writes(...)` sub-call embedded inside an atom_info arg list.
|
||||
-- Returns the kind ("atom_bind" / "atom_reads" / "atom_writes") and the inner content, or nil if the token isn't a recognized sub-call form.
|
||||
-- Flattened via a prefix lookup instead of a nested if/elseif chain.
|
||||
local REGS_CALL_PREFIX = {
|
||||
["atom_writes("] = { kind = "atom_writes", inner_offset = 13 },
|
||||
["atom_reads("] = { kind = "atom_reads", inner_offset = 12 },
|
||||
["atom_bind("] = { kind = "atom_bind", inner_offset = 11, single_ident = true },
|
||||
}
|
||||
|
||||
local function parse_regs_call(s)
|
||||
if s:sub(-1) ~= ")" then return nil end
|
||||
-- Try longest prefix first so "atom_writes(" wins over "atom_reads(" when both 12-char prefixes would otherwise match.
|
||||
-- Lengths:
|
||||
-- atom_writes( = 12 chars, offset 13
|
||||
-- atom_reads( = 11 chars, offset 12
|
||||
-- atom_bind( = 10 chars, offset 11
|
||||
local spec = REGS_CALL_PREFIX[s:sub(1, 12)]
|
||||
if not spec then spec = REGS_CALL_PREFIX[s:sub(1, 11)] end
|
||||
if not spec then spec = REGS_CALL_PREFIX[s:sub(1, 10)] end
|
||||
if not spec then return nil end
|
||||
local inner = s:sub(spec.inner_offset, -2)
|
||||
if spec.single_ident then
|
||||
-- atom_bind takes a single Binds_* type ident. Trim and pass through.
|
||||
return spec.kind, trim(inner)
|
||||
end
|
||||
return spec.kind, inner
|
||||
end
|
||||
|
||||
-- Resolve any phase_* / R_* alias macros in a register list.
|
||||
-- (Phase / region / cadence aliases have been dropped. Kept as an identity function so callers can stay uniform.)
|
||||
local function resolve_reg_aliases(regs) return regs end
|
||||
|
||||
-- Parse a comma-separated inner content (e.g. inside atom_reads(...)) into a list of trimmed identifiers with aliases resolved.
|
||||
local function parse_regs_list(inner)
|
||||
local out = {}
|
||||
for _, r in ipairs(split_csv_top(inner)) do
|
||||
local trimmed = trim(r)
|
||||
if trimmed ~= "" then out[#out + 1] = trimmed end
|
||||
end
|
||||
return resolve_reg_aliases(out)
|
||||
end
|
||||
|
||||
-- Parse a single token (from split_csv_top) into an arg entry.
|
||||
-- Three forms: register-list call, bare identifier, "other" (preserved as text).
|
||||
local function parse_arg_token(s)
|
||||
local kind, inner = parse_regs_call(s)
|
||||
if kind then
|
||||
if kind == "atom_bind" then
|
||||
return { kind = kind, value = inner } -- single ident, not a list
|
||||
end
|
||||
return { kind = kind, value = parse_regs_list(inner) }
|
||||
end
|
||||
local id = read_ident(s, 1)
|
||||
if id and trim(s) == id then
|
||||
return { kind = "ident", value = id }
|
||||
end
|
||||
return { kind = "other", value = s }
|
||||
end
|
||||
|
||||
--- Extract identifier args from a parenthesized group.
|
||||
--- Returns a list of {kind, value} pairs where kind is one of:
|
||||
--- "ident" -- a bare identifier (e.g. phase_work)
|
||||
--- "atom_reads" -- an atom_reads(...) call: value is the register list
|
||||
--- "atom_writes" -- an atom_writes(...) call: value is the register list
|
||||
--- "other" -- something we can't classify (preserved as text)
|
||||
local function parse_atom_annot_args(inner)
|
||||
local args = {}
|
||||
for _, tok in ipairs(split_csv_top(inner)) do
|
||||
local s = trim(tok)
|
||||
if s ~= "" then
|
||||
args[#args + 1] = parse_arg_token(s)
|
||||
end
|
||||
end
|
||||
return args
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Parse TAPE_WORDS(mac_X, N) pragma directives
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Skip preprocessor directives (lines starting with `#`).
|
||||
--- Returns the position past the newline at the end of the line.
|
||||
--- @param source string
|
||||
--- @param pos integer
|
||||
--- @return integer
|
||||
local function skip_preprocessor_line(source, pos)
|
||||
local str_len = #source
|
||||
local scan = pos
|
||||
while scan <= str_len and source:byte(scan) ~= BYTE_NEWLINE do
|
||||
scan = scan + 1
|
||||
end
|
||||
return scan + 1
|
||||
end
|
||||
|
||||
--- Parse `_Pragma("mac_X tape_atom words=N")` (operator form).
|
||||
--- @param source string
|
||||
--- @param ident_pos integer -- position of the `_Pragma` ident
|
||||
--- @param after_ident integer -- position just past the ident
|
||||
--- @return MacroEntry|nil, integer -- (entry or nil, new source position)
|
||||
local function parse_pragma_operator(source, ident_pos, after_ident)
|
||||
local open_paren = skip_ws_and_cmt(source, after_ident)
|
||||
if source:byte(open_paren) ~= BYTE_OPEN_PAREN then
|
||||
return nil, open_paren + 1
|
||||
end
|
||||
local str, str_end = read_parens(source, open_paren)
|
||||
-- scan: _Pragma(<string>)
|
||||
str = trim(str)
|
||||
if str:sub(1, 1) ~= '"' or str:sub(-1) ~= '"' then
|
||||
return nil, str_end
|
||||
end
|
||||
local inner = str:sub(2, -2)
|
||||
local space = find_byte(inner, BYTE_SPACE, 1)
|
||||
if not space then return nil, str_end end
|
||||
local name = inner:sub(1, space - 1)
|
||||
local rest = inner:sub(space + 1)
|
||||
local eq = find_byte(rest, BYTE_EQUALS, 1)
|
||||
if not eq then return nil, str_end end
|
||||
local key = trim(rest:sub(1, eq - 1))
|
||||
local val = trim(rest:sub(eq + 1))
|
||||
if key ~= TAPE_ATOM_WORDS_KEY and key ~= WORDS_KEY then return nil, str_end end
|
||||
return {
|
||||
line = source:sub(1, ident_pos) and 0 or 0, -- see line_of below
|
||||
name = name,
|
||||
words = tonumber(val) or 0,
|
||||
}, str_end
|
||||
end
|
||||
|
||||
--- Parse `#pragma mac_X tape_atom words=N` (directive form).
|
||||
--- @param source string
|
||||
--- @param ident_pos integer -- position of the `pragma` ident
|
||||
--- @param after_ident integer -- position just past the ident
|
||||
--- @return MacroEntry|nil, integer -- (entry or nil, new source position)
|
||||
local function parse_pragma_directive(source, ident_pos, after_ident)
|
||||
local str_len = #source
|
||||
local rest_start = skip_ws_and_cmt(source, after_ident)
|
||||
local eol = rest_start
|
||||
while eol <= str_len and source:byte(eol) ~= BYTE_NEWLINE do
|
||||
eol = eol + 1
|
||||
end
|
||||
local line_text = trim(source:sub(rest_start, eol - 1))
|
||||
-- scan: #pragma <mac_name> tape_atom words=<N>
|
||||
local tokens = split_ws(line_text)
|
||||
local entry
|
||||
if #tokens >= 3 and tokens[2] == "tape_atom" and tokens[3]:sub(1, WORDS_KEY_PREFIX_LEN) == WORDS_KEY_PREFIX then
|
||||
entry = {
|
||||
name = tokens[1],
|
||||
words = tonumber(tokens[3]:sub(WORDS_KEY_PREFIX_LEN + 1)) or 0,
|
||||
}
|
||||
elseif #tokens >= 2 and tokens[2]:sub(1, WORDS_KEY_PREFIX_LEN) == WORDS_KEY_PREFIX then
|
||||
entry = {
|
||||
name = tokens[1],
|
||||
words = tonumber(tokens[2]:sub(WORDS_KEY_PREFIX_LEN + 1)) or 0,
|
||||
}
|
||||
end
|
||||
if entry then
|
||||
local line_of = duffle.LineIndex(source)
|
||||
entry.line = line_of(ident_pos)
|
||||
end
|
||||
return entry, eol
|
||||
end
|
||||
|
||||
--- Find every `TAPE_WORDS(mac_X, N)` pragma in source.
|
||||
--- Accepts both forms:
|
||||
--- `_Pragma("mac_X tape_atom words=N")` (operator form)
|
||||
--- `#pragma mac_X tape_atom words=N` (directive form)
|
||||
--- @param source string
|
||||
--- @return MacroEntry[]
|
||||
local function find_macro_word_annotations(source)
|
||||
local out = {}
|
||||
local pos = 1
|
||||
local str_len = #source
|
||||
while pos <= str_len do
|
||||
pos = skip_ws_and_cmt(source, pos)
|
||||
if pos > str_len then break end
|
||||
|
||||
-- Skip preprocessor directives (lines starting with #).
|
||||
if source:byte(pos) == 35 then -- '#'
|
||||
pos = skip_preprocessor_line(source, pos)
|
||||
else
|
||||
local ident, after_ident = read_ident(source, pos)
|
||||
-- scan: <ident>
|
||||
if not ident then
|
||||
pos = pos + 1
|
||||
elseif ident == PRAGMA_OPERATOR then
|
||||
-- scan: _Pragma(...)
|
||||
local entry, new_pos = parse_pragma_operator(source, pos, after_ident)
|
||||
if entry then
|
||||
local line_of = duffle.LineIndex(source)
|
||||
entry.line = line_of(pos)
|
||||
out[#out + 1] = entry
|
||||
end
|
||||
pos = new_pos
|
||||
elseif ident == PRAGMA_IDENT then
|
||||
-- scan: #pragma <mac_name> tape_atom words=<N>
|
||||
local entry, new_pos = parse_pragma_directive(source, pos, after_ident)
|
||||
if entry then out[#out + 1] = entry end
|
||||
pos = new_pos
|
||||
else
|
||||
pos = after_ident
|
||||
end
|
||||
end
|
||||
end
|
||||
return out
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Parse `typedef Struct_(Binds_X) { ... };` declarations
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Walk a `Binds_X` body string and extract U4 fields (name + byte offset).
|
||||
--- Only U4 fields are tracked (Binds_* are always word arrays in this
|
||||
--- codebase -- pointers stored as U4, indices as U4, etc.).
|
||||
--- @param body string -- the brace-delimited body (without the braces)
|
||||
--- @return BindsField[] -- field list
|
||||
--- @return integer -- total byte size
|
||||
local function parse_binds_body(body)
|
||||
local fields = {}
|
||||
local byte_off = 0
|
||||
local pos = 1
|
||||
local body_len = #body
|
||||
while pos <= body_len do
|
||||
pos = skip_ws_and_cmt(body, pos)
|
||||
if pos > body_len then break end
|
||||
local type_ident, type_after = read_ident(body, pos)
|
||||
if not type_ident then
|
||||
pos = pos + 1
|
||||
elseif type_ident == U4_TYPE then
|
||||
local field_after = skip_ws_and_cmt(body, type_after)
|
||||
local fid, fafter = read_ident(body, field_after)
|
||||
if fid then
|
||||
fields[#fields + 1] = { name = fid, offset = byte_off }
|
||||
byte_off = byte_off + U4_BYTES
|
||||
end
|
||||
pos = fafter or (type_after + 1)
|
||||
else
|
||||
pos = type_after + 1
|
||||
end
|
||||
end
|
||||
return fields, byte_off
|
||||
end
|
||||
|
||||
--- Try to parse a `typedef Struct_(Binds_X) { ... };` declaration.
|
||||
--- Returns the parsed BindsStruct (if the form matched) and the new source position.
|
||||
--- If the form didn't match, returns nil + a position to continue scanning from.
|
||||
--- @param source string
|
||||
--- @param ident_pos integer -- position of the `typedef` ident start
|
||||
--- @param after_typedef integer -- position just past `typedef`
|
||||
--- @param line_of fun(pos: integer): integer
|
||||
--- @return BindsStruct|nil, integer
|
||||
local function parse_typedef_binds(source, ident_pos, after_typedef, line_of)
|
||||
local after_type = skip_ws_and_cmt(source, after_typedef)
|
||||
local type_ident, after_type_ident = read_ident(source, after_type)
|
||||
-- scan: typedef <type_ident>
|
||||
if type_ident ~= STRUCT_TYPE then
|
||||
return nil, after_type_ident or (after_type + 1)
|
||||
end
|
||||
|
||||
local open_paren = skip_ws_and_cmt(source, after_type_ident)
|
||||
if source:byte(open_paren) ~= BYTE_OPEN_PAREN then
|
||||
return nil, open_paren + 1
|
||||
end
|
||||
|
||||
local inner, after_paren = read_parens(source, open_paren)
|
||||
-- scan: typedef Struct_(<name>)
|
||||
local name = trim(inner)
|
||||
|
||||
local brace = scan_to_char(source, "{", after_paren)
|
||||
-- scan: typedef Struct_(<name>) {
|
||||
if not brace then return nil, open_paren + 1 end
|
||||
|
||||
local body, after_brace = read_braces(source, brace)
|
||||
-- scan: typedef Struct_(<name>) { <fields> }
|
||||
local fields, bytes = parse_binds_body(body)
|
||||
|
||||
-- Only emit Binds_* structs (other Struct_ typedefs are ignored).
|
||||
if name:sub(1, BINDS_PREFIX_LEN) ~= BINDS_PREFIX then
|
||||
return nil, after_brace
|
||||
end
|
||||
|
||||
return {
|
||||
line = line_of(ident_pos),
|
||||
name = name,
|
||||
fields = fields,
|
||||
bytes = bytes,
|
||||
}, after_brace
|
||||
end
|
||||
|
||||
--- Find every `Binds_*` struct declaration.
|
||||
--- @param source string
|
||||
--- @return BindsStruct[]
|
||||
local function find_binds_structs(source)
|
||||
local line_of = duffle.LineIndex(source)
|
||||
local out = {}
|
||||
local pos = 1
|
||||
local str_len = #source
|
||||
while pos <= str_len do
|
||||
pos = skip_ws_and_cmt(source, pos)
|
||||
if pos > str_len then break end
|
||||
|
||||
if source:byte(pos) == 35 then -- '#'
|
||||
pos = skip_preprocessor_line(source, pos)
|
||||
else
|
||||
local ident, after_ident = read_ident(source, pos)
|
||||
-- scan: <ident>
|
||||
if not ident then
|
||||
pos = pos + 1
|
||||
elseif ident == "typedef" then
|
||||
-- scan: typedef Struct_(<name>) { <fields> }
|
||||
local binds_struct, new_pos = parse_typedef_binds(source, pos, after_ident, line_of)
|
||||
if binds_struct then out[#out + 1] = binds_struct end
|
||||
pos = new_pos
|
||||
else
|
||||
pos = after_ident
|
||||
end
|
||||
end
|
||||
end
|
||||
return out
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Find every MipsAtom_(name) { ... } declaration in source
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Read the next identifier token from `s` starting at `pos`, where the identifier is a contiguous run of `[a-zA-Z0-9_]`
|
||||
--- characters (no underscore-starting alpha-only constraint).
|
||||
--- Returns the ident + the position just past it, or nil + pos if no identifier starts there.
|
||||
--- @param s string
|
||||
--- @param pos integer
|
||||
--- @return string|nil, integer
|
||||
local function read_alnum_ident(s, pos)
|
||||
local str_len = #s; while pos <= str_len and is_space(s:sub(pos, pos)) do pos = pos + 1 end
|
||||
local start = pos; while pos <= str_len and is_alnum(s:sub(pos, pos)) do pos = pos + 1 end
|
||||
if pos == start then return nil, pos end
|
||||
return s:sub(start, pos - 1), pos
|
||||
end
|
||||
|
||||
--- Find every `MipsAtom_(name)` declaration in source.
|
||||
--- (Just the name + source line; the body is parsed separately by `parse_mips_atom`.)
|
||||
--- @param source string
|
||||
--- @return Atom[]
|
||||
local function find_atom_names(source)
|
||||
local line_of = duffle.LineIndex(source)
|
||||
local out = {}
|
||||
local pos = 1
|
||||
local str_len = #source
|
||||
while pos <= str_len do
|
||||
pos = skip_ws_and_cmt(source, pos)
|
||||
if pos > str_len then break end
|
||||
|
||||
local ident, after_ident = read_ident(source, pos)
|
||||
-- scan: <ident>
|
||||
if not ident then
|
||||
pos = pos + 1
|
||||
elseif ident ~= ATOM_DECL then
|
||||
pos = after_ident
|
||||
else
|
||||
local open_paren = skip_ws_and_cmt(source, after_ident)
|
||||
if source:byte(open_paren) ~= BYTE_OPEN_PAREN then
|
||||
pos = open_paren + 1
|
||||
else
|
||||
local inner, after_paren = read_parens(source, open_paren)
|
||||
-- scan: MipsAtom_(<name>)
|
||||
local name, _ = read_alnum_ident(inner, 1)
|
||||
if name and name ~= "" then
|
||||
out[#out + 1] = { line = line_of(pos), name = name }
|
||||
end
|
||||
local brace = scan_to_char(source, "{", after_paren)
|
||||
-- scan: MipsAtom_(<name>) {
|
||||
if brace then
|
||||
local _, after_brace = read_braces(source, brace)
|
||||
-- scan: MipsAtom_(<name>) { <body> }
|
||||
pos = after_brace
|
||||
else
|
||||
pos = open_paren + 1
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
return out
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Find atom annotations (atom_annot / atom_init / atom_setup / atom_commit
|
||||
-- / atom_bind / atom_terminate)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- True iff the parsed arg is a register-list call (any recognized form).
|
||||
local function is_regs_arg(a) return a and (a.kind == "atom_reads" or a.kind == "atom_writes" or a.kind == "regs") end
|
||||
|
||||
--- Per-macro arg-shape handlers. Each takes (entry, args) and mutates Per-atom_info sub-call dispatch.
|
||||
--- Each takes (entry, args) and mutates entry.{reads, writes, binds, errors}.
|
||||
--- All sub-calls are order-independent; each is dispatched on its `kind` (atom_bind / atom_reads / atom_writes) when parsed.
|
||||
local ANNOT_ARG_HANDLERS = {}
|
||||
|
||||
-- atom_info(atom_bind(Binds_X), atom_reads(...), atom_writes(...))
|
||||
function ANNOT_ARG_HANDLERS.info(entry, args)
|
||||
for _, arg in ipairs(args) do
|
||||
if arg.kind == "atom_bind" then entry.binds = arg.value
|
||||
elseif arg.kind == "atom_reads" then entry.reads = arg.value
|
||||
elseif arg.kind == "atom_writes" then entry.writes = arg.value
|
||||
elseif arg.kind == "ident" then
|
||||
-- Reserved for future phase tokens. Currently ignored.
|
||||
-- (Could be reintroduced as `phase_*` sub-calls of atom_info.)
|
||||
else
|
||||
entry.errors[#entry.errors + 1] = string.format("unexpected atom_info arg kind=%s value=%s", arg.kind, tostring(arg.value))
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
-- Build a new annotation entry with the standard shape.
|
||||
local function new_annot_entry(line, ident, name, kind)
|
||||
return {
|
||||
line = line,
|
||||
macro = ident,
|
||||
name = name,
|
||||
kind = kind,
|
||||
binds = nil,
|
||||
reads = {},
|
||||
writes = {},
|
||||
errors = {},
|
||||
}
|
||||
end
|
||||
|
||||
--- Try to parse an `atom_info(...)` call right after the `MipsAtom_(name)` parens.
|
||||
--- Returns the annotation entry (if present) and the new source position past the atom_info call.
|
||||
--- Returns nil if no atom_info follows.
|
||||
--- @param source string
|
||||
--- @param atom_name string
|
||||
--- @param after_mipsatom_paren integer -- position past the MipsAtom_(...) close paren
|
||||
--- @param line_of fun(pos: integer): integer
|
||||
--- @return AtomAnnotation|nil, integer -- (entry or nil, new source position)
|
||||
local function parse_atom_info_call(source, atom_name, after_mipsatom_paren, line_of)
|
||||
local lookahead = skip_ws_and_cmt(source, after_mipsatom_paren)
|
||||
local look_ident, look_after = read_ident(source, lookahead)
|
||||
-- scan: MipsAtom_(<name>) <look_ident>
|
||||
if look_ident ~= ATOM_INFO then return nil, after_mipsatom_paren end
|
||||
|
||||
local info_open = skip_ws_and_cmt(source, look_after)
|
||||
if source:byte(info_open) ~= BYTE_OPEN_PAREN then return nil, info_open + 1 end
|
||||
|
||||
local info_inner, info_after = read_parens(source, info_open)
|
||||
-- scan: MipsAtom_(<name>) atom_info(<binds>, <reads>, <writes>)
|
||||
local args = parse_atom_annot_args(info_inner)
|
||||
local entry = new_annot_entry(line_of(lookahead), ATOM_INFO, atom_name, "info")
|
||||
ANNOT_ARG_HANDLERS.info(entry, args)
|
||||
return entry, info_after
|
||||
end
|
||||
|
||||
--- Find every `MipsAtom_(name) atom_info(...) { ... };` annotation in source.
|
||||
--- Returns a list of annotation entries. Atoms without a following `atom_info(...)` call produce NO entry
|
||||
--- (atoms without annotations are valid in the new minimal shape).
|
||||
--- @param source string
|
||||
--- @return AtomAnnotation[]
|
||||
local function find_atom_annotations(source)
|
||||
local line_of = duffle.LineIndex(source)
|
||||
local annots = {}
|
||||
local pos = 1
|
||||
local str_len = #source
|
||||
while pos <= str_len do
|
||||
pos = skip_ws_and_cmt(source, pos)
|
||||
if pos > str_len then break end
|
||||
|
||||
-- Skip preprocessor directives (lines starting with #).
|
||||
if source:byte(pos) == 35 then -- '#'
|
||||
pos = skip_preprocessor_line(source, pos)
|
||||
else
|
||||
local ident, after_ident = read_ident(source, pos)
|
||||
-- scan: <ident>
|
||||
if not ident then
|
||||
pos = pos + 1
|
||||
elseif ident == ATOM_DECL then
|
||||
local open_paren = skip_ws_and_cmt(source, after_ident)
|
||||
if source:byte(open_paren) ~= BYTE_OPEN_PAREN then
|
||||
pos = open_paren + 1
|
||||
else
|
||||
local inner, after_paren = read_parens(source, open_paren)
|
||||
-- scan: MipsAtom_(<name>)
|
||||
local name, _ = read_alnum_ident(inner, 1)
|
||||
|
||||
local entry, new_pos = parse_atom_info_call(source, name, after_paren, line_of)
|
||||
-- scan: MipsAtom_(<name>) atom_info(<binds>, <reads>, <writes>)
|
||||
if entry then annots[#annots + 1] = entry end
|
||||
pos = new_pos
|
||||
|
||||
-- Skip past the body { ... } if present.
|
||||
local brace = scan_to_char(source, "{", pos)
|
||||
-- scan: MipsAtom_(<name>) atom_info(...) {
|
||||
if brace then
|
||||
local _, after_brace = read_braces(source, brace)
|
||||
-- scan: MipsAtom_(<name>) atom_info(...) { <body> }
|
||||
pos = after_brace
|
||||
end
|
||||
end
|
||||
else
|
||||
pos = after_ident
|
||||
end
|
||||
end
|
||||
end
|
||||
return annots
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Validation (ported from tape_atom_annotation_pass.lua:1193-1405)
|
||||
-- Validation
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
--
|
||||
-- Pure check: read from src.scan, run validations, emit findings.
|
||||
-- No source walking; no parsing. The scan was done once upstream.
|
||||
|
||||
--- Validate one source against its pre-scanned SourceScan payload.
|
||||
--- @param ctx PassCtx
|
||||
--- @param src SourceFile
|
||||
--- @return AnnotatedResult
|
||||
local function validate(ctx, src)
|
||||
local source = src.text
|
||||
local scan = src.scan
|
||||
|
||||
local annots = find_atom_annotations(source)
|
||||
local macros = find_macro_word_annotations(source)
|
||||
local binds = find_binds_structs(source)
|
||||
local atoms = find_atom_names(source)
|
||||
-- Project the pre-scanned atoms to the AtomEntry shape this pass needs.
|
||||
local atoms = {}
|
||||
for _, a in ipairs(scan.atoms) do
|
||||
if a.kind == "atom" then
|
||||
atoms[#atoms + 1] = { line = a.line, name = a.raw_name }
|
||||
end
|
||||
end
|
||||
|
||||
-- Project the pre-scanned atom_infos to AtomAnnotation shape.
|
||||
local annots = {}
|
||||
for _, info in ipairs(scan.atom_infos) do
|
||||
annots[#annots + 1] = {
|
||||
line = info.info_line,
|
||||
macro = "atom_info",
|
||||
name = info.atom_name,
|
||||
kind = "info",
|
||||
binds = info.binds,
|
||||
reads = info.reads or {},
|
||||
writes = info.writes or {},
|
||||
errors = {},
|
||||
}
|
||||
end
|
||||
|
||||
-- Index atoms by name for lookup.
|
||||
local atom_index = {}
|
||||
for _, a in ipairs(atoms) do atom_index[a.name] = a end
|
||||
|
||||
-- Index binds by name for lookup.
|
||||
local binds_index = {}
|
||||
for _, b in ipairs(binds) do binds_index[b.name] = b end
|
||||
for _, b in ipairs(scan.binds) do binds_index[b.name] = b end
|
||||
|
||||
local errors = {}
|
||||
local warnings = {}
|
||||
@@ -736,9 +133,7 @@ local function validate(ctx, src)
|
||||
|
||||
-- 1. Every annotated atom must exist as a real MipsAtom_ declaration.
|
||||
for _, a in ipairs(annots) do
|
||||
if a.error then
|
||||
errors[#errors + 1] = {line = a.line, msg = a.error}
|
||||
elseif not atom_index[a.name] then
|
||||
if not atom_index[a.name] then
|
||||
errors[#errors + 1] = {
|
||||
line = a.line,
|
||||
msg = string.format("annotation for '%s' has no matching MipsAtom_(%s) { ... }", a.name, a.name),
|
||||
@@ -751,11 +146,11 @@ local function validate(ctx, src)
|
||||
end
|
||||
end
|
||||
|
||||
-- 2. Every atom may have AT MOST ONE annotation (no duplicates).
|
||||
-- 2. Every atom may have AT MOST ONE annotation (no duplicates).
|
||||
-- (Atoms with ZERO annotations are valid in the new minimal shape.)
|
||||
local count_per_atom = {}
|
||||
for _, a in ipairs(annots) do
|
||||
if a.name and not a.error then
|
||||
if a.name then
|
||||
count_per_atom[a.name] = (count_per_atom[a.name] or 0) + 1
|
||||
end
|
||||
end
|
||||
@@ -768,8 +163,7 @@ local function validate(ctx, src)
|
||||
end
|
||||
end
|
||||
|
||||
-- 3. (Phase validity check DROPPED. Phases were removed from the annotation DSL.
|
||||
-- They may be reintroduced later as sub-calls of atom_info, at which point ordering checks will go here.)
|
||||
-- 3. (Phase validity check DROPPED. Phases were removed from the annotation DSL.)
|
||||
|
||||
-- 4. BIND atoms must reference a real Binds_* struct.
|
||||
for _, a in ipairs(annots) do
|
||||
@@ -777,7 +171,7 @@ local function validate(ctx, src)
|
||||
if not binds_index[a.binds] then
|
||||
-- Demoted from error to warning (2026-07-10): the same condition is now caught by passes/static_analysis.lua's
|
||||
-- check_abi_handoff() as an error. Emitting a warning here keeps the annotation pass from being stop-on-error
|
||||
-- for the common test-fixture case, while still surfacing the issue in the report.
|
||||
-- for the common test-fixture case, while still surfacing the issue in the report.
|
||||
-- The static-analysis report remains the source of truth for build-stopping errors.
|
||||
warnings[#warnings + 1] = {
|
||||
line = a.line,
|
||||
@@ -791,8 +185,6 @@ local function validate(ctx, src)
|
||||
for _, a in ipairs(annots) do
|
||||
if a.binds and binds_index[a.binds] then
|
||||
local bs = binds_index[a.binds]
|
||||
local field_names = {}
|
||||
for _, f in ipairs(bs.fields) do field_names[f.name] = true end
|
||||
|
||||
for _, f in ipairs(bs.fields) do
|
||||
local candidate = "R_" .. f.name
|
||||
@@ -829,7 +221,6 @@ local function validate(ctx, src)
|
||||
|
||||
-- 7. TAPE_WORDS(mac_X, N) ↔ WORD_COUNT(mac_X, N) drift.
|
||||
-- Three outcomes: missing (error), mismatch (error), match (info).
|
||||
-- Flattened via early-return-style helper instead of 3-way elseif.
|
||||
local function check_macro_drift(m, declared)
|
||||
if not declared then
|
||||
errors[#errors + 1] = {
|
||||
@@ -850,7 +241,7 @@ local function validate(ctx, src)
|
||||
msg = string.format("OK: %s = %d words", m.name, m.words),
|
||||
}
|
||||
end
|
||||
for _, m in ipairs(macros) do
|
||||
for _, m in ipairs(scan.macros) do
|
||||
check_macro_drift(m, ctx.shared.word_counts[m.name])
|
||||
end
|
||||
|
||||
@@ -858,15 +249,14 @@ local function validate(ctx, src)
|
||||
info[#info + 1] = {
|
||||
line = 0,
|
||||
msg = string.format("scanned: %d atom(s), %d annotation(s), %d macro-word-decl(s), %d binds struct(s)",
|
||||
#atoms, #annots, #macros, #binds),
|
||||
#atoms, #annots, #scan.macros, #scan.binds),
|
||||
}
|
||||
|
||||
return {
|
||||
atoms = atoms,
|
||||
annots = annots,
|
||||
macros = macros,
|
||||
pragmas = pragmas,
|
||||
binds = binds,
|
||||
macros = scan.macros,
|
||||
binds = scan.binds,
|
||||
errors = errors,
|
||||
warnings = warnings,
|
||||
info = info,
|
||||
@@ -876,18 +266,12 @@ end
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Per-DIRECTORY (per-module) output: errors.h + annotations.txt
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
--
|
||||
-- Per-source reports were the old behavior; each source in the same directory produced its own <basename>.errors.h + <basename>.annotations.txt,
|
||||
-- which flooded build/gen/ with one report per header.
|
||||
-- Aggregates per-DIRECTORY (one errors.h + one annotations.txt per module basename).
|
||||
-- Directories with zero atoms/annotations are skipped (no file emitted).
|
||||
|
||||
--- Render `<dir_basename>.errors.h` with `#error` directives for every error found across all sources in the directory.
|
||||
--- Render `<dir_basename>.errors.h` with `#error` directives for every error found across all sources in the directory.
|
||||
--- Empty directories (no errors, no atoms) produce no file.
|
||||
local function emit_module_errors_h(ctx, dir_basename, atoms_count, errors, sources)
|
||||
if ctx.dry_run then return nil end
|
||||
if atoms_count == 0 and #errors == 0 then
|
||||
-- Skip dirs with nothing to report
|
||||
return nil
|
||||
end
|
||||
local out_path = ctx.out_root .. "/" .. dir_basename .. ".errors.h"
|
||||
@@ -900,8 +284,6 @@ local function emit_module_errors_h(ctx, dir_basename, atoms_count, errors, sour
|
||||
if #errors == 0 then
|
||||
lines[#lines + 1] = "// annotation pass OK"
|
||||
else
|
||||
-- Prefix each error with the source basename for traceability
|
||||
-- in the C compile log.
|
||||
for _, e in ipairs(errors) do
|
||||
local src_tag = ""
|
||||
if e.source then
|
||||
@@ -925,7 +307,6 @@ local function emit_module_annotations_stub(ctx, dir, dir_basename, atoms_count)
|
||||
dir_basename = dir_basename,
|
||||
atoms_count = atoms_count,
|
||||
}
|
||||
-- annotations.txt is written by report.lua
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
@@ -937,7 +318,6 @@ end
|
||||
local M = {}
|
||||
|
||||
-- Expose `validate` for downstream passes (e.g. report.lua) that need to re-render the per-source results into a per-MODULE report.
|
||||
-- Keeping it as a single shared function avoids the duplication that an earlier version of report.lua had.
|
||||
M.validate = validate
|
||||
|
||||
--- @param ctx PassCtx
|
||||
@@ -947,25 +327,24 @@ function M.run(ctx)
|
||||
local errors = {}
|
||||
local warnings = {}
|
||||
|
||||
-- Per-DIRECTORY (per-module) aggregation. Group sources by `src.dir`, validate every source in the dir, then emit ONE errors.h per dir
|
||||
-- (skipping dirs with no atoms AND no errors).
|
||||
-- The actual annotations.txt is rendered by passes/report.lua from the stashed per-module results below.
|
||||
local by_dir = {}
|
||||
for _, src in ipairs(ctx.sources) do
|
||||
by_dir[src.dir] = by_dir[src.dir] or {}
|
||||
table.insert(by_dir[src.dir], src)
|
||||
end
|
||||
-- Per-DIRECTORY (per-module) aggregation. Group sources by `src.dir`,
|
||||
-- validate every source in the dir, then emit ONE errors.h per dir.
|
||||
-- `ctx.by_dir` is pre-computed in build_ctx (shared across all passes).
|
||||
local by_dir = ctx.by_dir or duffle.group_sources_by_dir(ctx.sources)
|
||||
|
||||
for dir, dir_sources in pairs(by_dir) do
|
||||
-- Dir basename = last component of `dir` ("code/duffle" -> "duffle").
|
||||
local dir_basename = dir:match("([^/\\]+)$") or dir
|
||||
|
||||
-- Aggregate validate() results across the directory.
|
||||
local dir_atoms = 0
|
||||
local dir_errors = {}
|
||||
local dir_warnings = {}
|
||||
-- Per-source validate() results, cached for the report pass (it reads from this instead of re-validating each source).
|
||||
ctx.flags = ctx.flags or {}
|
||||
ctx.flags._annot_source_results = ctx.flags._annot_source_results or {}
|
||||
for _, src in ipairs(dir_sources) do
|
||||
local result = validate(ctx, src)
|
||||
result.source = src.path -- tag for downstream rendering
|
||||
ctx.flags._annot_source_results[src.path] = result -- stash so report.lua reads from cache instead of re-running validate()
|
||||
dir_atoms = dir_atoms + #result.atoms
|
||||
for _, e in ipairs(result.errors) do
|
||||
dir_errors[#dir_errors + 1] = { line = e.line, msg = e.msg, source = src.path }
|
||||
@@ -977,13 +356,11 @@ function M.run(ctx)
|
||||
end
|
||||
end
|
||||
|
||||
-- Emit one errors.h per dir.
|
||||
local err_path = emit_module_errors_h(ctx, dir_basename, dir_atoms, dir_errors, dir_sources)
|
||||
if err_path then
|
||||
table.insert(outputs, { errors_h = err_path })
|
||||
end
|
||||
|
||||
-- Stash for report pass.
|
||||
emit_module_annotations_stub(ctx, dir, dir_basename, dir_atoms)
|
||||
end
|
||||
|
||||
|
||||
+80
-199
@@ -1,8 +1,12 @@
|
||||
--- passes/components.lua — Component-macro header generator.
|
||||
---
|
||||
--- Walks every source for `MipsAtomComp_(ac_X) { body }` (and the function-form `MipsAtomComp_Proc_(ac_X, { body })`) declarations and
|
||||
--- emits a per-directory `<dir_basename>.macs.h` containing one `#define mac_X(sig) \` macro per component + `WORD_COUNT(mac_X, N)`
|
||||
--- entries for downstream offset computation.
|
||||
--- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`)
|
||||
--- for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations, then does
|
||||
--- per-source backward lookups for the function-args string (from the preceding `FI_ MipsAtom ac_X(...)`
|
||||
--- function declaration) and the preceding comment block (for LSP/IntelliSense signature docs).
|
||||
---
|
||||
--- Emits a per-directory `<dir_basename>.macs.h` containing one `#define mac_X(sig) \` macro per component
|
||||
--- + `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
|
||||
---
|
||||
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
|
||||
--- Lua 5.3 compatible.
|
||||
@@ -20,14 +24,14 @@
|
||||
-- Module-scope requires + package.path setup
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Resolve `arg[0]` to an absolute-ish script directory so that `require("duffle")` resolves against `scripts/` regardless of CWD.
|
||||
-- Bootstrap: same as entry scripts. See `ps1_meta.lua` for the rationale.
|
||||
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
||||
-- Uses `debug.getinfo` to find this file's own directory, so it works
|
||||
-- both standalone and when require'd from the orchestrator.
|
||||
local _src = debug.getinfo(1, "S").source:sub(2)
|
||||
local _dir = _src:match("(.*[/\\])") or "./"
|
||||
dofile(_dir .. "../duffle_paths.lua")
|
||||
local duffle = require("duffle")
|
||||
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
|
||||
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
local word_count_eval = require("word_count_eval")
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
@@ -36,7 +40,6 @@ local word_count_eval = require("word_count_eval")
|
||||
|
||||
-- Atom component declaration identifiers.
|
||||
local ATOM_COMP_PROC = "MipsAtomComp_Proc_"
|
||||
local ATOM_COMP = "MipsAtomComp_"
|
||||
local MIPS_ATOM = "MipsAtom" -- prefix on the function declaration that wraps an AtomComp_Proc_
|
||||
|
||||
-- Component-name prefixes.
|
||||
@@ -61,6 +64,7 @@ local GEN_SUBDIR = "gen"
|
||||
--- @field text string -- the full source text
|
||||
--- @field dir string -- the directory containing the source
|
||||
--- @field basename string -- filename without extension
|
||||
--- @field scan table -- pre-scanned SourceScan payload (from duffle.scan_source)
|
||||
|
||||
--- @class PassCtx
|
||||
--- @field sources SourceFile[] -- all source files in the build
|
||||
@@ -72,7 +76,7 @@ local GEN_SUBDIR = "gen"
|
||||
--- @field upstream table<string, table> -- per-pass upstream outputs
|
||||
--- @field flags table -- CLI flags
|
||||
--- @field dry_run boolean -- if true, compute but don't write
|
||||
--- @field verbose boolean -- if true, log diagnostic info
|
||||
--- @field verbose boolean -- log diagnostic info
|
||||
|
||||
--- @class PassResult
|
||||
--- @field outputs table[] -- {kind=, path=} entries describing emit files
|
||||
@@ -90,37 +94,6 @@ local GEN_SUBDIR = "gen"
|
||||
-- Local helpers (file I/O + path normalization)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Write content to disk in binary mode so LF line endings are preserved on Windows
|
||||
-- (text mode would convert LF -> CRLF, breaking byte-identical diffs against git-tracked gen/*.macs.h files which are stored as LF).
|
||||
-- @param path string
|
||||
-- @param content string
|
||||
local function write_file_lf(path, content)
|
||||
local f = io.open(path, "wb")
|
||||
if not f then error("Cannot write " .. path) end
|
||||
f:write(content); f:close()
|
||||
end
|
||||
|
||||
-- Convert a (possibly relative) path to an absolute Windows path.
|
||||
-- The pre-rework output's "// Source:" comment line used the absolute path (e.g. "C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h");
|
||||
-- If we want byte-identical output, we must normalize relative -> absolute before emitting that comment.
|
||||
-- @param path string
|
||||
-- @return string
|
||||
local function to_absolute_path(path)
|
||||
if #path >= 2 and path:sub(2, 2) == ":" then
|
||||
-- Already absolute; normalize slashes for consistency.
|
||||
return (path:gsub("/", "\\"))
|
||||
end
|
||||
local p = io.popen("cd")
|
||||
if not p then return path end
|
||||
local cwd = p:read("*l")
|
||||
p:close()
|
||||
if not cwd then return path end
|
||||
-- Normalize forward slashes to backslashes (Windows convention) on both the cwd AND the relative path tail, so the join is uniform.
|
||||
cwd = cwd:gsub("/", "\\")
|
||||
local tail = (path:gsub("/", "\\"))
|
||||
return cwd .. "\\" .. tail
|
||||
end
|
||||
|
||||
local M = {}
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
@@ -372,122 +345,30 @@ local function extract_arg_names(args_str)
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Component scanner (bare + function forms)
|
||||
-- Component projection (read from pre-scanned SourceScan)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Parse the inner content of an `AtomComp_(name, ...)` call.
|
||||
-- Returns (name, body_or_nil) — `body_or_nil` is non-nil iff this is the function-form `MipsAtomComp_Proc_(name, { body })` invocation.
|
||||
-- @param inner string -- the content between ( and ) of the AtomComp_ call
|
||||
--- @return string|nil, string|nil
|
||||
local function parse_atomcomp_inner(inner)
|
||||
local tokens = duffle.split_top_level_commas(inner)
|
||||
if #tokens == 1 then
|
||||
return duffle.trim(tokens[1]), nil
|
||||
elseif #tokens == 2 then
|
||||
local name = duffle.trim(tokens[1])
|
||||
local body_raw = duffle.trim(tokens[2])
|
||||
-- Strip leading { and trailing } if present.
|
||||
local body
|
||||
if #body_raw >= 2 and body_raw:sub(1, 1) == "{" and body_raw:sub(-1) == "}" then
|
||||
body = duffle.trim(body_raw:sub(2, -2))
|
||||
else
|
||||
body = body_raw
|
||||
end
|
||||
return name, body
|
||||
end
|
||||
return nil, nil
|
||||
end
|
||||
|
||||
-- (internal) Try to extract a bare-form `MipsAtomComp_(ac_X)` declaration.
|
||||
-- Bare form: `MipsAtomComp_(ac_X) { body }` — body comes from the brace block AFTER the parens.
|
||||
-- @param source string
|
||||
-- @param name string -- the `ac_X` ident from the parens
|
||||
--- @param ident_pos integer -- position of the `MipsAtomComp_` ident start
|
||||
--- @param after_paren integer -- position just past the closing `)`
|
||||
--- @param line_of fun(pos: integer): integer
|
||||
--- @param args string|nil -- function-args from preceding function decl
|
||||
--- @param comment string -- preceding comment block
|
||||
--- @return Component|nil, integer -- the component + new source position
|
||||
local function make_bare_component(source, name, ident_pos, after_paren, line_of, args, comment)
|
||||
local brace = duffle.scan_to_char(source, "{", after_paren)
|
||||
-- scan: <ident>(<name>) {
|
||||
if not brace then return nil, after_paren + 1 end
|
||||
local body, after_brace = duffle.read_braces(source, brace)
|
||||
-- scan: <ident>(<name>) { <body> }
|
||||
return {
|
||||
line = line_of(ident_pos),
|
||||
name = name:sub(AC_PREFIX_LEN + 1), -- strip "ac_" prefix
|
||||
body = body,
|
||||
args = args,
|
||||
comment = comment,
|
||||
}, after_brace
|
||||
end
|
||||
|
||||
-- (internal) Build the function-form `MipsAtomComp_Proc_` component. Body
|
||||
-- came from inside the parens; no following brace block.
|
||||
local function make_proc_component(name, body, ident_pos, line_of, args, comment)
|
||||
return {
|
||||
line = line_of(ident_pos),
|
||||
name = name:sub(AC_PREFIX_LEN + 1),
|
||||
body = body,
|
||||
args = args,
|
||||
comment = comment,
|
||||
}
|
||||
end
|
||||
|
||||
--- Find every `MipsAtomComp_(ac_<X>) { body }` declaration in source.
|
||||
--- Supports BOTH the bare form and the function form:
|
||||
--- Bare: `MipsAtomComp_(ac_X) { body }`
|
||||
--- Function: `MipsAtomComp_Proc_(ac_X, { body })` (with a preceding
|
||||
--- `"FI_ MipsAtom ac_X(args)"` function declaration)
|
||||
---
|
||||
--- @param source string
|
||||
--- @return Component[]
|
||||
local function find_component_atoms(source)
|
||||
local line_of = duffle.LineIndex(source)
|
||||
local out = {}
|
||||
local pos = 1
|
||||
local src_len = #source
|
||||
while pos <= src_len do
|
||||
pos = duffle.skip_ws_and_cmt(source, pos)
|
||||
if pos > src_len then break end
|
||||
|
||||
local ident, after_ident = duffle.read_ident(source, pos)
|
||||
-- scan: <ident>
|
||||
local is_comp = ident == ATOM_COMP or ident == ATOM_COMP_PROC
|
||||
if not ident then
|
||||
pos = pos + 1
|
||||
elseif not is_comp then
|
||||
pos = after_ident
|
||||
else
|
||||
local open_paren = duffle.skip_ws_and_cmt(source, after_ident)
|
||||
if source:sub(open_paren, open_paren) ~= "(" then
|
||||
pos = open_paren + 1
|
||||
else
|
||||
local inner, after_paren = duffle.read_parens(source, open_paren)
|
||||
-- scan: <ident>(<args>)
|
||||
local name, body = parse_atomcomp_inner(inner)
|
||||
-- scan: <ident>(<name>) OR <ident>(<name>, { <body> })
|
||||
if not name or name:sub(1, AC_PREFIX_LEN) ~= AC_PREFIX then
|
||||
pos = open_paren + 1
|
||||
else
|
||||
local args = find_function_args_for(source, name, open_paren)
|
||||
local comment = preceding_comment_block(source, pos)
|
||||
if body == nil then
|
||||
-- Bare form: body comes from the brace block after the parens.
|
||||
-- scan: <ident>(<name>) {
|
||||
local comp, new_pos = make_bare_component(source, name, pos, after_paren, line_of, args, comment)
|
||||
-- scan: <ident>(<name>) { <body> }
|
||||
if comp then out[#out + 1] = comp end
|
||||
pos = new_pos
|
||||
else
|
||||
-- Function form: body was inside the parens.
|
||||
-- scan: <ident>(<name>, { <body> })
|
||||
out[#out + 1] = make_proc_component(name, body, pos, line_of, args, comment)
|
||||
pos = after_paren
|
||||
end
|
||||
end
|
||||
end
|
||||
-- Project pre-scanned MipsAtomComp_ / MipsAtomComp_Proc_ entries into Component shape.
|
||||
-- Does per-source backward lookups for args (preceding function decl) and comment (preceding comment block).
|
||||
-- Carries `body_tokens` forward from scan-source so word_count_rec reads from the precomputed table
|
||||
-- instead of calling duffle.tokenize_body again.
|
||||
-- @param source string -- the full source text (needed for backward lookups)
|
||||
-- @param scan table -- SourceScan from duffle.scan_source
|
||||
-- @return Component[]
|
||||
local function project_components(source, scan)
|
||||
local out = {}
|
||||
for _, a in ipairs(scan.atoms) do
|
||||
if a.kind == "comp_bare" or a.kind == "comp_proc" then
|
||||
local args = find_function_args_for(source, a.raw_name, a.ident_pos)
|
||||
local comment = preceding_comment_block(source, a.ident_pos)
|
||||
out[#out + 1] = {
|
||||
line = a.line,
|
||||
name = a.name,
|
||||
body = a.body,
|
||||
body_tokens = a.body_tokens,
|
||||
args = args,
|
||||
comment = comment,
|
||||
}
|
||||
end
|
||||
end
|
||||
return out
|
||||
@@ -499,7 +380,7 @@ end
|
||||
|
||||
-- Convert `//` line comments to `/* */` block comments in a token.
|
||||
--
|
||||
-- C macros use `\` line-continuations; a `//` comment before `\` would consume the continuation,
|
||||
-- C macros use `\` line-continuations; a `//` comment before `\` would consume the continuation,
|
||||
-- breaking the macro. We convert `//` to `/* */` so the multi-line macro structure is preserved.
|
||||
--
|
||||
-- Skips `//` sequences that are inside string or character literals
|
||||
@@ -553,8 +434,8 @@ local function strip_mac_prefix(ident)
|
||||
return ident
|
||||
end
|
||||
|
||||
-- (internal) Recursive word-count lookup. `cache` is the memoization table across all calls to `compute_component_word_count`;
|
||||
-- the in-progress -1 sentinel detects cycles (A -> B -> A).
|
||||
-- (internal) Recursive word-count lookup. `cache` is the memoization table shared across all components
|
||||
-- in a single source's `count_all_components` pass; the in-progress -1 sentinel detects cycles (A -> B -> A).
|
||||
-- @param name string -- the component name (without `mac_`)
|
||||
-- @param comp_by_name table<string, Component>
|
||||
-- @param wc table<string, integer>
|
||||
@@ -567,8 +448,9 @@ local function word_count_rec(name, comp_by_name, wc, cache)
|
||||
local n
|
||||
if cc then
|
||||
n = 0
|
||||
for _, t in ipairs(duffle.split_top_level_commas(cc.body)) do
|
||||
local trimmed = duffle.trim(t)
|
||||
local tokens = cc.body_tokens
|
||||
for _, t in ipairs(tokens) do
|
||||
local trimmed = t.tok
|
||||
if trimmed ~= "" then
|
||||
local lookup = strip_mac_prefix(duffle.read_ident(trimmed, 1))
|
||||
if lookup and comp_by_name[lookup] then
|
||||
@@ -591,25 +473,25 @@ local function word_count_rec(name, comp_by_name, wc, cache)
|
||||
return n
|
||||
end
|
||||
|
||||
--- Compute the word count of a component body, accounting for macro expansion.
|
||||
--- Each comma-separated entry in the body is a "slot" that contributes its own word count.
|
||||
--- For most entries (regular MIPS instructions) the count is 1.
|
||||
--- For `mac_Y(...)` calls, the count is the word count of `mac_Y` (recursive lookup through `components`).
|
||||
--- For encoding macros with a known multi-word count (e.g. `mask_upper` = 2),
|
||||
--- the count is taken from `word_counts`.
|
||||
--- Compute word counts for every component in `components` in a single pass.
|
||||
--- The name-lookup table + memoization cache are built ONCE (per source) instead of per-component,
|
||||
--- so the cache survives across siblings and a component's recursive `mac_Y(...)` references hit memoized values
|
||||
--- instead of re-walking the body. Previously each call rebuilt both tables (O(N) tables per call → O(N^2)).
|
||||
---
|
||||
--- The lookup is memoized via `word_count_rec` to avoid infinite recursion (e.g. if two components referenced each other).
|
||||
--- This is the same algorithm as the original `tape_atom_annotation_pass.lua` (commit 7d20a4d).
|
||||
--- Cycle detection (A -> B -> A) is preserved via the in-progress `-1` sentinel in `cache`.
|
||||
---
|
||||
--- @param c Component
|
||||
--- @param components Component[]
|
||||
--- @param wc table<string, integer>
|
||||
--- @return integer
|
||||
local function compute_component_word_count(c, components, wc)
|
||||
--- @return table<string, integer> -- map of component name (without `mac_`) -> word count
|
||||
local function count_all_components(components, wc)
|
||||
local comp_by_name = {}
|
||||
for _, cc in ipairs(components) do comp_by_name[cc.name] = cc end
|
||||
local cache = {}
|
||||
return word_count_rec(c.name, comp_by_name, wc, cache)
|
||||
local cache = {}
|
||||
local counts = {}
|
||||
for _, c in ipairs(components) do
|
||||
counts[c.name] = word_count_rec(c.name, comp_by_name, wc, cache)
|
||||
end
|
||||
return counts
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
@@ -640,12 +522,7 @@ end
|
||||
--- @param body string
|
||||
--- @return string[]
|
||||
local function tokens_from_body(body)
|
||||
local out = {}
|
||||
for _, t in ipairs(duffle.split_top_level_commas(body)) do
|
||||
local trimmed = duffle.trim(t)
|
||||
if trimmed ~= "" then out[#out + 1] = trimmed end
|
||||
end
|
||||
return out
|
||||
return duffle.tokenize_body_simple(body)
|
||||
end
|
||||
|
||||
--- Determine the macro signature: function-args list (function form) or variadic-ignored (bare form).
|
||||
@@ -682,13 +559,13 @@ local function emit_macro_body(lines, c, sig, tokens)
|
||||
strip_trailing_continuation(lines)
|
||||
end
|
||||
|
||||
--- Build the list of lines for one component
|
||||
--- Build the list of lines for one component
|
||||
--- (signature comment, `#define mac_X(...)` line with backslash-continued tokens, then `WORD_COUNT(mac_X, N)` entry).
|
||||
--- @param c Component
|
||||
--- @param components Component[]
|
||||
--- @param wc table<string, integer>
|
||||
--- @return string[] -- list of lines for this component
|
||||
local function build_component_lines(c, components, wc)
|
||||
local function build_component_lines(c, counts)
|
||||
local lines = {}
|
||||
|
||||
if c.comment and c.comment ~= "" then
|
||||
@@ -699,7 +576,8 @@ local function build_component_lines(c, components, wc)
|
||||
|
||||
local tokens = tokens_from_body(c.body)
|
||||
local sig = signature_from_args(c.args)
|
||||
local n = compute_component_word_count(c, components, wc)
|
||||
-- Direct lookup against the per-source precomputed `counts` table (built once by count_all_components).
|
||||
local n = counts[c.name]
|
||||
|
||||
if n > 0 then
|
||||
emit_macro_body(lines, c, sig, tokens)
|
||||
@@ -716,19 +594,19 @@ end
|
||||
-- Per-source emit logic
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Build the boilerplate header lines (the `#ifdef INTELLISENSE_DIRECTIVES` block,
|
||||
-- Build the boilerplate header lines (the `#ifdef INTELLISENSE_DIRECTIVES` block,
|
||||
-- the `// Auto-generated` comment, the `// Source:` line, and the self-contained `WORD_COUNT` macro definition).
|
||||
-- @param src SourceFile
|
||||
-- @return string[]
|
||||
local function header_boilerplate(src)
|
||||
return {
|
||||
-- #pragma once wrapped in #ifdef INTELLISENSE_DIRECTIVES, matching the convention in lottes_tape.h.
|
||||
-- #pragma once wrapped in #ifdef INTELLISENSE_DIRECTIVES, matching the convention in lottes_tape.h.
|
||||
-- The build does manual unity includes (the user controls include order), so the pragma is only active for IDE/tooling.
|
||||
"#ifdef INTELLISENSE_DIRECTIVES",
|
||||
"#pragma once",
|
||||
"#endif",
|
||||
"// Auto-generated by tape_atom_annotation_pass.lua — DO NOT EDIT",
|
||||
"// Source: " .. to_absolute_path(src.path),
|
||||
"// Auto-generated by ps1_meta.lua — DO NOT EDIT",
|
||||
"// Source: " .. duffle.to_absolute_path(src.path),
|
||||
"// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)",
|
||||
"",
|
||||
-- Self-contained: define WORD_COUNT if not already defined.
|
||||
@@ -742,8 +620,8 @@ local function header_boilerplate(src)
|
||||
end
|
||||
|
||||
-- Compute the output path for one source's `.macs.h` file.
|
||||
-- The pre-rework convention uses the *directory* basename
|
||||
-- (not the source file basename) — e.g. `code/duffle/lottes_tape.h` produces `code/duffle/gen/duffle.macs.h`.
|
||||
-- The pre-rework convention uses the *directory* basename
|
||||
-- (not the source file basename) — e.g. `code/duffle/lottes_tape.h` produces `code/duffle/gen/duffle.macs.h`.
|
||||
-- This matches what the C codebase #includes.
|
||||
-- @param src SourceFile
|
||||
-- @return string -- the output directory
|
||||
@@ -762,16 +640,16 @@ end
|
||||
--- @param ctx PassCtx
|
||||
--- @param src SourceFile
|
||||
--- @param components Component[]
|
||||
--- @param counts table<string, integer> -- precomputed word counts (from count_all_components)
|
||||
--- @return string|nil -- path to the written file (nil if no components)
|
||||
local function emit_component_macros_h(ctx, src, components)
|
||||
local function emit_component_macros_h(ctx, src, components, counts)
|
||||
if #components == 0 then return nil end
|
||||
|
||||
local out_dir, out_path = compute_macs_h_path(src)
|
||||
local lines = header_boilerplate(src)
|
||||
|
||||
local wc = ctx.shared.word_counts
|
||||
for _, c in ipairs(components) do
|
||||
for _, l in ipairs(build_component_lines(c, components, wc)) do
|
||||
for _, l in ipairs(build_component_lines(c, counts)) do
|
||||
lines[#lines + 1] = l
|
||||
end
|
||||
end
|
||||
@@ -784,7 +662,7 @@ local function emit_component_macros_h(ctx, src, components)
|
||||
end
|
||||
|
||||
duffle.ensure_dir(out_dir)
|
||||
write_file_lf(out_path, content)
|
||||
duffle.write_file_lf(out_path, content)
|
||||
print(string.format(" -> %s", out_path))
|
||||
return out_path
|
||||
end
|
||||
@@ -793,14 +671,15 @@ end
|
||||
-- Pass entry
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- (internal) Extend `ctx.shared.word_counts` with this source's component macros
|
||||
-- (internal) Extend `ctx.shared.word_counts` with this source's component macros
|
||||
-- so offsets sees them without re-reading the file.
|
||||
-- @param ctx PassCtx
|
||||
-- @param components Component[]
|
||||
local function update_shared_word_counts(ctx, components)
|
||||
-- @param counts table<string, integer> -- precomputed word counts (from count_all_components)
|
||||
local function update_shared_word_counts(ctx, components, counts)
|
||||
local wc = ctx.shared.word_counts
|
||||
for _, c in ipairs(components) do
|
||||
wc["mac_" .. c.name] = compute_component_word_count(c, components, wc)
|
||||
wc["mac_" .. c.name] = counts[c.name]
|
||||
end
|
||||
end
|
||||
|
||||
@@ -812,13 +691,15 @@ function M.run(ctx)
|
||||
local warnings = {}
|
||||
|
||||
for _, src in ipairs(ctx.sources) do
|
||||
-- find_component_atoms operates on src.text
|
||||
local components = find_component_atoms(src.text)
|
||||
-- project_components reads from src.scan + does backward lookups on src.text
|
||||
local components = project_components(src.text, src.scan)
|
||||
if #components > 0 then
|
||||
local macs_path = emit_component_macros_h(ctx, src, components)
|
||||
-- Compute word counts for ALL components once (was: rebuilt per call inside the helpers).
|
||||
local counts = count_all_components(components, ctx.shared.word_counts)
|
||||
local macs_path = emit_component_macros_h(ctx, src, components, counts)
|
||||
if macs_path then
|
||||
outputs[#outputs + 1] = { macs_h = macs_path }
|
||||
update_shared_word_counts(ctx, components)
|
||||
update_shared_word_counts(ctx, components, counts)
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
+61
-230
@@ -1,8 +1,9 @@
|
||||
--- passes/offsets.lua — Branch-offset generator.
|
||||
---
|
||||
--- Scans every source for `MipsAtom_(name) { ... }` (and the raw `MipsCode code_<name> { ... }` form) declarations,
|
||||
--- computes the word offset from each `atom_offset(F, T)` marker to its target `atom_label(T)` declaration,
|
||||
--- and emits `<dir_basename>.offsets.h` with one `#define _atom_offset_F_T = N` per branch.
|
||||
--- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`)
|
||||
--- for `MipsAtom_(name)` and `MipsCode code_<name>` declarations, computes the word offset
|
||||
--- from each `atom_offset(F, T)` marker to its target `atom_label(T)` declaration, and emits
|
||||
--- `<dir_basename>.offsets.h` with one `#define _atom_offset_F_T = N` per branch.
|
||||
---
|
||||
--- The offset is `target_word - branch_word - 1` (the standard MIPS branch-immediate encoding: branch_offset = relative_pc_in_words - 1).
|
||||
---
|
||||
@@ -13,15 +14,14 @@
|
||||
-- Module-scope requires + package.path setup
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Resolve `arg[0]` to an absolute-ish script directory so that `require("duffle")` resolves against `scripts/` regardless of CWD.
|
||||
-- Bootstrap: see `ps1_meta.lua` for the rationale.
|
||||
-- Bootstrap: same as entry scripts. See `ps1_meta.lua` for the rationale.
|
||||
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
||||
-- Uses `debug.getinfo` to find this file's own directory, so it works
|
||||
-- both standalone and when require'd from the orchestrator.
|
||||
local _src = debug.getinfo(1, "S").source:sub(2)
|
||||
local _dir = _src:match("(.*[/\\])") or "./"
|
||||
dofile(_dir .. "../duffle_paths.lua")
|
||||
local duffle = require("duffle")
|
||||
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
|
||||
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
local word_count_eval = require("word_count_eval")
|
||||
local count_token_words = word_count_eval.count_token_words
|
||||
|
||||
@@ -29,21 +29,6 @@ local count_token_words = word_count_eval.count_token_words
|
||||
-- Constants
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- C qualifier keywords that may precede a `MipsAtom_` declaration
|
||||
-- (and should be skipped by `skip_qualifiers`).
|
||||
local QUALIFIER_KEYWORDS = {
|
||||
["static"] = true, ["const"] = true, ["volatile"] = true,
|
||||
["extern"] = true, ["register"] = true, ["auto"] = true,
|
||||
["inline"] = true, ["typedef"] = true,
|
||||
["internal"] = true, ["LP_"] = true, ["global"] = true, ["gkknown"] = true,
|
||||
}
|
||||
|
||||
-- Atom declaration identifiers.
|
||||
local ATOM_PREFIX = "MipsAtom_"
|
||||
local CODE_DECL = "MipsCode"
|
||||
local CODE_RAW_PREFIX = "code_" -- raw atom form: `MipsCode code_<name> { ... }`
|
||||
local CODE_RAW_PREFIX_LEN = 5 -- = #CODE_RAW_PREFIX
|
||||
|
||||
-- Marker-call identifiers inside atom bodies.
|
||||
local LABEL_MARKER = "atom_label"
|
||||
local OFFSET_MARKER = "atom_offset"
|
||||
@@ -64,6 +49,7 @@ local OFFSET_MACRO_COL = 44
|
||||
--- @field text string -- the full source text
|
||||
--- @field dir string -- the directory containing the source
|
||||
--- @field basename string -- filename without extension
|
||||
--- @field scan table -- pre-scanned SourceScan payload (from duffle.scan_source)
|
||||
|
||||
--- @class PassCtx
|
||||
--- @field sources SourceFile[] -- all source files in the build
|
||||
@@ -75,17 +61,13 @@ local OFFSET_MACRO_COL = 44
|
||||
--- @field upstream table<string, table> -- per-pass upstream outputs
|
||||
--- @field flags table -- CLI flags
|
||||
--- @field dry_run boolean -- if true, compute but don't write
|
||||
--- @field verbose boolean -- if true, log diagnostic info
|
||||
--- @field verbose boolean -- log diagnostic info
|
||||
|
||||
--- @class PassResult
|
||||
--- @field outputs table[] -- {kind=, path=} entries describing emit files
|
||||
--- @field errors table[] -- {line=, msg=} entries; build-stops
|
||||
--- @field warnings table[] -- {line=, msg=} entries; build-succeeds
|
||||
|
||||
--- @class Atom
|
||||
--- @field name string -- atom name (e.g. "cube_g4_face")
|
||||
--- @field body string -- the brace-delimited body (without the braces)
|
||||
|
||||
--- @class BranchOffset
|
||||
--- @field tag string -- the marker tag (e.g. "F" in `atom_offset(F, T)`)
|
||||
--- @field target string -- the target label name (e.g. "T" in `atom_offset(F, T)`)
|
||||
@@ -98,44 +80,7 @@ local OFFSET_MACRO_COL = 44
|
||||
--- @field offsets BranchOffset[] -- per-branch offset list
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Local helpers
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Returns true if `s` starts with `prefix`.
|
||||
-- @param s string
|
||||
-- @param prefix string
|
||||
-- @return boolean
|
||||
local function starts_with(s, prefix)
|
||||
if #s < #prefix then return false end
|
||||
for pos = 1, #prefix do
|
||||
if s:sub(pos, pos) ~= prefix:sub(pos, pos) then return false end
|
||||
end
|
||||
return true
|
||||
end
|
||||
|
||||
-- Replace every non-alphanumeric char in `s` with underscore.
|
||||
-- @param s string
|
||||
-- @return string
|
||||
local function to_alnum_underscore(s)
|
||||
local out = ""
|
||||
for pos = 1, #s do
|
||||
local ch = s:sub(pos, pos)
|
||||
if duffle.is_alnum(ch) then out = out .. ch else out = out .. "_" end
|
||||
end
|
||||
return out
|
||||
end
|
||||
|
||||
-- Right-pad `s` with spaces to width `w`. If `s` is already `w` or
|
||||
-- wider, no padding is added.
|
||||
-- @param s string
|
||||
-- @param w integer
|
||||
-- @return string
|
||||
local function pad_right(s, w)
|
||||
return s .. string.rep(" ", math.max(0, w - #s))
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Marker-call helpers
|
||||
-- Per-token marker-call helpers (atom_label / atom_offset inside bodies)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Extract comma-separated identifier args from a parenthesized group after a function-like macro call.
|
||||
@@ -219,164 +164,20 @@ local function scan_for_atom_markers(token, at_pos, labels, branches)
|
||||
end
|
||||
end
|
||||
|
||||
--- Find the end position (just past the closing ')') of the first atom_label/atom_offset call in `tok`. Returns 0 if no such call.
|
||||
--- @param tok string
|
||||
--- @return integer -- 0 if no marker call found; otherwise end-1 (just past ')')
|
||||
local function find_marker_call_end(tok)
|
||||
local pos = 1
|
||||
local tok_len = #tok
|
||||
while pos <= tok_len do
|
||||
pos = duffle.skip_ws_and_cmt(tok, pos)
|
||||
if pos > tok_len then break end
|
||||
local ch = tok:sub(pos, pos)
|
||||
if duffle.is_space(ch) then
|
||||
pos = pos + 1
|
||||
elseif ch == "/" then
|
||||
-- comment — skip past it (delegated to duffle.skip_str_or_cmt)
|
||||
local nx = duffle.skip_str_or_cmt(tok, pos)
|
||||
pos = (nx > pos) and nx or (pos + 1)
|
||||
else
|
||||
local ident, after_ident = duffle.read_ident(tok, pos)
|
||||
-- scan: <ident>
|
||||
if ident == LABEL_MARKER or ident == OFFSET_MARKER then
|
||||
-- scan: atom_label(<name>) OR atom_offset(<tag>, <target>)
|
||||
local open_paren = duffle.skip_ws_and_cmt(tok, after_ident)
|
||||
if tok:sub(open_paren, open_paren) == "(" then
|
||||
local _, end_paren = duffle.read_parens(tok, open_paren)
|
||||
return end_paren - 1
|
||||
end
|
||||
return 0
|
||||
end
|
||||
pos = after_ident or (pos + 1)
|
||||
end
|
||||
end
|
||||
return 0
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Atom scanner
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Skip C qualifier keywords (`static`, `const`, etc.) and return the position past the last qualifier.
|
||||
--- @param source string
|
||||
--- @param pos integer
|
||||
--- @return integer
|
||||
local function skip_qualifiers(source, pos)
|
||||
while true do
|
||||
pos = duffle.skip_ws_and_cmt(source, pos)
|
||||
local ident, after = duffle.read_ident(source, pos)
|
||||
if not ident then return pos end
|
||||
if QUALIFIER_KEYWORDS[ident] then pos = after else return pos end
|
||||
end
|
||||
end
|
||||
|
||||
-- (internal) Try to parse the wrapped atom form: `MipsAtom_(<name>) { ... }`.
|
||||
-- Returns the parsed Atom (name + body + position past body), or nil if the form didn't match.
|
||||
-- @param source_text string
|
||||
-- @param after_pos integer -- position just past `MipsAtom_`
|
||||
-- @return Atom|nil
|
||||
local function try_wrapped_atom(source_text, after_pos)
|
||||
local paren_pos = duffle.skip_ws_and_cmt(source_text, after_pos)
|
||||
if source_text:sub(paren_pos, paren_pos) ~= "(" then return nil end
|
||||
local inner, after_paren = duffle.read_parens(source_text, paren_pos)
|
||||
-- scan: MipsAtom_(<name>)
|
||||
|
||||
local name_start = 1
|
||||
while name_start <= #inner and duffle.is_space(inner:sub(name_start, name_start)) do
|
||||
name_start = name_start + 1
|
||||
end
|
||||
local name_end = name_start
|
||||
while name_end <= #inner and duffle.is_alnum(inner:sub(name_end, name_end)) do
|
||||
name_end = name_end + 1
|
||||
end
|
||||
local name = inner:sub(name_start, name_end - 1)
|
||||
if name == "" then return nil end
|
||||
|
||||
local brace_pos = duffle.scan_to_char(source_text, "{", after_paren)
|
||||
-- scan: MipsAtom_(<name>) {
|
||||
if not brace_pos then return nil end
|
||||
local body, after_brace = duffle.read_braces(source_text, brace_pos)
|
||||
-- scan: MipsAtom_(<name>) { <body> }
|
||||
return { name = name, body = body, after_brace = after_brace }
|
||||
end
|
||||
|
||||
-- (internal) Try to parse the raw atom form: `MipsCode code_<name> { ... }`.
|
||||
-- @param source_text string
|
||||
-- @param after_pos integer -- position just past `MipsCode`
|
||||
-- @return Atom|nil
|
||||
local function try_raw_atom(source_text, after_pos)
|
||||
local next_pos = duffle.skip_ws_and_cmt(source_text, after_pos)
|
||||
local next_ident, next_after = duffle.read_ident(source_text, next_pos)
|
||||
-- scan: MipsCode <next_ident>
|
||||
if not next_ident then return nil end
|
||||
if not starts_with(next_ident, CODE_RAW_PREFIX) then return nil end
|
||||
if #next_ident <= CODE_RAW_PREFIX_LEN then return nil end
|
||||
local atom_name = next_ident:sub(CODE_RAW_PREFIX_LEN + 1)
|
||||
-- scan: MipsCode code_<name>
|
||||
local brace_pos = duffle.scan_to_char(source_text, "{", next_after)
|
||||
-- scan: MipsCode code_<name> {
|
||||
if not brace_pos then return nil end
|
||||
local body, after_brace = duffle.read_braces(source_text, brace_pos)
|
||||
-- scan: MipsCode code_<name> { <body> }
|
||||
return { name = atom_name, body = body, after_brace = after_brace }
|
||||
end
|
||||
|
||||
--- Find every `MipsAtom_(name) { ... }` (or raw `MipsCode code_<name> { ... }`) declaration in a source.
|
||||
--- @param source_text string
|
||||
--- @return Atom[]
|
||||
local function find_atoms(source_text)
|
||||
local atoms = {}
|
||||
local pos = 1
|
||||
local src_len = #source_text
|
||||
|
||||
while pos <= src_len do
|
||||
pos = duffle.skip_ws_and_cmt(source_text, pos); if pos > src_len then break end
|
||||
pos = skip_qualifiers(source_text, pos); if pos > src_len then break end
|
||||
|
||||
local ident, after = duffle.read_ident(source_text, pos)
|
||||
-- scan: <ident>
|
||||
if not ident then
|
||||
pos = pos + 1
|
||||
elseif ident == ATOM_PREFIX then
|
||||
-- scan: MipsAtom_(<name>) { <body> }
|
||||
local atom = try_wrapped_atom(source_text, after)
|
||||
if atom then
|
||||
atoms[#atoms + 1] = { name = atom.name, body = atom.body }
|
||||
pos = atom.after_brace
|
||||
else
|
||||
pos = pos + 1
|
||||
end
|
||||
elseif ident == CODE_DECL then
|
||||
-- scan: MipsCode code_<name> { <body> }
|
||||
local atom = try_raw_atom(source_text, after)
|
||||
if atom then
|
||||
atoms[#atoms + 1] = { name = atom.name, body = atom.body }
|
||||
pos = atom.after_brace
|
||||
else
|
||||
pos = after
|
||||
end
|
||||
else
|
||||
pos = after
|
||||
end
|
||||
end
|
||||
return atoms
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Per-atom body scan
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- (internal) Count words emitted by the rest of `tok` after a marker call
|
||||
-- (the marker call itself emits 0 words, but the source pattern may bundle the marker with the next instruction on the same line,
|
||||
-- separated by no top-level comma).
|
||||
-- (internal) Count words emitted by the rest of `tok` after a marker call
|
||||
-- (the marker call itself emits 0 words, but the source pattern may bundle the marker with the next instruction on the same line,
|
||||
-- separated by no top-level comma).
|
||||
-- Returns the word count contributed by that rest.
|
||||
-- @param tok string
|
||||
-- @param word_counts table
|
||||
-- @return integer
|
||||
local function count_marker_rest(tok, word_counts)
|
||||
local marker_end = find_marker_call_end(tok)
|
||||
if marker_end <= 0 or marker_end >= #tok then return 0 end
|
||||
local rest = duffle.trim(tok:sub(marker_end + 1))
|
||||
-- duffle.find_marker_call_end returns the position PAST the closing `)` of the marker call
|
||||
-- (or nil if `tok` isn't a marker call). Canonical impl in duffle.lua is faster than the
|
||||
-- file-local copy that used to live here (byte-indexed, no `tok:sub` per char).
|
||||
local marker_end = duffle.find_marker_call_end(tok)
|
||||
if not marker_end or marker_end >= #tok then return 0 end
|
||||
local rest = duffle.trim(tok:sub(marker_end))
|
||||
if rest == "" then return 0 end
|
||||
return count_token_words(rest, word_counts)
|
||||
end
|
||||
@@ -394,11 +195,17 @@ end
|
||||
--- @param body string
|
||||
--- @param word_counts table
|
||||
--- @return table<string, integer>, table[], integer
|
||||
local function scan_atom_body(body, word_counts)
|
||||
-- scan_atom_body: walk pre-tokenized body for atom_label/atom_offset markers + word counts.
|
||||
-- Uses `atom.body_tokens` from the SourceScan payload (pre-tokenized by scan-source pass).
|
||||
-- @param body_tokens table[] -- {{tok=string, rel=integer}, ...} from duffle.tokenize_body
|
||||
-- @param word_counts table
|
||||
-- @return table, table, integer -- labels, branches, total_words
|
||||
local function scan_atom_body(body_tokens, word_counts)
|
||||
local pos = 0
|
||||
local labels = {}
|
||||
local branches = {}
|
||||
for _, tok in ipairs(duffle.split_top_level_commas(body)) do
|
||||
for _, t in ipairs(body_tokens) do
|
||||
local tok = t.tok
|
||||
if is_marker_token(tok) then
|
||||
-- Marker call: record at the current pos, do NOT advance pos.
|
||||
scan_for_atom_markers(tok, pos, labels, branches)
|
||||
@@ -433,8 +240,15 @@ local function compute_offsets(labels, branches)
|
||||
return results
|
||||
end
|
||||
|
||||
-- (internal) Build a constant-table entry `{macro_name, enum_name, value}`
|
||||
-- from a BranchOffset.
|
||||
-- Right-pad `s` with spaces to width `w`. If `s` is already `w` or wider, no padding is added.
|
||||
-- @param s string
|
||||
-- @param w integer
|
||||
-- @return string
|
||||
local function pad_right(s, w)
|
||||
return s .. string.rep(" ", math.max(0, w - #s))
|
||||
end
|
||||
|
||||
-- (internal) Build a constant-table entry `{macro_name, enum_name, value}` from a BranchOffset.
|
||||
-- @param r BranchOffset
|
||||
-- @return table
|
||||
local function make_offset_const(r)
|
||||
@@ -499,18 +313,35 @@ end
|
||||
|
||||
local M = {}
|
||||
|
||||
-- (internal) Process one source: find atoms, scan bodies, write header.
|
||||
-- Project the pre-scanned SourceScan entries into the {name, body, body_tokens} shape this pass needs.
|
||||
-- MipsAtom_ entries have kind="atom"; MipsCode code_<name> entries have kind="raw_atom".
|
||||
-- `body_tokens` is set by scan-source on every `scan.atoms[i]` / `scan.raw_atoms[i]`; we carry it forward
|
||||
-- so `scan_atom_body` reads from the precomputed table directly (no per-atom tokenize_body fallback).
|
||||
-- @param scan table -- SourceScan from duffle.scan_source
|
||||
-- @return table[] -- list of {name=, body=, body_tokens=}
|
||||
local function project_atoms(scan)
|
||||
local out = {}
|
||||
for _, a in ipairs(scan.atoms) do
|
||||
out[#out + 1] = { name = a.raw_name, body = a.body, body_tokens = a.body_tokens }
|
||||
end
|
||||
for _, a in ipairs(scan.raw_atoms) do
|
||||
out[#out + 1] = { name = a.name, body = a.body, body_tokens = a.body_tokens }
|
||||
end
|
||||
return out
|
||||
end
|
||||
|
||||
-- (internal) Process one source: project atoms from scan, scan bodies, write header.
|
||||
-- Returns the offsets_h path if a header was written, or nil.
|
||||
-- @param ctx PassCtx
|
||||
-- @param src SourceFile
|
||||
-- @return string|nil -- the offsets_h path
|
||||
local function process_source(ctx, src)
|
||||
local atoms = find_atoms(src.text)
|
||||
local atoms = project_atoms(src.scan)
|
||||
if #atoms == 0 then return nil end
|
||||
|
||||
local atoms_data = {}
|
||||
for _, atom in ipairs(atoms) do
|
||||
local labels, branches, total = scan_atom_body(atom.body, ctx.shared.word_counts)
|
||||
local labels, branches, total = scan_atom_body(atom.body_tokens, ctx.shared.word_counts)
|
||||
atoms_data[#atoms_data + 1] = {
|
||||
name = atom.name,
|
||||
total_words = total,
|
||||
@@ -526,9 +357,9 @@ local function process_source(ctx, src)
|
||||
return out_path
|
||||
end
|
||||
|
||||
--- Run the offsets pass.
|
||||
--- For each source, emits a per-module `<dir_basename>.offsets.h` containing `#define _atom_offset_F_T = N` constants for every `atom_offset(F, T)` reference
|
||||
--- in the source's atoms.
|
||||
--- Run the offsets pass.
|
||||
--- For each source, emits a per-module `<dir_basename>.offsets.h` containing `#define _atom_offset_F_T = N` constants
|
||||
--- for every `atom_offset(F, T)` reference in the source's atoms.
|
||||
--- @param ctx PassCtx
|
||||
--- @return PassResult
|
||||
function M.run(ctx)
|
||||
|
||||
+21
-35
@@ -2,19 +2,14 @@
|
||||
--- project-wide summary writer.
|
||||
---
|
||||
--- Two output files per build:
|
||||
--- - `build/gen/<dir_basename>.annotations.txt` — one per source-directory
|
||||
--- containing atoms; aggregates across all sources in the directory.
|
||||
--- - `build/gen/<dir_basename>.annotations.txt` — one per source-directory containing atoms; aggregates across all sources in the directory.
|
||||
--- - `build/gen/annotation_validation.txt` — the project summary.
|
||||
---
|
||||
--- The annotation pass stashes per-MODULE summary entries in
|
||||
--- `ctx.flags._annot_results` (set by `passes/annotation.lua`). This
|
||||
--- pass re-validates each source via `annotation.validate()` to get
|
||||
--- the detailed per-source results needed for the report. The cost is
|
||||
--- acceptable: `validate()` is fast (~5ms per source) and runs once.
|
||||
--- The annotation pass stashes per-MODULE summary entries in `ctx.flags._annot_results` (set by `passes/annotation.lua`).
|
||||
--- This pass re-validates each source via `annotation.validate()` to get the detailed per-source results needed for the report.
|
||||
---
|
||||
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
|
||||
--- Lua 5.3 compatible. See
|
||||
--- `C:\projects\Pikuma\ps1-ai\conductor\code_styleguides\lua.md`.
|
||||
--- Lua 5.3 compatible.
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Module-scope requires + package.path setup
|
||||
@@ -26,10 +21,10 @@
|
||||
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
||||
-- Uses `debug.getinfo` to find this file's own directory, so it works
|
||||
-- both standalone and when require'd from the orchestrator.
|
||||
local _src = debug.getinfo(1, "S").source:sub(2)
|
||||
local _dir = _src:match("(.*[/\\])") or "./"
|
||||
dofile(_dir .. "../duffle_paths.lua")
|
||||
local duffle = require("duffle")
|
||||
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
|
||||
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Constants
|
||||
@@ -298,7 +293,7 @@ local function render_module_report(dir, sources, results)
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Per-project summary (ported from tape_atom_annotation_pass.lua:1488-1528)
|
||||
-- Per-project summary
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Render the per-project summary (`build/gen/annotation_validation.txt`).
|
||||
@@ -351,32 +346,23 @@ end
|
||||
-- Orchestration helpers
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Group source files by their `dir` field. Used to mirror the per-DIRECTORY partitioning the annotation pass uses.
|
||||
-- @param sources SourceFile[]
|
||||
-- @return table<string, SourceFile[]> -- map of dir -> sources in that dir
|
||||
local function group_sources_by_dir(sources)
|
||||
local by_dir = {}
|
||||
for _, src in ipairs(sources) do
|
||||
by_dir[src.dir] = by_dir[src.dir] or {}
|
||||
table.insert(by_dir[src.dir], src)
|
||||
end
|
||||
return by_dir
|
||||
end
|
||||
|
||||
-- (internal) Validate each source in `dir_sources` via the annotation pass, tagging each result with `result.source = src.path` for downstream rendering.
|
||||
-- (internal) Pull per-source validate() results from the annotation pass's stash.
|
||||
-- The annotation pass runs first in the dep chain and caches results in `ctx.flags._annot_source_results`; we read from there instead of re-validating each source.
|
||||
-- Returns the list of module results + the flat list of all results (for the project-wide summary).
|
||||
-- @param ctx PassCtx
|
||||
-- @param dir_sources SourceFile[]
|
||||
-- @return AnnotationResult[], AnnotationResult[]
|
||||
local function validate_module_sources(ctx, dir_sources)
|
||||
local annotation = require("passes.annotation")
|
||||
local function lookup_module_results(ctx, dir_sources)
|
||||
local src_cache = (ctx.flags and ctx.flags._annot_source_results) or {}
|
||||
local module_results = {}
|
||||
local all_results = {}
|
||||
for _, src in ipairs(dir_sources) do
|
||||
local result = annotation.validate(ctx, src)
|
||||
result.source = src.path
|
||||
module_results[#module_results + 1] = result
|
||||
all_results[#all_results + 1] = result
|
||||
local result = src_cache[src.path]
|
||||
if result then
|
||||
result.source = src.path -- defensive (annotation tags it too; this guards against cache misses from earlier iterations)
|
||||
module_results[#module_results + 1] = result
|
||||
all_results[#all_results + 1] = result
|
||||
end
|
||||
end
|
||||
return module_results, all_results
|
||||
end
|
||||
@@ -418,7 +404,7 @@ function M.run(ctx)
|
||||
local warnings = {}
|
||||
|
||||
local module_entries = (ctx.flags and ctx.flags._annot_results) or {}
|
||||
local by_dir = group_sources_by_dir(ctx.sources)
|
||||
local by_dir = ctx.by_dir or duffle.group_sources_by_dir(ctx.sources)
|
||||
|
||||
if not ctx.dry_run then duffle.ensure_dir(ctx.out_root) end
|
||||
|
||||
@@ -428,7 +414,7 @@ function M.run(ctx)
|
||||
|
||||
if entry.atoms_count > 0 or #(by_dir[entry.dir] or {}) > 0 then
|
||||
local dir_sources = by_dir[entry.dir] or {}
|
||||
local module_results, all_results = validate_module_sources(ctx, dir_sources)
|
||||
local module_results, all_results = lookup_module_results(ctx, dir_sources)
|
||||
for _, r in ipairs(all_results) do
|
||||
all_results_for_summary[#all_results_for_summary + 1] = r
|
||||
end
|
||||
|
||||
@@ -0,0 +1,484 @@
|
||||
--- passes/scan_source.lua — Source pre-scan pass (the "mega entity" pass).
|
||||
---
|
||||
--- Single source-walk pass that produces the fat `SourceScan` payload consumed by all downstream passes. Walks each `ctx.sources` entry once,
|
||||
--- extracting every construct type the metaprograms need:
|
||||
---
|
||||
--- MipsAtom_ (kind = "atom", with optional atom_info inner)
|
||||
--- MipsAtomComp_ (kind = "comp_bare")
|
||||
--- MipsAtomComp_Proc_ (kind = "comp_proc", body inside last {})
|
||||
--- MipsCode code_<name> (kind = "raw_atom", offsets pass only)
|
||||
--- typedef Struct_(Binds_X) { fields }
|
||||
--- #pragma mac_X tape_atom words=N + _Pragma("...")
|
||||
---
|
||||
--- The result is attached to each `src.scan` so downstream passes can read from `src.scan.atoms` / `src.scan.binds` / etc. without re-walking the source.
|
||||
--- This is the first pass in the dep graph (no deps).
|
||||
--- Every other pass that reads source structure depends on this one — see `ps1_meta.lua :: PASSES`.
|
||||
---
|
||||
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
|
||||
--- Lua 5.3 compatible.
|
||||
|
||||
-- Bootstrap: same as entry scripts. See `ps1_meta.lua` for the rationale.
|
||||
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
||||
-- Uses `debug.getinfo` to find this file's own directory, so it works
|
||||
-- both standalone and when require'd from the orchestrator.
|
||||
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
|
||||
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Type declarations
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- @class SourceScan
|
||||
--- @field atoms AtomEntry[] -- MipsAtom_ + MipsAtomComp_ + MipsAtomComp_Proc_
|
||||
--- @field raw_atoms AtomEntry[] -- MipsCode code_<name> { body } (offsets pass only)
|
||||
--- @field binds BindsEntry[] -- typedef Struct_(Binds_X) { fields } (fields pre-parsed)
|
||||
--- @field atom_infos AtomInfoEntry[] -- MipsAtom_(name) atom_info(...) (sub-calls pre-parsed)
|
||||
--- @field macros MacroEntry[] -- #pragma mac_X tape_atom words=N + _Pragma("...")
|
||||
--- @field line_of fun(pos: integer): integer -- shared LineIndex closure
|
||||
|
||||
--- @class SourceFile
|
||||
--- @field path string -- absolute path to the source file
|
||||
--- @field text string -- the full source text
|
||||
--- @field dir string -- the directory containing the source
|
||||
--- @field basename string -- filename without extension
|
||||
--- @field scan table -- pre-scanned SourceScan payload (set by this pass)
|
||||
|
||||
--- @class PassCtx
|
||||
--- @field sources SourceFile[]
|
||||
--- @field metadata_path string
|
||||
--- @field shared table
|
||||
--- @field out_root string
|
||||
--- @field project_root string
|
||||
--- @field upstream table<string, table>
|
||||
--- @field flags table
|
||||
--- @field dry_run boolean
|
||||
--- @field verbose boolean
|
||||
|
||||
--- @class PassResult
|
||||
--- @field outputs table[]
|
||||
--- @field errors table[]
|
||||
--- @field warnings table[]
|
||||
|
||||
--- @class AtomEntry
|
||||
--- @field line integer
|
||||
--- @field name string -- atom name (for components: without ac_ prefix)
|
||||
--- @field body string -- brace-delimited body (without the braces)
|
||||
--- @field body_off integer -- char offset of body[1] in source
|
||||
--- @field kind string -- "atom" | "comp_bare" | "comp_proc" | "raw_atom"
|
||||
--- @field raw_name string -- un-stripped name (for components: with ac_ prefix)
|
||||
--- @field ident_pos integer -- position of the MipsAtom_/MipsAtomComp_ ident start
|
||||
--- @field after_paren integer -- position past the closing paren
|
||||
--- @field args string|nil -- populated by components pass (backward lookup)
|
||||
--- @field comment string|nil -- populated by components pass (backward lookup)
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Local helpers
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- C qualifier keywords that may precede a MipsAtom_ / MipsCode declaration.
|
||||
-- (typedef is NOT a qualifier here — it's a separate construct (`typedef Struct_(Binds_X) { ... };`)
|
||||
-- and must be read as an ident so the typedef check below can match it.)
|
||||
local QUALIFIER_KEYWORDS = {
|
||||
["static"] = true, ["const"] = true, ["volatile"] = true, ["extern"] = true,
|
||||
["register"] = true, ["auto"] = true, ["inline"] = true,
|
||||
["internal"] = true, ["LP_"] = true, ["global"] = true, ["gkknown"] = true,
|
||||
}
|
||||
|
||||
-- Parse the U4 fields from a Binds_X body. Returns (fields, byte_count).
|
||||
local function scan_binds_fields(body)
|
||||
local fields = {}
|
||||
local byte_off = 0
|
||||
local body_pos = 1
|
||||
while body_pos <= #body do
|
||||
body_pos = duffle.skip_ws_and_cmt(body, body_pos)
|
||||
if body_pos > #body then break end
|
||||
local type_ident, type_end = duffle.read_ident(body, body_pos)
|
||||
if not type_ident then
|
||||
body_pos = body_pos + 1
|
||||
elseif type_ident == "U4" then
|
||||
local field_ident, field_end = duffle.read_ident(body, duffle.skip_ws_and_cmt(body, type_end))
|
||||
if field_ident then
|
||||
fields[#fields + 1] = { name = field_ident, offset = byte_off }
|
||||
byte_off = byte_off + 4
|
||||
end
|
||||
body_pos = field_end or (type_end + 1)
|
||||
else
|
||||
body_pos = type_end + 1
|
||||
end
|
||||
end
|
||||
return fields, byte_off
|
||||
end
|
||||
|
||||
-- Parse the register list from inside `atom_reads(...)` or `atom_writes(...)`.
|
||||
local function scan_reg_list(sub_inner)
|
||||
local regs = {}
|
||||
local sub_inner_pos = 1
|
||||
while sub_inner_pos <= #sub_inner do
|
||||
sub_inner_pos = duffle.skip_ws_and_cmt(sub_inner, sub_inner_pos)
|
||||
if sub_inner_pos > #sub_inner then break end
|
||||
local reg_ident, reg_end = duffle.read_ident(sub_inner, sub_inner_pos)
|
||||
if reg_ident then
|
||||
regs[#regs + 1] = duffle.trim(reg_ident)
|
||||
sub_inner_pos = reg_end
|
||||
else
|
||||
sub_inner_pos = sub_inner_pos + 1
|
||||
end
|
||||
if sub_inner_pos > #sub_inner then break end
|
||||
if sub_inner:sub(sub_inner_pos, sub_inner_pos) == "," then sub_inner_pos = sub_inner_pos + 1 end
|
||||
end
|
||||
return regs
|
||||
end
|
||||
|
||||
-- Parse the sub-calls inside `atom_info(atom_bind(...), atom_reads(...), atom_writes(...))`.
|
||||
-- Returns (binds, reads, writes).
|
||||
local function scan_atom_info_subcalls(info_inner)
|
||||
local binds, reads, writes = nil, nil, nil
|
||||
local sub_pos = 1
|
||||
while sub_pos <= #info_inner do
|
||||
sub_pos = duffle.skip_ws_and_cmt(info_inner, sub_pos)
|
||||
if sub_pos > #info_inner then break end
|
||||
local sub_ident, sub_end = duffle.read_ident(info_inner, sub_pos)
|
||||
if not sub_ident then
|
||||
sub_pos = sub_pos + 1
|
||||
elseif sub_ident == "atom_bind" then
|
||||
local sub_open = duffle.skip_ws_and_cmt(info_inner, sub_end)
|
||||
if info_inner:sub(sub_open, sub_open) == "(" then
|
||||
local sub_inner, sub_after2 = duffle.read_parens(info_inner, sub_open)
|
||||
-- scan: atom_bind(<Binds_X>)
|
||||
binds = duffle.trim(sub_inner)
|
||||
sub_pos = sub_after2
|
||||
else
|
||||
sub_pos = sub_open + 1
|
||||
end
|
||||
elseif sub_ident == "atom_reads" or sub_ident == "atom_writes" then
|
||||
local kind = sub_ident
|
||||
local sub_open = duffle.skip_ws_and_cmt(info_inner, sub_end)
|
||||
if info_inner:sub(sub_open, sub_open) == "(" then
|
||||
local sub_inner, sub_after2 = duffle.read_parens(info_inner, sub_open)
|
||||
-- scan: atom_reads(<regs>) OR atom_writes(<regs>)
|
||||
local regs = scan_reg_list(sub_inner)
|
||||
if kind == "atom_reads" then reads = regs else writes = regs end
|
||||
sub_pos = sub_after2
|
||||
else
|
||||
sub_pos = sub_open + 1
|
||||
end
|
||||
else
|
||||
sub_pos = sub_end
|
||||
end
|
||||
end
|
||||
return binds, reads, writes
|
||||
end
|
||||
|
||||
-- Skip C qualifier keywords and return the position past the last one.
|
||||
local function scan_skip_qualifiers(source, pos)
|
||||
while true do
|
||||
pos = duffle.skip_ws_and_cmt(source, pos)
|
||||
local ident, after = duffle.read_ident(source, pos)
|
||||
if not ident then return pos end
|
||||
if QUALIFIER_KEYWORDS[ident] then pos = after else return pos end
|
||||
end
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- The single source walker
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Single-pass source scan. Walks the source ONCE and extracts every construct type the metaprogram passes need.
|
||||
--- Returns a fat SourceScan table. Each pass filters from this payload instead of re-walking the source.
|
||||
--- @param source string
|
||||
--- @return table -- SourceScan { atoms, raw_atoms, binds, atom_infos, macros, line_of }
|
||||
local function scan_source(source)
|
||||
local line_of = duffle.LineIndex(source)
|
||||
local atoms = {}
|
||||
local raw_atoms = {}
|
||||
local binds = {}
|
||||
local atom_infos = {}
|
||||
local macros = {}
|
||||
local pos = 1
|
||||
local src_len = #source
|
||||
|
||||
while pos <= src_len do
|
||||
pos = duffle.skip_ws_and_cmt(source, pos)
|
||||
if pos > src_len then break end
|
||||
|
||||
-- Skip preprocessor directives (#define / #include / #pragma / etc).
|
||||
-- _Pragma is an operator (not a directive) — it doesn't start with #.
|
||||
local pp_pos = duffle.skip_preprocessor_line(source, pos)
|
||||
if pp_pos then pos = pp_pos; goto continue end
|
||||
|
||||
-- Skip C qualifiers (static, const, etc.) that may precede a declaration.
|
||||
pos = scan_skip_qualifiers(source, pos)
|
||||
if pos > src_len then break end
|
||||
|
||||
local ident, ident_end = duffle.read_ident(source, pos)
|
||||
-- scan: <ident>
|
||||
if not ident then pos = pos + 1; goto continue end
|
||||
|
||||
-- ── MipsAtom_ / MipsAtomComp_ / MipsAtomComp_Proc_ ──
|
||||
if ident == "MipsAtom_" or ident == "MipsAtomComp_" or ident == "MipsAtomComp_Proc_" then
|
||||
local is_atom = ident == "MipsAtom_"
|
||||
local is_comp = ident == "MipsAtomComp_"
|
||||
local is_proc = ident == "MipsAtomComp_Proc_"
|
||||
local kind = is_atom and "atom" or (is_comp and "comp_bare" or "comp_proc")
|
||||
local open_paren = duffle.skip_ws_and_cmt(source, ident_end)
|
||||
if source:sub(open_paren, open_paren) ~= "(" then pos = open_paren + 1; goto continue end
|
||||
|
||||
local inner, after_paren = duffle.read_parens(source, open_paren)
|
||||
-- scan: <ident>(<args>)
|
||||
|
||||
if is_proc then
|
||||
-- MipsAtomComp_Proc_(name, { body }) — body is inside the LAST { } in args.
|
||||
local last_brace_pos
|
||||
for search_pos = #inner, 1, -1 do
|
||||
if inner:sub(search_pos, search_pos) == "{" then last_brace_pos = search_pos; break end
|
||||
end
|
||||
if last_brace_pos then
|
||||
local depth = 1
|
||||
local inner_pos = last_brace_pos + 1
|
||||
while inner_pos <= #inner and depth > 0 do
|
||||
local c = inner:byte(inner_pos)
|
||||
if c == 123 then depth = depth + 1; inner_pos = inner_pos + 1
|
||||
elseif c == 125 then depth = depth - 1; if depth == 0 then break end; inner_pos = inner_pos + 1
|
||||
elseif c == 40 then local _, a = duffle.read_parens(inner, inner_pos); inner_pos = a
|
||||
elseif c == 91 then local _, a = duffle.read_brackets(inner, inner_pos); inner_pos = a
|
||||
elseif c == 34 or c == 39 then inner_pos = duffle.skip_str_or_cmt(inner, inner_pos) + 1
|
||||
else inner_pos = inner_pos + 1 end
|
||||
end
|
||||
if depth == 0 then
|
||||
-- scan: <ident>(<name>, { <body> })
|
||||
local name_match = inner:match("^%s*([%w_]+)")
|
||||
local raw_name = name_match or "?"
|
||||
-- Strip "ac_" prefix for component names (components pass convention).
|
||||
local name = raw_name
|
||||
if #raw_name > 3 and raw_name:sub(1, 3) == "ac_" then
|
||||
name = raw_name:sub(4)
|
||||
end
|
||||
local body = inner:sub(last_brace_pos + 1, inner_pos - 1)
|
||||
local body_off = open_paren + 1 + last_brace_pos
|
||||
atoms[#atoms + 1] = {
|
||||
line = line_of(pos), name = name, body = body, body_off = body_off + 1,
|
||||
kind = kind, raw_name = raw_name,
|
||||
ident_pos = pos, after_paren = after_paren,
|
||||
args = nil, comment = nil,
|
||||
}
|
||||
end
|
||||
end
|
||||
pos = after_paren
|
||||
else
|
||||
-- MipsAtom_(name) { body } OR MipsAtomComp_(name) { body }
|
||||
local name_start = 1
|
||||
while name_start <= #inner and inner:sub(name_start, name_start):match("[%s]") do name_start = name_start + 1 end
|
||||
local name_end = name_start
|
||||
while name_end <= #inner and inner:sub(name_end, name_end):match("[%w_]") do name_end = name_end + 1 end
|
||||
local raw_name = inner:sub(name_start, name_end - 1)
|
||||
-- scan: <ident>(<name>)
|
||||
if raw_name ~= "" then
|
||||
local brace = duffle.scan_to_char(source, "{", after_paren)
|
||||
-- scan: <ident>(<name>) {
|
||||
if brace then
|
||||
local body, after_brace = duffle.read_braces(source, brace)
|
||||
-- scan: <ident>(<name>) { <body> }
|
||||
-- Strip "ac_" prefix for component names (components pass convention).
|
||||
local disp_name = raw_name
|
||||
if is_comp and #raw_name > 3 and raw_name:sub(1, 3) == "ac_" then
|
||||
disp_name = raw_name:sub(4)
|
||||
end
|
||||
atoms[#atoms + 1] = {
|
||||
line = line_of(pos), name = disp_name, body = body, body_off = brace + 1,
|
||||
kind = kind, raw_name = raw_name,
|
||||
ident_pos = pos, after_paren = after_paren,
|
||||
args = nil, comment = nil,
|
||||
}
|
||||
pos = after_brace
|
||||
else
|
||||
pos = open_paren + 1
|
||||
end
|
||||
else
|
||||
pos = open_paren + 1
|
||||
end
|
||||
end
|
||||
|
||||
-- For MipsAtom_ entries: check if atom_info(...) follows.
|
||||
if is_atom then
|
||||
local lookahead = duffle.skip_ws_and_cmt(source, after_paren)
|
||||
local look_ident, look_end = duffle.read_ident(source, lookahead)
|
||||
-- scan: MipsAtom_(<name>) <look_ident>
|
||||
if look_ident == "atom_info" then
|
||||
local info_open = duffle.skip_ws_and_cmt(source, look_end)
|
||||
if source:sub(info_open, info_open) == "(" then
|
||||
local info_inner, info_after = duffle.read_parens(source, info_open)
|
||||
-- scan: MipsAtom_(<name>) atom_info(<binds>, <reads>, <writes>)
|
||||
-- Find the atom name from the just-parsed atom entry (last one added).
|
||||
local last_atom = atoms[#atoms]
|
||||
local atom_name = last_atom and last_atom.raw_name or "?"
|
||||
local ai_binds, ai_reads, ai_writes = scan_atom_info_subcalls(info_inner)
|
||||
atom_infos[#atom_infos + 1] = {
|
||||
atom_name = atom_name, binds = ai_binds,
|
||||
reads = ai_reads or {}, writes = ai_writes or {},
|
||||
info_line = line_of(lookahead),
|
||||
}
|
||||
-- Don't advance pos past info_after — the body { ... } still needs to be skipped
|
||||
-- by the brace scan below. But if there's no body (forward decl), advance.
|
||||
local body_brace = duffle.scan_to_char(source, "{", info_after)
|
||||
if body_brace then
|
||||
local _, after_body = duffle.read_braces(source, body_brace)
|
||||
pos = after_body
|
||||
else
|
||||
pos = info_after
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
goto continue
|
||||
end
|
||||
|
||||
-- ── MipsCode code_<name> { body } (raw atom form — offsets pass only) ──
|
||||
if ident == "MipsCode" then
|
||||
local next_pos = duffle.skip_ws_and_cmt(source, ident_end)
|
||||
local next_ident, next_after = duffle.read_ident(source, next_pos)
|
||||
-- scan: MipsCode <next_ident>
|
||||
if next_ident and #next_ident > 5 and next_ident:sub(1, 5) == "code_" then
|
||||
local atom_name = next_ident:sub(6)
|
||||
-- scan: MipsCode code_<name>
|
||||
local brace_pos = duffle.scan_to_char(source, "{", next_after)
|
||||
-- scan: MipsCode code_<name> {
|
||||
if brace_pos then
|
||||
local body, after_brace = duffle.read_braces(source, brace_pos)
|
||||
-- scan: MipsCode code_<name> { <body> }
|
||||
raw_atoms[#raw_atoms + 1] = {
|
||||
line = line_of(pos), name = atom_name, body = body, body_off = brace_pos + 1,
|
||||
kind = "raw_atom", raw_name = atom_name,
|
||||
}
|
||||
pos = after_brace
|
||||
goto continue
|
||||
end
|
||||
end
|
||||
pos = ident_end
|
||||
goto continue
|
||||
end
|
||||
|
||||
-- ── typedef Struct_(Binds_X) { fields } ──
|
||||
if ident == "typedef" then
|
||||
local after_typedef = duffle.skip_ws_and_cmt(source, ident_end)
|
||||
local id2, id2_end = duffle.read_ident(source, after_typedef)
|
||||
-- scan: typedef <id2>
|
||||
if id2 == "Struct_" then
|
||||
local open_paren = duffle.skip_ws_and_cmt(source, id2_end)
|
||||
if source:sub(open_paren, open_paren) == "(" then
|
||||
local inner, after_paren = duffle.read_parens(source, open_paren)
|
||||
-- scan: typedef Struct_(<name>)
|
||||
local name = duffle.trim(inner)
|
||||
local brace = duffle.scan_to_char(source, "{", after_paren)
|
||||
-- scan: typedef Struct_(<name>) {
|
||||
if brace then
|
||||
local body, after_brace = duffle.read_braces(source, brace)
|
||||
-- scan: typedef Struct_(<name>) { <fields> }
|
||||
if name:sub(1, 6) == "Binds_" then
|
||||
local fields, byte_off = scan_binds_fields(body)
|
||||
binds[#binds + 1] = { line = line_of(pos), name = name, fields = fields, bytes = byte_off }
|
||||
end
|
||||
pos = after_brace
|
||||
goto continue
|
||||
end
|
||||
pos = open_paren + 1
|
||||
goto continue
|
||||
end
|
||||
pos = id2_end or (after_typedef + 1)
|
||||
goto continue
|
||||
end
|
||||
pos = ident_end
|
||||
goto continue
|
||||
end
|
||||
|
||||
-- ── _Pragma("mac_X tape_atom words=N") (operator form) ──
|
||||
if ident == "_Pragma" then
|
||||
local open_paren = duffle.skip_ws_and_cmt(source, ident_end)
|
||||
if source:sub(open_paren, open_paren) == "(" then
|
||||
local str, str_end = duffle.read_parens(source, open_paren)
|
||||
-- scan: _Pragma(<string>)
|
||||
str = duffle.trim(str)
|
||||
if str:sub(1, 1) == '"' and str:sub(-1) == '"' then
|
||||
local inner = str:sub(2, -2)
|
||||
local space = duffle.find_byte(inner, 32, 1)
|
||||
if space then
|
||||
local name = inner:sub(1, space - 1)
|
||||
local rest = inner:sub(space + 1)
|
||||
local eq = duffle.find_byte(rest, 61, 1)
|
||||
if eq then
|
||||
local key = duffle.trim(rest:sub(1, eq - 1))
|
||||
local val = duffle.trim(rest:sub(eq + 1))
|
||||
if key == "tape_atom words" or key == "words" then
|
||||
macros[#macros + 1] = { line = line_of(pos), name = name, words = tonumber(val) or 0 }
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
pos = str_end
|
||||
goto continue
|
||||
end
|
||||
pos = open_paren + 1
|
||||
goto continue
|
||||
end
|
||||
|
||||
-- ── #pragma mac_X tape_atom words=N (directive form) ──
|
||||
-- (preprocessor skip above handles # lines, but pragma is an ident here
|
||||
-- only if it appeared without a leading # — which happens when the
|
||||
-- preprocessor skip didn't fire because the # was on a previous line.
|
||||
-- The annotation pass handles this via its own skip_preprocessor_line,
|
||||
-- but scan_source handles it here by checking the ident.)
|
||||
if ident == "pragma" then
|
||||
-- This shouldn't normally fire — #pragma lines are skipped by
|
||||
-- skip_preprocessor_line above. If we get here, it's a _Pragma
|
||||
-- variant or a non-#-prefixed pragma. Just advance.
|
||||
pos = ident_end
|
||||
goto continue
|
||||
end
|
||||
|
||||
-- ── Unrecognized ident — advance past it ──
|
||||
pos = ident_end
|
||||
|
||||
::continue::
|
||||
end
|
||||
|
||||
return {
|
||||
atoms = atoms,
|
||||
raw_atoms = raw_atoms,
|
||||
binds = binds,
|
||||
atom_infos = atom_infos,
|
||||
macros = macros,
|
||||
line_of = line_of,
|
||||
}
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- M — module exports
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- @class M
|
||||
|
||||
local M = {}
|
||||
|
||||
--- Walk each source once and attach the fat SourceScan payload to `src.scan`.
|
||||
--- No output files; this is a pure in-memory pre-processing pass.
|
||||
--- @param ctx PassCtx
|
||||
--- @return PassResult
|
||||
function M.run(ctx)
|
||||
for _, src in ipairs(ctx.sources) do
|
||||
src.scan = scan_source(src.text)
|
||||
-- Pre-tokenize each atom body once (plex: single source of truth).
|
||||
-- Downstream passes (offsets, word-counts, components, static-analysis) read from
|
||||
-- `atom.body_tokens` instead of calling `split_top_level_commas` / `tokenize_body` independently.
|
||||
-- The tokens are memoized in duffle.lua's cache, so re-access is O(1).
|
||||
for _, atom in ipairs(src.scan.atoms) do
|
||||
atom.body_tokens = duffle.tokenize_body(atom.body)
|
||||
end
|
||||
for _, atom in ipairs(src.scan.raw_atoms or {}) do
|
||||
atom.body_tokens = duffle.tokenize_body(atom.body)
|
||||
end
|
||||
end
|
||||
return { outputs = {}, errors = {}, warnings = {} }
|
||||
end
|
||||
|
||||
return M
|
||||
+524
-1122
File diff suppressed because it is too large
Load Diff
@@ -4,7 +4,6 @@
|
||||
--- 1. **Public utilities** (used by `passes/components.lua`, `passes/offsets.lua`, `passes/annotation.lua`):
|
||||
--- - `M.count_token_words(token, wc)` — words emitted by one token
|
||||
--- - `M.scan_dir(dir, suffix)` — glob walk for *.macs.h
|
||||
--- - `M.count_body_words(body, wc)` — words emitted by an atom body
|
||||
--- 2. **Pass entry** `M.run(ctx)` — loads metadata.h + *.macs.h into `ctx.shared.word_counts` for downstream passes.
|
||||
--- 3. **Internal helpers** for the body scanner.
|
||||
---
|
||||
@@ -20,22 +19,26 @@
|
||||
-- Bootstrap: see `ps1_meta.lua` for the rationale.
|
||||
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
||||
-- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator.
|
||||
local _src = debug.getinfo(1, "S").source:sub(2)
|
||||
local _dir = _src:match("(.*[/\\])") or "./"
|
||||
dofile(_dir .. "../duffle_paths.lua")
|
||||
local duffle = require("duffle")
|
||||
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
|
||||
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Constants
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Windows separator chars — used to convert `dir /b /s` output (which uses `\`) into POSIX paths (which our scripts expect).
|
||||
-- Windows separator chars — used to convert `dir` output (which uses `\`) into POSIX paths (which our scripts expect).
|
||||
local PATH_SEP_BACKSLASH = "\\"
|
||||
local PATH_SEP_FORWARD = "/"
|
||||
|
||||
-- Glob command for Windows directory walk. `dir /b /s` lists all matching files recursively with bare paths (no headers);
|
||||
-- `2>nul` discards the "file not found" stderr when nothing matches.
|
||||
local DIR_GLOB_CMD = 'dir /b /s "%s\\%s" 2>nul'
|
||||
-- Fallback glob command (subprocess). Used when `lfs` (LuaFileSystem) is not available.
|
||||
-- Scoped to `code\` to avoid walking `.git/`, `toolchain/`, `build/`, etc.
|
||||
local DIR_GLOB_CMD = 'dir /b /s "%s\\code\\%s" 2>nul'
|
||||
|
||||
-- Try to load lfs (LuaFileSystem). If available, scan_dir uses native directory enumeration (~2ms)
|
||||
-- instead of spawning `dir /b /s` as a subprocess (~56ms). Built by update_deps.ps1 into toolchain/lfs/lfs.dll.
|
||||
local lfs = pcall(require, "lfs") and require("lfs") or nil
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Type declarations
|
||||
@@ -120,36 +123,50 @@ end
|
||||
-- If a build removes/creates .macs.h files mid-process, the caller can invalidate by calling `M._invalidate_scan_cache()`.
|
||||
local SCAN_CACHE_KEY = "__word_count_eval_scan_cache__"
|
||||
|
||||
--- Recursively scan a directory for files matching a glob suffix.
|
||||
--- No regex per the no_regex constraint — uses plain byte matching via `dir /b /s` on Windows.
|
||||
--- Scan `code/` for files matching `suffix` (e.g. `*.macs.h`).
|
||||
--- Uses `lfs` (LuaFileSystem) when available — native directory enumeration at ~2ms.
|
||||
--- Falls back to `dir /b /s` subprocess (~56ms) when `lfs` is not compiled.
|
||||
---
|
||||
--- @param dir string -- directory to scan (absolute or relative)
|
||||
--- @param dir string -- project root directory
|
||||
--- @param suffix string -- file pattern, e.g. "*.macs.h"
|
||||
--- @return string[]
|
||||
function M.scan_dir(dir, suffix)
|
||||
local key = dir .. "\0" .. suffix
|
||||
|
||||
-- Check the in-process cache first. (Mostly helps when a build triggers multiple `M.run` calls -- e.g.
|
||||
-- the audit_lua_nesting script's stress tests but the cost is ~free either way.)
|
||||
local cache = package.loaded[SCAN_CACHE_KEY]
|
||||
if cache and cache[key] then return cache[key] end
|
||||
|
||||
local results = {}
|
||||
local pipe = io.popen(DIR_GLOB_CMD:format(dir, suffix))
|
||||
if not pipe then
|
||||
-- Cache the empty result too (avoids re-scan if the dir is genuinely empty -- e.g. a clean build before components has run yet).
|
||||
cache = cache or {}
|
||||
cache[key] = results
|
||||
package.loaded[SCAN_CACHE_KEY] = cache
|
||||
return results
|
||||
end
|
||||
for raw_line in pipe:lines() do
|
||||
local path = raw_line:gsub(PATH_SEP_BACKSLASH, PATH_SEP_FORWARD)
|
||||
results[#results + 1] = path
|
||||
end
|
||||
pipe:close()
|
||||
|
||||
-- Cache the result.
|
||||
if lfs then
|
||||
-- Native walk: list code/<module>/gen/ for matching files. Zero subprocess spawns.
|
||||
local code_dir = dir .. "/code"
|
||||
if lfs.attributes(code_dir, "mode") == "directory" then
|
||||
for mod_name in lfs.dir(code_dir) do
|
||||
if mod_name ~= "." and mod_name ~= ".." then
|
||||
local gen_path = code_dir .. "/" .. mod_name .. "/gen"
|
||||
if lfs.attributes(gen_path, "mode") == "directory" then
|
||||
for fname in lfs.dir(gen_path) do
|
||||
if fname:match("%.macs%.h$") then
|
||||
results[#results + 1] = gen_path .. "/" .. fname
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
else
|
||||
-- Fallback: single `dir /b /s` subprocess scoped to code\.
|
||||
local pipe = io.popen(DIR_GLOB_CMD:format(dir, suffix))
|
||||
if pipe then
|
||||
for raw_line in pipe:lines() do
|
||||
results[#results + 1] = raw_line:gsub(PATH_SEP_BACKSLASH, PATH_SEP_FORWARD)
|
||||
end
|
||||
pipe:close()
|
||||
end
|
||||
end
|
||||
|
||||
-- Cache the result (including empty results).
|
||||
cache = cache or {}
|
||||
cache[key] = results
|
||||
package.loaded[SCAN_CACHE_KEY] = cache
|
||||
@@ -160,88 +177,6 @@ end
|
||||
--- Invalidate the scan cache (call after creating new .macs.h files in the same Lua process — usually not needed).
|
||||
function M._invalidate_scan_cache() package.loaded[SCAN_CACHE_KEY] = nil end
|
||||
|
||||
-- ┌────────────────────────────────────────────────────────────────────┐
|
||||
-- │ Shared utility: count_body_words │
|
||||
-- └────────────────────────────────────────────────────────────────────┘
|
||||
|
||||
--- Count words emitted by an entire atom body (a brace-delimited block).
|
||||
--- Splits by top-level commas; for each token, delegates to count_token_words.
|
||||
--- Handles `atom_label(name)` / `atom_offset(tag, name)` markers
|
||||
--- (record at current pos, do NOT advance pos; if the marker call bundles an instruction after it, count that instruction too).
|
||||
---
|
||||
--- @param body string -- brace-delimited atom body (without braces)
|
||||
--- @param wc WordCounts -- the shared word-count table
|
||||
--- @return integer -- total words
|
||||
function M.count_body_words(body, wc)
|
||||
local total = 0
|
||||
for _, tok in ipairs(duffle.split_top_level_commas(body)) do
|
||||
local pos = 1
|
||||
local tok_len = #tok
|
||||
while pos <= tok_len and duffle.is_space(tok:sub(pos, pos)) do
|
||||
pos = pos + 1
|
||||
end
|
||||
local leading_ident = duffle.read_ident(tok, pos)
|
||||
local is_marker = leading_ident == "atom_label" or leading_ident == "atom_offset"
|
||||
if is_marker then
|
||||
-- Marker call: record at current pos, do NOT advance pos.
|
||||
-- But the source pattern may bundle the marker with the next instruction on a new line (no top-level comma between them).
|
||||
-- In that case, the rest of `tok` after the marker call is a real instruction that must still be counted.
|
||||
local marker_end = M.find_marker_call_end(tok)
|
||||
if marker_end > 0 and marker_end < #tok then
|
||||
local rest = duffle.trim(tok:sub(marker_end + 1))
|
||||
if rest ~= "" then
|
||||
total = total + M.count_token_words(rest, wc)
|
||||
end
|
||||
end
|
||||
else
|
||||
total = total + M.count_token_words(tok, wc)
|
||||
end
|
||||
end
|
||||
return total
|
||||
end
|
||||
|
||||
--- Find the end position (just past the closing ')') of the first atom_label/atom_offset call in `tok`. Returns 0 if no such call.
|
||||
--- Internal helper for count_body_words.
|
||||
---
|
||||
--- @param tok string
|
||||
--- @return integer -- 0 if no marker call found
|
||||
function M.find_marker_call_end(tok)
|
||||
local pos = 1
|
||||
local tok_len = #tok
|
||||
while pos <= tok_len do
|
||||
pos = duffle.skip_ws_and_cmt(tok, pos)
|
||||
if pos > tok_len then break end
|
||||
local ch = tok:sub(pos, pos)
|
||||
if duffle.is_space(ch) then
|
||||
pos = pos + 1
|
||||
elseif ch == "/" then
|
||||
-- comment — skip past it (delegated to duffle.skip_str_or_cmt)
|
||||
local nx = duffle.skip_str_or_cmt(tok, pos)
|
||||
pos = (nx > pos) and nx or (pos + 1)
|
||||
else
|
||||
local ident, after_ident = duffle.read_ident(tok, pos)
|
||||
local marker_end = find_marker_end(tok, ident, after_ident)
|
||||
if marker_end > 0 then return marker_end end
|
||||
pos = after_ident or (pos + 1)
|
||||
end
|
||||
end
|
||||
return 0
|
||||
end
|
||||
|
||||
-- (internal) If `ident` is `atom_label`/`atom_offset` followed by `(...)`, return the position just past the closing ')'.
|
||||
-- Otherwise 0.
|
||||
-- @param tok string
|
||||
-- @param ident string|nil
|
||||
-- @param after_ident integer
|
||||
-- @return integer
|
||||
local function find_marker_end(tok, ident, after_ident)
|
||||
if ident ~= "atom_label" and ident ~= "atom_offset" then return 0 end
|
||||
local open_paren = duffle.skip_ws_and_cmt(tok, after_ident)
|
||||
if tok:sub(open_paren, open_paren) ~= "(" then return 0 end
|
||||
local _, end_paren = duffle.read_parens(tok, open_paren)
|
||||
return end_paren - 1
|
||||
end
|
||||
|
||||
-- ┌────────────────────────────────────────────────────────────────────┐
|
||||
-- │ Pass entry: M.run(ctx) — "word-counts" pass │
|
||||
-- └────────────────────────────────────────────────────────────────────┘
|
||||
|
||||
+79
-52
@@ -6,24 +6,23 @@
|
||||
---
|
||||
--- **Architecture**:
|
||||
--- - **PASSES table** — declarative dep graph (data, not code).
|
||||
--- - **FLAG_HANDLERS table** — per-flag CLI dispatchers (handler-map
|
||||
--- pattern; replaces an 8-way if/elseif chain).
|
||||
--- - **parse_args** → **build_ctx** → **topo_sort** → **dispatch_passes**.
|
||||
--- - **FLAG_HANDLERS table** — per-flag CLI dispatchers (handler-map pattern; replaces an 8-way if/elseif chain).
|
||||
--- - **parse_args** → **build_ctx** (just opens + reads source files; no inline scanning) → **topo_sort** → **dispatch_passes**.
|
||||
--- - The first pass in the dep graph is `scan-source` (see `passes/scan_source.lua`).
|
||||
--- It calls `duffle.scan_source` once per source to produce the fat `SourceScan` payload, which is attached to each `src.scan`.
|
||||
--- Every other pass that reads source structure depends on `scan-source` and consumes `src.scan` as a read-only payload.
|
||||
---
|
||||
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
|
||||
--- Lua 5.3 compatible.
|
||||
---
|
||||
--- Lua 5.3 compatible.
|
||||
---
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Module-scope requires + package.path setup
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Resolve `arg[0]` to an absolute-ish script directory so that `require("duffle")` resolves against `scripts/` regardless of CWD.
|
||||
-- Note: this boilerplate is duplicated in 6 other entry scripts; a
|
||||
-- Phase-6 extraction target (`duffle.setup_package_path()`).
|
||||
-- Bootstrap: load `duffle_paths.lua` (uses `git rev-parse` to find the repo root, then sets package.path + package.cpath).
|
||||
-- After this line, `require("duffle")` and `require("passes.X")` both resolve.
|
||||
dofile((arg[0]:match("(.*[/\\])") or "./") .. "duffle_paths.lua")
|
||||
local duffle = require("duffle")
|
||||
-- Bootstrap: load `duffle_paths.lua` via `arg[0]` (this script's own path).
|
||||
-- That single statement: (a) sets `package.path` + `package.cpath` (via cached `git rev-parse`),
|
||||
-- (b) at the bottom returns `require("duffle")`. So the dofile's return value is the duffle module.
|
||||
local duffle = dofile((arg[0]:match("(.*[/\\])") or "./") .. "duffle_paths.lua")
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Constants
|
||||
@@ -104,6 +103,13 @@ local PASS_FLAG_DISPATCH_KEY = "__pass__"
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
local PASSES = {
|
||||
["scan-source"] = {
|
||||
module = "passes.scan_source",
|
||||
kind = "shared",
|
||||
deps = {},
|
||||
desc = "Walk each source once; produce the fat SourceScan payload for downstream passes",
|
||||
out = {},
|
||||
},
|
||||
["word-counts"] = {
|
||||
module = "passes.word_count_eval",
|
||||
kind = "shared",
|
||||
@@ -114,14 +120,14 @@ local PASSES = {
|
||||
components = {
|
||||
module = "passes.components",
|
||||
kind = "header-output",
|
||||
deps = {"word-counts"},
|
||||
deps = {"scan-source", "word-counts"},
|
||||
desc = "Emit mac_X macros from MipsAtomComp_ declarations",
|
||||
out = { { kind = "header", path_template = "<source_dir>/gen/<basename>.macs.h" } },
|
||||
},
|
||||
annotation = {
|
||||
module = "passes.annotation",
|
||||
kind = "validation",
|
||||
deps = {"word-counts"},
|
||||
deps = {"scan-source", "word-counts"},
|
||||
desc = "Validate atom DSL usage; emit errors.h + annotations.txt",
|
||||
out = {
|
||||
{ kind = "report", path_template = "<out_root>/<basename>.errors.h" },
|
||||
@@ -131,14 +137,14 @@ local PASSES = {
|
||||
offsets = {
|
||||
module = "passes.offsets",
|
||||
kind = "header-output",
|
||||
deps = {"word-counts", "components"},
|
||||
deps = {"scan-source", "word-counts", "components"},
|
||||
desc = "Compute branch offsets for atom_label / atom_offset",
|
||||
out = { { kind = "header", path_template = "<source_dir>/gen/<basename>.offsets.h" } },
|
||||
},
|
||||
["static-analysis"] = {
|
||||
module = "passes.static_analysis",
|
||||
kind = "validation",
|
||||
deps = {"word-counts", "components"},
|
||||
deps = {"scan-source", "word-counts", "components"},
|
||||
desc = "[FUTURE] GTE pipeline-fill, mac_yield uniformity, etc.",
|
||||
out = { { kind = "report", path_template = "<out_root>/<basename>.static_analysis.txt" } },
|
||||
},
|
||||
@@ -167,11 +173,12 @@ local PASS_FLAG_TO_NAME = {
|
||||
["--offsets"] = "offsets",
|
||||
["--static-analysis"] = "static-analysis",
|
||||
["--report"] = "report",
|
||||
["--scan-source"] = "scan-source",
|
||||
["--all"] = ALL_PASSES_SENTINEL,
|
||||
}
|
||||
|
||||
local ALL_PASS_NAMES = {
|
||||
"word-counts", "components", "annotation",
|
||||
"scan-source", "word-counts", "components", "annotation",
|
||||
"offsets", "static-analysis", "report",
|
||||
}
|
||||
|
||||
@@ -337,6 +344,9 @@ local function build_ctx(args)
|
||||
dir = dir:sub(1, -2)
|
||||
end
|
||||
|
||||
-- src.scan is populated by the "scan-source" pass (the first pass in the
|
||||
-- dep graph). build_ctx just opens + reads the files; the scan itself
|
||||
-- happens in the pass module, not inline in the orchestrator.
|
||||
sources[#sources + 1] = {
|
||||
path = path,
|
||||
text = text,
|
||||
@@ -345,8 +355,14 @@ local function build_ctx(args)
|
||||
}
|
||||
end
|
||||
|
||||
-- Pre-compute the per-directory grouping once (Fleury: expose structure).
|
||||
-- Three passes (annotation, report, static-analysis) call group_sources_by_dir with the same ctx.sources;
|
||||
-- computing it here and stashing on ctx.by_dir eliminates 2 redundant calls.
|
||||
local by_dir = duffle.group_sources_by_dir(sources)
|
||||
|
||||
return {
|
||||
sources = sources,
|
||||
by_dir = by_dir,
|
||||
metadata_path = args.metadata,
|
||||
shared = {},
|
||||
upstream = {},
|
||||
@@ -509,41 +525,52 @@ local function render_dep_graph(passes, requested, closed)
|
||||
end
|
||||
add("")
|
||||
|
||||
add("[ps1_meta] Pass graph (read top-to-bottom):")
|
||||
-- Data-driven ASCII graph built from the actual PASSES table.
|
||||
-- Shows the source -> scan_source -> pass chain. Each pass is
|
||||
-- shown once; edges are "feeds into" arrows based on deps.
|
||||
add("[ps1_meta] Pass graph (read top-to-bottom; edges = 'feeds into'):")
|
||||
add("")
|
||||
add(" metadata.h")
|
||||
add(" |")
|
||||
add(" v")
|
||||
add(" +-----------+ +-----------------+ +-----------------+")
|
||||
add(" | word- |-->| components |-->| offsets |")
|
||||
add(" | counts | +-----------------+ +-----------------+")
|
||||
add(" | (load) | | ^")
|
||||
add(" +-----------+ | |")
|
||||
add(" | v |")
|
||||
add(" | code/<module>/gen/<basename>.macs.h |")
|
||||
add(" | (header - co-located for #include) |")
|
||||
add(" | |")
|
||||
add(" | +-----------------+ |")
|
||||
add(" +---------->| annotation |--------------+")
|
||||
add(" | +-----------------+ |")
|
||||
add(" | | |")
|
||||
add(" | v |")
|
||||
add(" | build/gen/<basename>.errors.h |")
|
||||
add(" | build/gen/<basename>.annotations.txt |")
|
||||
add(" | (report - NOT #included) |")
|
||||
add(" | |")
|
||||
add(" | +-----------------+ |")
|
||||
add(" +---------->| static-analysis |--------------+")
|
||||
add(" +-----------------+")
|
||||
add(" |")
|
||||
add(" v")
|
||||
add(" +---------------+")
|
||||
add(" | report |")
|
||||
add(" +---------------+")
|
||||
add(" |")
|
||||
add(" v")
|
||||
add(" build/gen/annotation_validation.txt")
|
||||
add(" (project summary)")
|
||||
|
||||
-- Compute which passes feed which other passes (reverse of deps).
|
||||
local feeds = {} -- feeds[X] = list of passes that X feeds into
|
||||
for _, name in ipairs(closed) do feeds[name] = {} end
|
||||
for name, p in pairs(passes) do
|
||||
for _, dep in ipairs(p.deps) do
|
||||
if feeds[dep] then feeds[dep][#feeds[dep] + 1] = name end
|
||||
end
|
||||
end
|
||||
|
||||
-- Layout: source -> scan_source -> word-counts -> {components, annotation, offsets, static-analysis} -> report
|
||||
-- Outputs are listed under each pass.
|
||||
local outputs_for = function(name)
|
||||
local p = passes[name]
|
||||
if not p or not p.out or #p.out == 0 then return "" end
|
||||
local outs = {}
|
||||
for _, o in ipairs(p.out) do outs[#outs + 1] = o.path_template end
|
||||
return table.concat(outs, ", ")
|
||||
end
|
||||
|
||||
add(" +-----------+ +-------------------+ +-----------------+")
|
||||
add(" | source |-->| scan_source |--->| word-counts |")
|
||||
add(" | files | | (scan_source.lua) | | (load) |")
|
||||
add(" +-----------+ +-------------------+ +-----------------+")
|
||||
add(" (single walk) |")
|
||||
add(" |")
|
||||
add(" +-------------------+-------------------+-----------+")
|
||||
add(" v v v v")
|
||||
add(" +--------------+ +--------------+ +--------------+ +---------------+")
|
||||
add(" | components | | annotation | | offsets | |static-analysis|")
|
||||
add(" +--------------+ +--------------+ +--------------+ +---------------+")
|
||||
add(" |<src>/gen/ | |build/gen/ | |<src>/gen/ | |build/gen/ |")
|
||||
add(" |<base>.macs.h | |<base>.errors | |<base>.offsets| |<base>.static |")
|
||||
add(" | (header) | | .h | | .h | | _analysis |")
|
||||
add(" +------+-------+ | +annot.txt | | (header) | | .txt |")
|
||||
add(" | +------+-------+ +--------------+ +------+--------+")
|
||||
add(" v v v")
|
||||
add(" +------+----------------+ +------+-------+ |")
|
||||
add(" |offsets|static-analysis| |report| |<--------------------+")
|
||||
add(" | | | +------+-------+")
|
||||
add(" +-------+---------------+")
|
||||
|
||||
return table.concat(lines, "\n") .. "\n"
|
||||
end
|
||||
@@ -573,7 +600,7 @@ end
|
||||
-- @param result PassResult
|
||||
-- @return boolean
|
||||
local function report_validation_errors(pass_name, pass, result)
|
||||
local has_errors = result.errors and #result.errors > 0
|
||||
local has_errors = result.errors and #result.errors > 0
|
||||
if not (has_errors and PASS_KIND_STOP_ON_ERROR[pass.kind]) then
|
||||
return false
|
||||
end
|
||||
|
||||
+15
-13
@@ -10,19 +10,6 @@ $ErrorActionPreference = 'Stop'
|
||||
$misc = join-path $PSScriptRoot 'helpers/misc.ps1'
|
||||
. $misc
|
||||
|
||||
# TODO(Ed): Review usage of these deps
|
||||
# I originally cloned them when starting to get to the C runtime usage of the course
|
||||
# However, based on the heavy reliance of the PSX.Dev extension I might fallback; also
|
||||
# The gdb server doesn't need the full repo and were only using the src/mips
|
||||
# which has a standalone repo (nuggets)
|
||||
# armips may not be used at all but I'm not sure...
|
||||
#
|
||||
# PCSX-Redux: built via MSBuild (VS2022) — automated in the build section below.
|
||||
# Requires: VS2022 with C++ desktop workload + PlatformToolset=v143 retarget.
|
||||
# The .vcxproj files request v145; we pass /p:PlatformToolset=v143 to MSBuild.
|
||||
# NuGet packages are restored automatically on first build.
|
||||
# Output: toolchain\pcsx-redux\vsprojects\x64\Debug\pcsx-redux.exe
|
||||
|
||||
$url_armips = 'https://github.com/Kingcom/armips.git'
|
||||
$url_pcsx_redux = 'https://github.com/grumpycoders/pcsx-redux.git'
|
||||
$url_psyq_iwyu = 'https://github.com/johnbaumann/psyq_include_what_you_use.git'
|
||||
@@ -114,6 +101,21 @@ push-location $path_lpeg
|
||||
& gcc @lpeg_compile_args
|
||||
pop-location
|
||||
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
# lfs (LuaFileSystem) — compiled from pcsx-redux's vendored luafilesystem source.
|
||||
# Used by word_count_eval.lua :: scan_dir for native directory enumeration (~2ms)
|
||||
# instead of spawning `dir /b /s` as a subprocess (~56ms).
|
||||
# Source: toolchain/pcsx-redux/third_party/luafilesystem/src/lfs.c
|
||||
# Output: toolchain/lfs/lfs.dll
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
$path_lfs = join-path $path_toolchain 'lfs'
|
||||
verify-path $path_lfs
|
||||
$lfs_src = join-path $path_pcsx_redux 'third_party\luafilesystem\src\lfs.c'
|
||||
$lfs_dll = join-path $path_lfs 'lfs.dll'
|
||||
$lfs_dll_import = join-path $luajit_lib_dir 'libluajit-5.1.dll.a'
|
||||
& gcc -O2 -shared "-I$lua_inc_dir" -o $lfs_dll $lfs_src $lfs_dll_import
|
||||
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
# OpenBIOS — built from the PCSX-Redux source tree via make + mipsel-none-elf
|
||||
#
|
||||
|
||||
Reference in New Issue
Block a user