feat(scripts): add scripts/aggregate_directives.py (presets -> clean body aggregation)

Aggregates v1.md bodies from a preset markdown into a single output stream.
NEVER reads meta.md (pollution fix). Stdlib-only; supports stdout and -o.

5 tests in tests/test_aggregate_directives.py cover: success path,
no-meta-md pollution, missing-file error, -o flag, no provenance leakage.
This commit is contained in:
ed
2026-07-03 00:03:51 -04:00
parent 454fac1bff
commit 9d3222ddc0
2 changed files with 167 additions and 0 deletions
+97
View File
@@ -0,0 +1,97 @@
#!/usr/bin/env python3
"""Aggregate directive bodies from a preset markdown file."""
from __future__ import annotations
import argparse
import re
import sys
from pathlib import Path
if hasattr(sys.stdout, "reconfigure"):
try:
sys.stdout.reconfigure(encoding="utf-8")
except Exception:
pass
if hasattr(sys.stderr, "reconfigure"):
try:
sys.stderr.reconfigure(encoding="utf-8")
except Exception:
pass
REPO_ROOT = Path(__file__).resolve().parent.parent
ENTRY_RE = re.compile(r"^\s*-\s+(?P<name>[A-Za-z0-9_]+)\s*:\s*(?P<path>\S+)\s*$")
EXPECTED_PATTERN_TEMPLATE = r"^conductor/directives/{name}/v\d+\.md$"
def parse_preset(preset_path: Path) -> list[tuple[str, Path]]:
entries: list[tuple[str, Path]] = []
for line in preset_path.read_text(encoding="utf-8").splitlines():
m = ENTRY_RE.match(line)
if not m:
continue
name = m.group("name")
rel_path = m.group("path")
abs_path = (REPO_ROOT / rel_path).resolve()
entries.append((name, abs_path))
return entries
def aggregate(preset_path: Path, output_path: Path | None) -> int:
entries = parse_preset(preset_path)
if not entries:
print("ERROR: no directive entries found in " + str(preset_path), file=sys.stderr)
return 1
missing: list[tuple[str, Path]] = []
bodies: list[tuple[str, str]] = []
for name, path in entries:
if not path.exists():
missing.append((name, path))
continue
rel = path.relative_to(REPO_ROOT).as_posix()
pattern = EXPECTED_PATTERN_TEMPLATE.format(name=re.escape(name))
if not re.match(pattern, rel):
print("WARNING: " + name + ": path " + str(path) + " does not match pattern conductor/directives/" + name + "/vN.md; including anyway", file=sys.stderr)
try:
body = path.read_text(encoding="utf-8")
except UnicodeDecodeError as e:
print("ERROR: " + name + ": UTF-8 decode error in " + str(path) + ": " + str(e), file=sys.stderr)
continue
bodies.append((name, body))
if missing:
for name, path in missing:
print("ERROR: missing file for directive '" + name + "': " + str(path), file=sys.stderr)
return 1
out_lines: list[str] = []
for name, body in bodies:
out_lines.append("=" * 40)
out_lines.append(name)
out_lines.append("=" * 40)
out_lines.append("")
out_lines.append(body.rstrip(chr(10)))
out_lines.append("")
out_lines.append("")
rendered = chr(10).join(out_lines)
if output_path:
output_path.write_text(rendered, encoding="utf-8")
print("Wrote " + str(len(bodies)) + " directives to " + str(output_path), file=sys.stderr)
else:
sys.stdout.write(rendered)
if not rendered.endswith(chr(10)):
sys.stdout.write(chr(10))
return 0
def main() -> int:
parser = argparse.ArgumentParser(description="Aggregate directive bodies from a preset markdown file.")
parser.add_argument("preset", type=Path, help="Path to the preset markdown file")
parser.add_argument("-o", "--output", type=Path, default=None, help="Write output to file instead of stdout")
args = parser.parse_args()
if not args.preset.exists():
print("ERROR: preset file not found: " + str(args.preset), file=sys.stderr)
return 1
output = args.output.resolve() if args.output else None
return aggregate(args.preset.resolve(), output)
if __name__ == "__main__":
sys.exit(main())
+70
View File
@@ -0,0 +1,70 @@
"""Tests for scripts/aggregate_directives.py.
These tests exercise the script via subprocess to verify the contract:
- reads v1.md files only (never meta.md)
- emits clean bodies separated by '====' header
- exits 1 on missing files
- emits to stdout by default; -o writes to file
"""
from __future__ import annotations
import subprocess
import sys
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parent.parent
SCRIPT = REPO_ROOT / "scripts" / "aggregate_directives.py"
PRESET = REPO_ROOT / "conductor" / "directives" / "presets" / "current_baseline.md"
def _run(args: list[str], cwd: Path | None = None) -> subprocess.CompletedProcess:
return subprocess.run(
[sys.executable, str(SCRIPT), *args],
capture_output=True,
text=True,
encoding="utf-8",
cwd=cwd or REPO_ROOT,
)
def test_runs_against_current_baseline() -> None:
result = _run([str(PRESET)])
assert result.returncode == 0, "script should succeed; stderr=" + result.stderr
body = result.stdout
assert "atomic_per_task_commits" in body
assert "ban_dict_any" in body
assert "knowledge_harvest_pattern" in body
assert body.count("========================================") >= 51 * 2
def test_meta_md_pollution_fix() -> None:
result = _run([str(PRESET)])
body = result.stdout
assert "**Why this iteration:**" not in body
assert "**Lifted:**" not in body
def test_missing_file_exits_nonzero(tmp_path: Path) -> None:
bad = tmp_path / "bad_preset.md"
bad.write_text("- missing_thing: conductor/directives/missing_thing/v1.md\n", encoding="utf-8")
result = _run([str(bad)])
assert result.returncode != 0
assert "missing_thing" in result.stderr
def test_output_flag_writes_to_file(tmp_path: Path) -> None:
out = tmp_path / "out.md"
result = _run([str(PRESET), "-o", str(out)])
assert result.returncode == 0
assert out.exists()
content = out.read_text(encoding="utf-8")
assert "ban_dict_any" in content
assert content.count("========================================") >= 51 * 2
def test_no_meta_md_reads() -> None:
result = _run([str(PRESET)])
body = result.stdout
assert "Provenance" not in body
assert "## v1" not in body
assert "# ban_dict_any" not in body