Files
document-haness/scripts/verify-pipeline.py
T

354 lines
15 KiB
Python
Executable File

#!/usr/bin/env python3
from __future__ import annotations
import argparse
import importlib.util
import json
import sys
from pathlib import Path
REQUIRED_PATHS = (
# 스킬 — 분석에서 게시까지
".agents/skills/running-tech-log-pipeline/SKILL.md",
".agents/skills/running-tech-log-pipeline/references/stage-contracts.md",
".agents/skills/running-tech-log-pipeline/references/subagent-prompts.md",
".agents/skills/running-tech-log-pipeline/templates/run.json",
".agents/skills/publishing-tech-log-to-studio/SKILL.md",
".agents/skills/publishing-tech-log-to-studio/references/studio-form-map.md",
".agents/skills/publishing-tech-log-to-studio/references/playwright-recipes.md",
".agents/skills/analyzing-codebase-for-tech-log/SKILL.md",
".agents/skills/deriving-tech-log-root-tree/SKILL.md",
".agents/skills/deriving-tech-log-root-tree/references/decomposition-checklist.md",
".agents/skills/deriving-tech-log-root-tree/references/candidate-disposition.md",
".agents/skills/deriving-tech-log-root-tree/references/example-tech-log-tree.md",
".agents/skills/writing-tech-log-records/SKILL.md",
".agents/skills/writing-tech-log-records/references/tech-log-tree-contract.md",
".agents/skills/writing-tech-log-records/references/record-kinds.md",
".agents/skills/writing-tech-log-records/references/body-syntax.md",
".agents/skills/writing-tech-log-records/references/code-tables-diagrams.md",
".agents/skills/writing-tech-log-records/references/review-checklist.md",
".agents/skills/writing-tech-log-records/references/from-ssot-to-records.md",
".agents/skills/writing-tech-log-records/references/writing-each-kind.md",
".agents/skills/rewriting-technical-prose-naturally/SKILL.md",
".agents/skills/rewriting-technical-prose-naturally/references/protected-content.md",
".agents/skills/rewriting-technical-prose-naturally/references/editorial-rules.md",
".agents/skills/writing-tech-log-records/references/explaining.md",
".agents/skills/writing-tech-log-records/references/ai-tells.md",
".agents/skills/rewriting-technical-prose-naturally/references/research-method.md",
".agents/skills/rewriting-technical-prose-naturally/scripts/check_prose.mjs",
".agents/skills/rewriting-technical-prose-naturally/scripts/style_profile.mjs",
".agents/skills/writing-as-the-person-who-did-it/SKILL.md",
".agents/skills/writing-as-the-person-who-did-it/references/voice-moves.md",
".agents/skills/writing-as-the-person-who-did-it/scripts/check_voice.mjs",
".agents/skills/technical-visualizer/SKILL.md",
".agents/skills/refactoring-from-analysis/SKILL.md",
# 프로젝트 폴더 틀 — 끝난 프로젝트의 모양. 작업 재료는 여기 없다
"docs/_templates/README.md",
"docs/_templates/final/document.md",
"docs/_templates/final/evidence/meta/evidence.json",
"docs/_templates/tech-log-studio/tech-log-tree.json",
# 분석하는 동안에만 있는 작업 재료의 틀
".agents/skills/analyzing-codebase-for-tech-log/templates/state.json",
".agents/skills/analyzing-codebase-for-tech-log/templates/source-index.md",
".agents/skills/analyzing-codebase-for-tech-log/templates/analysis/00-project-overview.md",
".agents/skills/analyzing-codebase-for-tech-log/templates/analysis/module.md",
".agents/skills/writing-tech-log-records/templates/case.md",
".agents/skills/writing-tech-log-records/templates/concept.md",
".agents/skills/writing-tech-log-records/templates/reference.md",
".agents/skills/writing-tech-log-records/templates/question.md",
".agents/skills/writing-tech-log-records/templates/decision.md",
# 도구
"scripts/techviz",
"scripts/build-tech-log-tree.py",
"scripts/techlog.py",
"scripts/verify-tech-log-tree.py",
"scripts/verify-pipeline-run.py",
"scripts/check-figure-overlap.py",
"scripts/verify-project-layout.py",
"scripts/fold-analysis-into-final.py",
"scripts/fold-studio-contract-into-index.py",
"scripts/terminal-evidence/render_terminal.py",
"scripts/terminal-evidence/README.md",
".agents/skills/writing-tech-log-records/scripts/check_body.mjs",
".agents/skills/writing-tech-log-records/scripts/check_evidence.mjs",
)
# 글감 계약이 요구하는 칸. 틀이 이것들을 보여 주지 않으면 아무도 채우지 않는다
INDEX_TOKENS = (
"readerQuestion",
"candidateScope",
"sourceRepository",
"readinessValues",
"dispositionValues",
"KEEP_IN_SSOT",
"classification",
"missing-verification",
"basis-version",
"relations",
"candidates",
"dispositionReview",
)
QUEUE_TOKENS = ("version:", "activeProject:", "projects:")
REFACTOR_QUEUE_TOKENS = ("version:", "activeItem:", "items:")
STATE_REANALYSIS_TOKENS = (
'"reanalysis"',
'"baselineRevision"',
'"targetRevision"',
'"mode"',
'"changedPaths"',
'"impactedScopes"',
)
ALLOWED_QUEUE_STATUSES = {"PENDING", "IN_PROGRESS", "COMPLETE", "REANALYZE", "BLOCKED", "SKIPPED"}
def _parse_analysis_queue(text: str):
active = None
projects = []
current = None
for raw in text.splitlines():
stripped = raw.strip()
if not stripped or stripped.startswith("#"):
continue
if raw.startswith("activeProject:"):
value = raw.split(":", 1)[1].strip().strip("\"'")
active = None if value in {"", "null", "~"} else value
continue
if stripped.startswith("- name:"):
name = stripped.split(":", 1)[1].strip().strip("\"'")
current = {"name": name, "status": None}
projects.append(current)
continue
if current is not None and stripped.startswith("status:"):
current["status"] = stripped.split(":", 1)[1].strip().strip("\"'")
return active, projects
def _verify_refactor_queue(path: Path) -> list[str]:
errors: list[str] = []
text = path.read_text(encoding="utf-8", errors="replace")
for token in REFACTOR_QUEUE_TOKENS:
if token not in text:
errors.append(f"refactor queue missing token: {token}")
return errors
def _verify_analysis_queue(path: Path) -> list[str]:
errors = []
text = path.read_text(encoding="utf-8", errors="replace")
for token in QUEUE_TOKENS:
if token not in text:
errors.append(f"analysis queue missing token: {token}")
if errors:
return errors
active, projects = _parse_analysis_queue(text)
names = [p["name"] for p in projects]
if len(names) != len(set(names)):
errors.append("analysis queue contains duplicate project names")
for project in projects:
status = project.get("status")
if status not in ALLOWED_QUEUE_STATUSES:
errors.append(f"analysis queue invalid status for {project['name']}: {status}")
in_progress = [p["name"] for p in projects if p.get("status") == "IN_PROGRESS"]
if len(in_progress) > 1:
errors.append("analysis queue has multiple IN_PROGRESS projects")
owned = [p["name"] for p in projects if p.get("status") in {"IN_PROGRESS", "BLOCKED"}]
if len(owned) > 1:
errors.append("analysis queue has multiple active-owned projects")
if active is None:
if owned:
errors.append("activeProject does not match active-owned project")
elif owned != [active]:
errors.append("activeProject does not match active-owned project")
return errors
FORBIDDEN_LITERAL = "document-" + "haness"
TEXT_SUFFIXES = {".md", ".json", ".py", ".sh", ".txt", ".yaml", ".yml", ".toml"}
def _iter_pipeline_text_files(shared_root: Path):
for rel_root in (".agents", "docs/_templates", "scripts"):
root = shared_root / rel_root
if not root.exists():
continue
for path in root.rglob("*"):
if not path.is_file() or path.suffix.lower() not in TEXT_SUFFIXES:
continue
if "__pycache__" in path.parts:
continue
yield path
def verify_pipeline(shared_root: Path) -> list[str]:
shared_root = Path(shared_root)
errors: list[str] = []
for rel in REQUIRED_PATHS:
if not (shared_root / rel).exists():
errors.append(f"missing required path: {rel}")
queue = shared_root / "docs/analysis-queue.yaml"
if queue.exists():
errors.extend(_verify_analysis_queue(queue))
refactor_queue = shared_root / "docs/refactor-queue.yaml"
if refactor_queue.exists():
errors.extend(_verify_refactor_queue(refactor_queue))
state_template = shared_root / (
".agents/skills/analyzing-codebase-for-tech-log/templates/state.json")
if state_template.exists():
state_text = state_template.read_text(encoding="utf-8", errors="replace")
for token in STATE_REANALYSIS_TOKENS:
if token not in state_text:
errors.append(f"state template missing reanalysis token: {token}")
index = shared_root / "docs/_templates/tech-log-studio/tech-log-tree.json"
if index.exists():
text = index.read_text(encoding="utf-8", errors="replace")
for token in INDEX_TOKENS:
if token not in text:
errors.append(f"tech-log-tree template missing token: {token}")
for path in _iter_pipeline_text_files(shared_root):
text = path.read_text(encoding="utf-8", errors="replace")
if FORBIDDEN_LITERAL in text:
rel = path.relative_to(shared_root)
errors.append(f"forbidden legacy dependency in {rel}: {FORBIDDEN_LITERAL}")
if path.is_symlink():
try:
target = str(path.resolve())
except OSError:
target = ""
if FORBIDDEN_LITERAL in target:
rel = path.relative_to(shared_root)
errors.append(f"forbidden legacy symlink target in {rel}: {target}")
return errors
def _load(shared_root: Path, filename: str, name: str):
path = shared_root / "scripts" / filename
if not path.exists():
return None
sys.path.insert(0, str(shared_root / "scripts"))
spec = importlib.util.spec_from_file_location(name, path)
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module
def verify_projects(shared_root: Path) -> list:
"""실제 프로젝트의 분해 계약 정합성. 템플릿에 토큰이 있는지와는 다른 것이다."""
verifier = _load(shared_root, "verify-tech-log-tree.py", "verify_tech_log_tree")
if verifier is None:
return []
projects = sorted(p.parent.name for p in shared_root.glob("docs/*/tech-log-studio")
if not p.parent.name.startswith("_"))
return [verifier.verify(name) for name in projects]
def verify_layouts(shared_root: Path) -> list:
"""프로젝트 폴더가 같은 모양인지. 틀은 docs/_templates 다."""
verifier = _load(shared_root, "verify-project-layout.py", "verify_project_layout")
if verifier is None:
return []
projects = sorted({p.parent.parent.name
for p in shared_root.glob("docs/*/final/document.md")
if not p.parent.parent.name.startswith("_")})
return [verifier.verify(name) for name in projects]
def verify_runs(shared_root: Path):
"""`runs/<프로젝트>/<runId>/run.json` 이 절차를 지켰는지 본다.
원장이 없는 것은 정상이다 — 파이프라인을 한 줄기로 돌린 적이 없다는 뜻이다.
있는데 안 지킨 것만 잡는다.
"""
module = _load(shared_root, "verify-pipeline-run.py", "verify_pipeline_run")
ledgers = sorted((shared_root / "runs").glob("*/*/run.json"))
return [module.verify(str(p)) for p in ledgers]
def main() -> int:
parser = argparse.ArgumentParser(description="Verify the Tech Log documentation pipeline workspace.")
parser.add_argument("shared_root", nargs="?", type=Path,
default=Path(__file__).resolve().parent.parent)
parser.add_argument("--skip-projects", action="store_true",
help="틀과 스킬만 본다. 프로젝트 트리 정합성은 보지 않는다")
parser.add_argument("--samples", type=int, default=2)
args = parser.parse_args()
errors = verify_pipeline(args.shared_root)
reports = [] if args.skip_projects else verify_projects(args.shared_root)
layouts = [] if args.skip_projects else verify_layouts(args.shared_root)
runs = [] if args.skip_projects else verify_runs(args.shared_root)
project_errors = (sum(r.error_count for r in reports)
+ sum(r.error_count for r in layouts)
+ sum(r.error_count for r in runs))
if errors:
print("PIPELINE CONTRACT: FAIL")
for error in errors:
print(f"- {error}")
else:
print("PIPELINE CONTRACT: PASS")
print(f"- required paths: {len(REQUIRED_PATHS)}")
print("- analysis queue contract: valid")
print("- tech-log-tree contract: present")
print("- forbidden legacy dependency: absent")
if layouts:
layout_errors = sum(r.error_count for r in layouts)
print()
print(f"PROJECT LAYOUT: {'FAIL' if layout_errors else 'PASS'}"
f" — 프로젝트 {len(layouts)} · error {layout_errors} ·"
f" warn {sum(r.warn_count for r in layouts)}")
for report in layouts:
verifier_render(report, args.samples)
if runs:
run_errors = sum(r.error_count for r in runs)
print()
print(f"PIPELINE RUNS: {'FAIL' if run_errors else 'PASS'}"
f" — 런 {len(runs)} · error {run_errors} ·"
f" warn {sum(r.warn_count for r in runs)}")
for report in runs:
verifier_render(report, args.samples)
if reports:
tree_errors = sum(r.error_count for r in reports)
print()
print(f"TECH LOG TREES: {'FAIL' if tree_errors else 'PASS'}"
f" — 프로젝트 {len(reports)} · error {tree_errors} ·"
f" warn {sum(r.warn_count for r in reports)}")
for report in reports:
verifier_render(report, args.samples)
return 1 if errors or project_errors else 0
def verifier_render(report, samples: int) -> None:
facts = " · ".join(
f"{k}={json.dumps(v, ensure_ascii=False) if isinstance(v, dict) else v}"
for k, v in report.facts.items())
print(f" [{report.project}] {facts}")
for label, bucket, mark in (("error", report.errors, "✗"), ("warn", report.warns, "!")):
for rule, details in sorted(bucket.items(), key=lambda kv: -len(kv[1])):
print(f" {mark} {label} {len(details):>4} {rule}")
for detail in details[:samples]:
if detail:
print(f" · {detail}")
if samples and len(details) > samples:
print(f" … 외 {len(details) - samples}건")
if __name__ == "__main__":
raise SystemExit(main())