Files

271 lines
14 KiB
Python

#!/usr/bin/env python3
"""#18 behavioral gaps — settings.json 실제 배선, context-package 생성 E2E,
workspace isolation, CI 아티팩트 실존. 문자열 존재가 아니라 동작을 검증한다.
standalone (no pytest). exit 0 = all pass.
"""
import json
import os
import subprocess
import sys
import tempfile
import yaml
ROOT = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
HOOKS = os.path.join(ROOT, ".claude", "hooks")
PY = sys.executable
sys.path.insert(0, HOOKS)
passed = failed = 0
def check(name, ok):
global passed, failed
if ok:
passed += 1
print(f" PASS {name}")
else:
failed += 1
print(f" FAIL {name}")
def run(script, args, env=None):
e = dict(os.environ)
e["CLAUDE_PROJECT_DIR"] = ROOT
if env:
e.update(env)
return subprocess.run([PY, os.path.join(HOOKS, script)] + args,
capture_output=True, text=True, env=e)
print("== (A) settings.json 실제 배선(5개 이벤트가 올바른 스크립트로) ==")
cfg = json.load(open(os.path.join(ROOT, ".claude", "settings.json")))
hooks = cfg.get("hooks", {})
WIRING = {
"PreToolUse": "guard_tools.py",
"PostToolUse": "evidence_ledger.py",
"SubagentStart": "subagent_register.py",
"SubagentStop": "stop_validate.py",
"Stop": "stop_validate.py",
}
for ev, script in WIRING.items():
cmds = [h.get("command", "") for g in hooks.get(ev, []) for h in g.get("hooks", [])]
check(f"settings.json wires {ev} -> {script}", any(script in c for c in cmds))
# Stop 은 --main 플래그로 메인 세션 전용(advisory)임을 확인
_stop_cmds = [h.get("command", "") for g in hooks.get("Stop", []) for h in g.get("hooks", [])]
check("Stop hook uses --main (advisory main-session)", any("--main" in c for c in _stop_cmds))
# SubagentStop 은 --main 이 아니어야(fail-closed)
_ss_cmds = [h.get("command", "") for g in hooks.get("SubagentStop", []) for h in g.get("hooks", [])]
check("SubagentStop is NOT --main (fail-closed)", all("--main" not in c for c in _ss_cmds))
print("== (B) context-package 생성 E2E (compile -> validate -> fill -> pass) ==")
import context_package as CP # noqa: E402
WS = tempfile.mkdtemp(prefix="ci_ws_")
env_ws = {"ORGOS_WORKSPACE": WS}
# P3-B cutover: generic spawn E2E 는 상류 의존 없는 DAG-source 역할을 쓴다. arch-app(consumer)은
# 활성화 후 solution-architecture required-input(both-active hard)을 요구해 spawn 이 막힌다
# — 이 테스트는 패키지 머시너리(compile→validate→fill→pass)만 검증하므로 required-inputs 0 인
# arch-bizanalyst(requirements-analysis, DAG-source)로 교체. required-inputs 강제 자체는
# test_p3b_enforcement/test_p3b_cutover 가 별도로 검증.
r = run("context_package.py",
["--compile", "--workflow", "wf-ci", "--task", "t1", "--role", "arch-bizanalyst",
"--tier", "heavy", "--mode", "converge"], env=env_ws)
pkg_rel = r.stdout.strip().splitlines()[-1] if r.stdout.strip() else ""
pkg_path = os.path.join(ROOT, pkg_rel)
check("compile emits a .pkg.yaml path", pkg_rel.endswith(".pkg.yaml") and os.path.exists(pkg_path))
pkg = yaml.safe_load(open(pkg_path)) if os.path.exists(pkg_path) else {}
# #17: heavy tier -> model/effort 가 패키지에 실려나온다
check("compiled package carries model=opus (heavy #17)", pkg.get("model") == "opus")
check("compiled package carries effort=high (heavy #17)", pkg.get("effort") == "high")
# 스켈레톤은 placeholder가 남아있어 validate 실패해야(스폰 전 강제)
v_skel = run("context_package.py", [pkg_path], env=env_ws)
check("skeleton package fails validate (placeholders unfilled -> spawn 금지)", v_skel.returncode == 1)
# placeholder를 채우면 통과
# P0-2: must-read 의 경로형 항목은 실존해야 한다(위장 방지). 실제 파일을 만든다.
with open(os.path.join(WS, "decision-packet.md"), "w") as _dp:
_dp.write("# decision packet\n")
pkg.update({
"target-repo": "some/repo", "objective": "arch-app 관점 설계 리뷰",
"allowed-tools": ["Read", "Grep", "Glob", "Write"],
"task-boundaries": "이 역할 관점만", "non-goals": ["구현"],
"must-read": ["decision-packet.md"], "acceptance-tests": ["설계 일관성 체크"],
"evidence-plan": ["설계문서 인용(E3)"],
})
filled_path = pkg_path.replace(".pkg.yaml", "-filled.pkg.yaml")
yaml.safe_dump(pkg, open(filled_path, "w"), allow_unicode=True)
v_fill = run("context_package.py", [filled_path], env=env_ws)
check("filled package passes validate (spawn 허용)", v_fill.returncode == 0)
# 회귀: model/effort 결정이 SoT(governance-tiers)에서 온다
check("light tier -> sonnet/low (SoT)", CP.model_effort_for_tier("light") == {"model": "sonnet", "effort": "low"})
# #13: required-fields 를 context-package-spec.yaml(SoT)에서 읽는다(하드코딩 아님) + P0 하한 보장
check("#13 context_package reads required-fields from spec (not hardcoded)",
CP._required_from_spec() is not None)
check("#13 P0 floor always enforced even if spec omits them",
all(f in CP.REQUIRED_FIELDS for f in CP.P0_REQUIRED))
print("== (C) workspace isolation (두 워크스페이스가 서로의 산출물을 안 읽음) ==")
import _workspace as W # noqa: E402
WSA = tempfile.mkdtemp(prefix="ci_wsA_")
WSB = tempfile.mkdtemp(prefix="ci_wsB_")
def _records_dir(ws):
e = dict(os.environ); e["ORGOS_WORKSPACE"] = ws; e["CLAUDE_PROJECT_DIR"] = ROOT
out = subprocess.run(
[PY, "-c", "import sys; sys.path.insert(0, r'%s'); import _workspace as W; print(W.records_dir())" % HOOKS],
capture_output=True, text=True, env=e)
return out.stdout.strip()
ra, rb = _records_dir(WSA), _records_dir(WSB)
check("distinct workspaces resolve to distinct records dirs", ra != rb and WSA in ra and WSB in rb)
# 미설정이면 조용한 기본값 없이 중단(finding #5)
e_unset = dict(os.environ); e_unset.pop("ORGOS_WORKSPACE", None)
# 저장소의 실제 포인터 유무에 테스트 결과가 좌우되지 않도록 포인터 없는 격리 project root를 쓴다.
NO_POINTER_ROOT = tempfile.mkdtemp(prefix="ci_no_pointer_")
e_unset["CLAUDE_PROJECT_DIR"] = NO_POINTER_ROOT
out = subprocess.run(
[PY, "-c",
"import sys; sys.path.insert(0, r'%s'); import _workspace as W;\n"
"try:\n W.workspace_name(); print('RESOLVED')\nexcept W.WorkspaceNotSetError:\n print('HALT')" % HOOKS],
capture_output=True, text=True, env=e_unset)
check("unset workspace halts (no silent test default #5)", "HALT" in out.stdout)
print("== (D) CI 아티팩트 실존 + 파싱 ==")
check("requirements.txt exists + pins PyYAML",
"PyYAML==" in open(os.path.join(ROOT, "requirements.txt")).read())
tv = yaml.safe_load(open(os.path.join(ROOT, ".claude", "tool-versions.yaml")))["tool-versions"]
check("tool-versions.yaml has required python+pyyaml",
"python" in tv["required"] and "pyyaml" in tv["required"])
check(".github/workflows/ci.yml exists + runs run_all",
"run_all.py" in open(os.path.join(ROOT, ".github", "workflows", "ci.yml")).read())
check("run_all.py exists (single test runner)",
os.path.exists(os.path.join(ROOT, ".claude", "tests", "run_all.py")))
print("== (E) #19 KPI collector: 아티팩트에서 파생 KPI 실측 + 미측정 정직 표시 ==")
KWS = tempfile.mkdtemp(prefix="ci_kpi_")
_rdir = os.path.join(KWS, "completion-records", "wf-k")
os.makedirs(_rdir, exist_ok=True)
os.makedirs(os.path.join(KWS, "state"), exist_ok=True)
os.makedirs(os.path.join(KWS, "reports"), exist_ok=True)
kenv = {"ORGOS_WORKSPACE": KWS}
def _wr_artifact(artifact_id, kind, producer, payload):
path = os.path.join(_rdir, f"{artifact_id}.report.yaml")
report = {
"report-type": "workflow-artifact", "artifact-kind": kind,
"artifact-version": 1, "tier": "light",
"identity": {"artifact-id": artifact_id, "workflow-id": "wf-k",
"stage": "intake", "producer-role-id": producer},
"payload": payload,
"report-header": {
"bottom-line": f"{kind} KPI fixture", "decision-needed": {"needed": False},
"confidence": {"value": "Med", "derived-from": "evidence"},
"risks": [], "evidence": [{"source-uri": "README.md", "grade": "E3"}],
},
}
with open(path, "w", encoding="utf-8") as fh:
yaml.safe_dump(report, fh, allow_unicode=True, sort_keys=False)
return path
_kpi_init = run("state_engine.py", ["init-workflow", "--workflow", "wf-k", "--tier", "light"], env=kenv)
check("#19 trusted workflow init", _kpi_init.returncode == 0)
brief1 = _wr_artifact("exec-ceo-1", "decision-brief", "EXEC-CEO", {"mode": "converge", "tier": "light", "candidate-families": ["FAM-CPO", "FAM-CTO", "FAM-CFO"]})
brief2 = _wr_artifact("exec-ceo-2", "decision-brief", "EXEC-CEO", {"mode": "converge", "tier": "light", "candidate-families": ["FAM-CPO", "FAM-CTO", "FAM-CFO"]})
profile = _wr_artifact("exec-ceo-profile-1", "workload-profile", "EXEC-CEO", {
"surfaces": {"ui": False, "public-api": False, "persistence": False, "infrastructure": False},
"risk": {"security-bearing": False, "data-migration": False, "external-side-effect": False,
"risk-level": "Low", "reversibility": "two-way-door", "blast-radius": "single-role",
"privacy": False, "regulatory": False, "slo-impact": False},
"required-capabilities": ["kpi-test"], "product-feature": False,
})
_kpi_ops = [
("submit first decision brief", run(
"state_engine.py", ["submit-artifact", "--workflow", "wf-k", "--report", brief1,
"--actor", "OPS-ORCH"], env=kenv)),
("review first decision brief", run(
"state_engine.py", ["review-artifact", "--workflow", "wf-k", "--report", brief1,
"--decision", "changes-requested", "--reviewer", "HUMAN-001"], env=kenv)),
("submit second decision brief", run(
"state_engine.py", ["submit-artifact", "--workflow", "wf-k", "--report", brief2,
"--actor", "OPS-ORCH"], env=kenv)),
("submit workload profile", run(
"state_engine.py", ["submit-artifact", "--workflow", "wf-k", "--report", profile,
"--actor", "OPS-ORCH"], env=kenv)),
("review workload profile", run(
"state_engine.py", ["review-artifact", "--workflow", "wf-k", "--report", profile,
"--decision", "accepted", "--reviewer", "HUMAN-001"], env=kenv)),
]
for _name, _result in _kpi_ops:
check(f"#19 trusted API: {_name}", _result.returncode == 0)
rd = run("kpi_ledger.py", ["derive", "--workflow", "wf-k"], env=kenv)
check("#19 derive computes metrics from artifacts (exit 0)", rd.returncode == 0)
check("#19 derive reports counts", "3 reports" in rd.stdout and "2 acceptance" in rd.stdout)
led = os.path.join(KWS, "state", "kpi-ledger.jsonl")
metrics = {}
if os.path.exists(led):
for line in open(led):
r = json.loads(line)
if r.get("source") == "derived" and r.get("metric") != "_counts":
metrics[r["metric"]] = r["value"]
check("#19 rework-rate derived = 0.3333 (1 changes-req / 3 submitted outputs)",
metrics.get("rework-rate") == 0.3333)
check("#19 distinct immutable revisions are not misclassified as duplicates",
metrics.get("duplicate-report-rate") == 0)
dd = run("kpi_ledger.py", ["dashboard"], env=kenv)
kmd = os.path.join(KWS, "reports", "KPI.md")
kbody = open(kmd).read() if os.path.exists(kmd) else ""
check("#19 dashboard renders KPI.md", os.path.exists(kmd))
check("#19 dashboard 정직: 미측정 KPI를 '미측정'으로 표시(위장 안 함)", "미측정" in kbody)
check("#19 dashboard shows derived value for rework-rate", "rework-rate" in kbody and "derived" in kbody)
print("== (F) 3주차 골든태스크 벤치마크: list/record/compare (격리 temp ROOT) ==")
import shutil as _sh # noqa: E402
BROOT = tempfile.mkdtemp(prefix="ci_bench_")
os.makedirs(os.path.join(BROOT, "benchmark"), exist_ok=True)
for fn in ("golden-tasks.yaml", "benchmark-rubric.yaml"):
_sh.copy(os.path.join(ROOT, "benchmark", fn), os.path.join(BROOT, "benchmark", fn))
benv = {"CLAUDE_PROJECT_DIR": BROOT}
def _brun(args):
e = dict(os.environ); e.update(benv)
return subprocess.run([PY, os.path.join(HOOKS, "benchmark.py")] + args,
capture_output=True, text=True, env=e)
_gt = yaml.safe_load(open(os.path.join(ROOT, "benchmark", "golden-tasks.yaml")))["golden-tasks"]
check("golden-tasks.yaml has >=10 tasks across categories",
len(_gt.get("tasks", [])) >= 10 and len(set(x["category"] for x in _gt["tasks"])) >= 4)
check("benchmark list runs", _brun(["list"]).returncode == 0)
_brun(["record", "--task", "GT-01", "--arm", "plain",
"--scores", "first-pass-acceptance=0,tests-pass-rate=0.7,rework-count=2"])
_brun(["record", "--task", "GT-01", "--arm", "harness",
"--scores", "first-pass-acceptance=1,tests-pass-rate=1.0,rework-count=0"])
# 미등록 task/arm은 거부(정합성)
check("record rejects unknown task", _brun(["record", "--task", "NOPE", "--arm", "plain",
"--scores", "x=1"]).returncode == 2)
_bc = _brun(["compare"])
check("benchmark compare runs", _bc.returncode == 0)
_bmd = os.path.join(BROOT, "benchmark", "BENCHMARK.md")
_bt = open(_bmd).read() if os.path.exists(_bmd) else ""
check("BENCHMARK.md shows harness win on first-pass-acceptance",
"first-pass-acceptance" in _bt and "하네스" in _bt)
check("BENCHMARK.md has weighted composite summary", "composite" in _bt)
# 표본 없는 dimension은 '미실행'으로 정직 표시
check("dims without samples marked 미실행 (정직)", "미실행" in _bt)
for d in (WS, WSA, WSB, NO_POINTER_ROOT, KWS, BROOT):
_sh.rmtree(d, ignore_errors=True)
print(f"\n{passed} passed, {failed} failed")
sys.exit(1 if failed else 0)