init: llm-wiki-haness 하네스 설계
This commit is contained in:
+161
@@ -0,0 +1,161 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import pathlib
|
||||
import re
|
||||
|
||||
ROOT = pathlib.Path(__file__).resolve().parents[5]
|
||||
TOPIC = "2026-07-18-keycloak-branch-note-consistency"
|
||||
SPEC = ROOT / "docs/superpowers/specs"
|
||||
BASE = SPEC / TOPIC
|
||||
MASTER = SPEC / f"{TOPIC}-report.md"
|
||||
CONTROLLER = SPEC / f"{TOPIC}-controller-verification.md"
|
||||
ADV_ROOT = SPEC / f"{TOPIC}-adversarial-review.md"
|
||||
|
||||
scope = sorted(
|
||||
str(path.relative_to(ROOT))
|
||||
for path in (ROOT / "raw/branch-notes").glob("*.md")
|
||||
if "keycloak" in path.name.lower() or "keycloak" in path.read_text(encoding="utf-8").lower()
|
||||
)
|
||||
matrix = (BASE / "evidence-matrix.md").read_text(encoding="utf-8")
|
||||
matrix_rows = []
|
||||
malformed = []
|
||||
for line in matrix.splitlines():
|
||||
if re.match(r"^\| \x60raw/branch-notes/[^\x60]+\.md\x60 \|", line):
|
||||
cells = [cell.strip() for cell in line.strip().strip("|").split("|")]
|
||||
if len(cells) != 4:
|
||||
malformed.append(line)
|
||||
continue
|
||||
matrix_rows.append((cells[0].strip("\x60"), cells[1], cells[2], cells[3]))
|
||||
if cells[1] not in {"READ_FULL", "READ_PARTIAL", "NOT_READ", "BLOCKED"}:
|
||||
malformed.append(line)
|
||||
|
||||
matrix_paths = [row[0] for row in matrix_rows]
|
||||
missing = sorted(set(scope) - set(matrix_paths))
|
||||
extra = sorted(set(matrix_paths) - set(scope))
|
||||
duplicates = sorted({path for path in matrix_paths if matrix_paths.count(path) > 1})
|
||||
nonexistent = sorted(path for path in matrix_paths if not (ROOT / path).is_file())
|
||||
|
||||
lane_files = sorted((BASE / "lanes").glob("lane-0[1-4]-*.md"))
|
||||
sections = {}
|
||||
schema_failures = []
|
||||
for lane in lane_files:
|
||||
text = lane.read_text(encoding="utf-8")
|
||||
matches = list(re.finditer(r"(?m)^### (L[1-4]-F\d{2})\b[^\n]*", text))
|
||||
for index, match in enumerate(matches):
|
||||
end = matches[index + 1].start() if index + 1 < len(matches) else len(text)
|
||||
sections.setdefault(match.group(1), []).append(text[match.start():end])
|
||||
|
||||
expected_ids = (
|
||||
[f"L1-F{i:02d}" for i in range(1, 15)]
|
||||
+ [f"L2-F{i:02d}" for i in range(1, 10)]
|
||||
+ [f"L3-F{i:02d}" for i in range(1, 17)]
|
||||
+ [f"L4-F{i:02d}" for i in range(1, 22)]
|
||||
)
|
||||
for finding_id, entries in sections.items():
|
||||
section = entries[0]
|
||||
checks = {
|
||||
"source_file": "raw/branch-notes/" in section,
|
||||
"severity": bool(re.search(r"Severity|심각도", section, re.I)),
|
||||
"falsification": bool(re.search(r"Falsification|Falsified|무효|반증", section, re.I)),
|
||||
"recommendation": bool(re.search(r"Required action|Action / why|synthesis recommendation|권고|조치", section, re.I)),
|
||||
}
|
||||
failed = [name for name, ok in checks.items() if not ok]
|
||||
if failed:
|
||||
schema_failures.append(f"{finding_id}:{','.join(failed)}")
|
||||
|
||||
per_file = (BASE / "per-file-findings.md").read_text(encoding="utf-8")
|
||||
per_file_sections = len(re.findall(r"(?m)^## \x60[^\x60]+\.md\x60$", per_file))
|
||||
proofs = (BASE / "sed-proofs.md").read_text(encoding="utf-8")
|
||||
proof_ids = re.findall(r"(?m)^\| \x60(L[1-4]-F\d{2})\x60 \|", proofs)
|
||||
adversarial = (BASE / "adversarial-review.md").read_text(encoding="utf-8")
|
||||
adv_ids = re.findall(r"(?m)^\| (L[1-4]-F\d{2}) \|", adversarial)
|
||||
generic = ["수동 보완책이 존재함", "일부 비핵심 경로", "치명적인 영향이 없음"]
|
||||
generic_hits = sum(adversarial.count(term) for term in generic)
|
||||
priority = (BASE / "priority-recommendations.md").read_text(encoding="utf-8")
|
||||
priority_ids = sorted(set(re.findall(r"L[1-4]-F\d{2}", priority)))
|
||||
|
||||
required = [
|
||||
MASTER,
|
||||
CONTROLLER,
|
||||
ADV_ROOT,
|
||||
BASE / "evidence-matrix.md",
|
||||
BASE / "per-file-findings.md",
|
||||
BASE / "sed-proofs.md",
|
||||
BASE / "priority-recommendations.md",
|
||||
BASE / "unresolved-risk-register.md",
|
||||
BASE / "adversarial-review.md",
|
||||
]
|
||||
missing_artifacts = [str(path.relative_to(ROOT)) for path in required if not path.is_file()]
|
||||
|
||||
link_files = [path for path in required if path.is_file()] + lane_files
|
||||
broken_links = []
|
||||
for source in link_files:
|
||||
text = source.read_text(encoding="utf-8")
|
||||
for match in re.finditer(r"\[[^\]]+\]\(([^)]+\.md)(?:#[^)]+)?\)", text):
|
||||
target = pathlib.Path(match.group(1))
|
||||
if str(target).startswith(("http:/", "https:/")):
|
||||
continue
|
||||
resolved = target if target.is_absolute() else (source.parent / target).resolve()
|
||||
if not resolved.is_file():
|
||||
broken_links.append(f"{source.relative_to(ROOT)}->{target}")
|
||||
|
||||
banned = [
|
||||
"100%", "완벽", "완전", "극한", "극단", "정밀한", "흔들림 없이",
|
||||
"절대로", "최강", "역사상 가장", "명품", "원천 차단", "보증", "폭사",
|
||||
]
|
||||
forbidden = []
|
||||
for path in [MASTER, CONTROLLER, BASE / "priority-recommendations.md", BASE / "unresolved-risk-register.md"]:
|
||||
if not path.is_file():
|
||||
continue
|
||||
text = re.sub(r"\x60\x60\x60.*?\x60\x60\x60", "", path.read_text(encoding="utf-8"), flags=re.S)
|
||||
text = re.sub(r"\x60[^\x60]*\x60", "", text)
|
||||
for line_number, line in enumerate(text.splitlines(), 1):
|
||||
for term in banned:
|
||||
if term in line:
|
||||
forbidden.append(f"{path.relative_to(ROOT)}:{line_number}:{term}")
|
||||
|
||||
claim_texts = [(ROOT / path).read_text(encoding="utf-8") for path in scope]
|
||||
values = {
|
||||
"raw_file_count": len(scope),
|
||||
"raw_line_count": sum(len(text.splitlines()) for text in claim_texts),
|
||||
"matrix_rows": len(matrix_rows),
|
||||
"read_full_rows": sum(row[1] == "READ_FULL" for row in matrix_rows),
|
||||
"blocked_rows": sum(row[1] == "BLOCKED" for row in matrix_rows),
|
||||
"missing_paths": len(missing),
|
||||
"extra_paths": len(extra),
|
||||
"duplicate_paths": len(duplicates),
|
||||
"nonexistent_paths": len(nonexistent),
|
||||
"malformed_matrix_rows": len(malformed),
|
||||
"lane_finding_count": sum(len(entries) for entries in sections.values()),
|
||||
"unique_finding_ids": len(sections),
|
||||
"duplicate_finding_ids": sum(len(entries) != 1 for entries in sections.values()),
|
||||
"missing_finding_ids": len(set(expected_ids) - set(sections)),
|
||||
"finding_schema_failures": len(schema_failures),
|
||||
"per_file_sections": per_file_sections,
|
||||
"sed_proof_rows": len(proof_ids),
|
||||
"sed_proof_unique_ids": len(set(proof_ids)),
|
||||
"adversarial_rows": len(adv_ids),
|
||||
"adversarial_unique_ids": len(set(adv_ids)),
|
||||
"adversarial_generic_hits": generic_hits,
|
||||
"priority_unique_ids": len(priority_ids),
|
||||
"unresolved_priority_ids": len(set(priority_ids) - set(expected_ids)),
|
||||
"broken_internal_links": len(broken_links),
|
||||
"forbidden_word_hits": len(forbidden),
|
||||
"missing_required_artifacts": len(missing_artifacts),
|
||||
"decision_evidence_map_files": sum("## Decision Evidence Map" in text for text in claim_texts),
|
||||
"claims_extracted_files": sum("## Claims Extracted" in text for text in claim_texts),
|
||||
"missing_claims_extracted": sum("## Claims Extracted" not in text for text in claim_texts),
|
||||
"unsupported_decision_files": sum("UNSUPPORTED_DECISION" in text for text in claim_texts),
|
||||
"unsupported_decision_occurrences": sum(text.count("UNSUPPORTED_DECISION") for text in claim_texts),
|
||||
"broken_claim_reference_literal_occurrences": sum(text.count("BROKEN_CLAIM_REFERENCE") for text in claim_texts),
|
||||
}
|
||||
for key, value in values.items():
|
||||
print(f"{key}={value}")
|
||||
for label, entries in [
|
||||
("missing", missing), ("extra", extra), ("duplicates", duplicates),
|
||||
("nonexistent", nonexistent), ("malformed", malformed),
|
||||
("schema_failures", schema_failures), ("missing_artifacts", missing_artifacts),
|
||||
("broken_links", broken_links), ("forbidden_hits", forbidden),
|
||||
]:
|
||||
if entries:
|
||||
print(f"{label}=" + " || ".join(entries))
|
||||
Reference in New Issue
Block a user