Files
document-haness/skills/technical-doc-flow/scripts/update_run.py
T

1158 lines
45 KiB
Python

#!/usr/bin/env python3
"""Atomically advance status and record validated artifact omissions."""
from __future__ import annotations
import argparse
from collections import Counter
import os
import sys
import unicodedata
from pathlib import Path
from typing import Any
from harness_common import (
DEFAULT_CONTRACT_PATH,
DEFAULT_RULES_PATH,
InputError,
atomic_write_json,
decode_utf8,
json_text,
load_json,
load_rules,
paths_alias,
require_regular_nonsymlink,
run_lock,
schema_version,
sha256_file,
sha256_text,
utc_now,
validate_with_schema,
)
from lint_document import lint as evaluate_lint
STATUSES = (
"initialized",
"evidence_ready",
"planned",
"drafted",
"reviewed",
"finalized",
"verified",
"hold_for_review",
"failed",
"incomplete",
)
TERMINAL_FAILURES = {"hold_for_review", "failed", "incomplete"}
TERMINAL_ERROR_FIELDS = {
"stage",
"code",
"message",
"affected_artifact",
"retryable",
"safe_next_action",
}
CHECKPOINT_JSON_SCHEMAS = {
"01_sources.json": "sources.schema.json",
"02_reader_contract.json": "reader-contract.schema.json",
"03_evidence_map.json": "evidence-map.schema.json",
"04_logic_map.json": "logic-map.schema.json",
"05_term_ledger.json": "term-ledger.schema.json",
"08_logic_review.json": "review.schema.json",
"08_reader_review.json": "review.schema.json",
"08_lint.json": "lint-report.schema.json",
}
STRUCTURAL_DRAFT_RULES = {
"DOC-H001",
"DOC-H002",
"DOC-M001",
"DOC-M002",
"DOC-M003",
"DOC-M004",
}
def load_runtime_contract() -> dict[str, Any]:
contract = load_json(DEFAULT_CONTRACT_PATH)
validate_with_schema(
contract,
"runtime-contract.schema.json",
str(DEFAULT_CONTRACT_PATH),
)
return contract
def required_artifacts(
contract: dict[str, Any], manifest: dict[str, Any], run_dir: Path
) -> set[str]:
artifacts = contract["artifacts"]
required = set(artifacts["always"]) | set(artifacts[manifest["route_hint"]])
if manifest["mode"] == "review":
required.update(artifacts["review_mode"])
required.discard("final.md")
return required
def artifact_universe(contract: dict[str, Any]) -> set[str]:
artifacts = contract["artifacts"]
universe: set[str] = set()
for branch in ("always", "light", "standard", "deep", "review_mode"):
universe.update(artifacts[branch])
return universe
def validate_run_manifest(manifest: dict[str, Any], manifest_path: Path) -> None:
schema_version(manifest, manifest_path)
validate_with_schema(manifest, "run.schema.json", str(manifest_path))
validate_run_history(manifest)
def validate_run_history(manifest: dict[str, Any]) -> None:
history = manifest.get("history")
if not isinstance(history, list) or not history:
raise InputError("00_run.json history는 비어 있지 않은 배열이어야 합니다.")
first = history[0]
if not isinstance(first, dict) or first.get("from") is not None or first.get("to") != "initialized":
raise InputError("run history는 (null -> initialized)로 시작해야 합니다.")
if not isinstance(first.get("reason"), str) or not first["reason"].strip():
raise InputError("run history[0].reason이 비어 있습니다.")
if first.get("error") is not None:
raise InputError("run history[0].error는 initialized 상태에서 null이어야 합니다.")
previous = "initialized"
for index, entry in enumerate(history[1:], start=1):
if not isinstance(entry, dict):
raise InputError(f"run history[{index}]는 객체여야 합니다.")
source = entry.get("from")
target = entry.get("to")
reason = entry.get("reason")
entry_error = entry.get("error")
if source != previous:
raise InputError(
f"run history[{index}].from={source!r}가 이전 상태 {previous!r}와 다릅니다."
)
if source not in STATUSES or target not in STATUSES:
raise InputError(f"run history[{index}] 상태가 잘못되었습니다.")
allowed = allowed_targets(manifest, source)
if target not in allowed:
raise InputError(f"run history[{index}] 전이가 허용되지 않습니다: {source} -> {target}")
if not isinstance(reason, str) or not reason.strip():
raise InputError(f"run history[{index}].reason이 비어 있습니다.")
if target == "verified" and (
source != "finalized" or reason != "verify_run pass"
):
raise InputError("verified history는 verify_run pass 전이만 허용합니다.")
if target in TERMINAL_FAILURES:
if not isinstance(entry_error, dict):
raise InputError(
f"run history[{index}].error는 terminal 상태에서 객체여야 합니다."
)
elif entry_error is not None:
raise InputError(
f"run history[{index}].error는 non-terminal 상태에서 null이어야 합니다."
)
previous = target
if previous != manifest.get("status"):
raise InputError(
f"run history 최종 상태 {previous!r}와 status {manifest.get('status')!r}가 다릅니다."
)
if manifest.get("error") != history[-1].get("error"):
raise InputError("00_run.json error는 마지막 history error snapshot과 같아야 합니다.")
def terminal_error_record(
current: str,
target: str,
reason: str,
overrides: dict[str, Any] | None,
) -> dict[str, Any]:
record: dict[str, Any] = {
"stage": current,
"code": f"TERMINAL_{target.upper()}",
"message": reason,
"affected_artifact": None,
"retryable": False,
"safe_next_action": (
"기록된 오류와 run artifact를 확인하고 blocker를 해결한 뒤 재개 여부를 판단한다."
),
}
if overrides is None:
return record
if not isinstance(overrides, dict):
raise InputError("terminal error override는 객체여야 합니다.")
unknown = sorted(set(overrides) - TERMINAL_ERROR_FIELDS)
if unknown:
raise InputError(f"알 수 없는 terminal error 필드입니다: {unknown}")
record.update(overrides)
return record
def update_omissions(
run_dir: Path,
manifest: dict[str, Any],
additions: list[tuple[str, str]],
removals: list[str],
contract: dict[str, Any],
) -> list[dict[str, str]]:
universe = artifact_universe(contract)
existing = manifest["omissions"]
existing_names = [item["artifact"] for item in existing]
if any(not isinstance(artifact, str) for artifact in removals):
raise InputError("--unomit artifact는 문자열이어야 합니다.")
normalized: list[dict[str, str]] = []
for item in additions:
if not isinstance(item, (list, tuple)) or len(item) != 2:
raise InputError("--omit에는 artifact와 reason 두 값이 필요합니다.")
artifact, reason = item
if not isinstance(artifact, str) or not isinstance(reason, str):
raise InputError("--omit artifact와 reason은 문자열이어야 합니다.")
if not reason.strip():
raise InputError(f"omission reason이 비어 있습니다: {artifact!r}")
normalized.append({"artifact": artifact, "reason": reason})
removal_counts = Counter(removals)
duplicate_removals = sorted(
name for name, count in removal_counts.items() if count > 1
)
unknown_removals = sorted(set(removals) - universe)
undeclared_removals = sorted(set(removals) - set(existing_names))
addition_names = {item["artifact"] for item in normalized}
conflicting = sorted(addition_names & set(removals))
removal_problems: list[str] = []
if duplicate_removals:
removal_problems.append(f"duplicate_unomit={duplicate_removals}")
if unknown_removals:
removal_problems.append(f"unknown_unomit={unknown_removals}")
if undeclared_removals:
removal_problems.append(f"not_declared={undeclared_removals}")
if conflicting:
removal_problems.append(f"omit_and_unomit={conflicting}")
if removal_problems:
raise InputError("잘못된 omission 철회: " + "; ".join(removal_problems))
removal_set = set(removals)
merged = [
*[item for item in existing if item["artifact"] not in removal_set],
*normalized,
]
declared = [item["artifact"] for item in merged]
required = required_artifacts(contract, manifest, run_dir)
duplicates = sorted(name for name, count in Counter(declared).items() if count > 1)
unknown = sorted(set(declared) - universe)
required_omitted = sorted(set(declared) & required)
present_omitted = sorted(
name for name in set(declared) if os.path.lexists(run_dir / name)
)
problems: list[str] = []
if duplicates:
problems.append(f"duplicate={duplicates}")
if unknown:
problems.append(f"unknown={unknown}")
if required_omitted:
problems.append(f"required={required_omitted}")
if present_omitted:
problems.append(f"present={present_omitted}")
if problems:
raise InputError("잘못된 omission 선언: " + "; ".join(problems))
return merged
def normal_targets(manifest: dict[str, Any]) -> dict[str, set[str]]:
route = manifest.get("route_hint")
mode = manifest.get("mode")
if route not in {"light", "standard", "deep"}:
raise InputError(f"알 수 없는 route_hint입니다: {route!r}")
if mode not in {"write", "revise", "review"}:
raise InputError(f"알 수 없는 mode입니다: {mode!r}")
if route == "light":
initialized = {"planned", "evidence_ready"}
else:
initialized = {"evidence_ready"}
targets: dict[str, set[str]] = {
"initialized": initialized,
"evidence_ready": {"planned"},
"planned": {"reviewed"} if mode == "review" else {"drafted"},
"drafted": {"finalized"} if route == "light" else {"reviewed"},
"reviewed": set() if mode == "review" else {"finalized"},
"finalized": {"verified"} if mode != "review" else set(),
"verified": set(),
"hold_for_review": set(),
"failed": set(),
"incomplete": set(),
}
if route == "light" and mode != "review":
targets["drafted"].add("reviewed")
return targets
def allowed_targets(manifest: dict[str, Any], source: str) -> set[str]:
"""Return legal next states without making failure states recoverable."""
normal = normal_targets(manifest).get(source, set())
if source in TERMINAL_FAILURES:
return set()
return normal | TERMINAL_FAILURES
def normalized_label(value: str) -> str:
return unicodedata.normalize("NFKC", " ".join(value.split())).casefold()
def checkpoint_file(run_dir: Path, name: str) -> Path:
return require_regular_nonsymlink(run_dir / name, name)
def optional_checkpoint_file(run_dir: Path, name: str) -> Path | None:
path = run_dir / name
if not os.path.lexists(path):
return None
return require_regular_nonsymlink(path, name)
def checkpoint_json(run_dir: Path, name: str) -> dict[str, Any]:
path = checkpoint_file(run_dir, name)
value = load_json(path)
schema_name = CHECKPOINT_JSON_SCHEMAS[name]
validate_with_schema(value, schema_name, str(path))
schema_version(value, path)
return value
def validate_utf8_checkpoint(run_dir: Path, name: str) -> Path:
path = checkpoint_file(run_dir, name)
try:
data = path.read_bytes()
except OSError as exc:
raise InputError(f"{name} 파일을 읽을 수 없습니다: {exc}") from exc
decode_utf8(data, name)
return path
def validate_draft_checkpoint(run_dir: Path) -> None:
draft_path = validate_utf8_checkpoint(run_dir, "07_draft.md")
report, _ = evaluate_lint(
argparse.Namespace(
document=str(draft_path),
logic_map=str(run_dir / "04_logic_map.json"),
term_ledger=str(run_dir / "05_term_ledger.json"),
reader_contract=str(run_dir / "02_reader_contract.json"),
baseline=None,
draft_baseline=None,
fail_on="error",
rules=str(DEFAULT_RULES_PATH),
)
)
structural = sorted(
{
finding["rule_id"]
for finding in report["findings"]
if finding["rule_id"] in STRUCTURAL_DRAFT_RULES
}
)
if structural:
raise InputError(
"07_draft.md structural lint가 실패했습니다: " + ", ".join(structural)
)
if report["document"]["sha256"] != sha256_file(draft_path):
raise InputError("07_draft.md가 structural lint 도중 변경되었습니다.")
def validate_review_draft_immutability(
run_dir: Path, manifest: dict[str, Any]
) -> None:
if manifest.get("mode") != "review":
return
draft_inventory = manifest.get("inputs", {}).get("draft")
if not isinstance(draft_inventory, dict) or not isinstance(
draft_inventory.get("sha256"), str
):
raise InputError("review mode의 초기 draft SHA-256 inventory가 없습니다.")
draft_path = validate_utf8_checkpoint(run_dir, "07_draft.md")
if sha256_file(draft_path) != draft_inventory["sha256"]:
raise InputError(
"review mode의 07_draft.md는 run 초기화 때 고정한 원본 draft와 "
"byte-identical이어야 합니다."
)
def current_artifact_hash(
run_dir: Path, name: str, *, optional: bool = False
) -> str | None:
path = (
optional_checkpoint_file(run_dir, name)
if optional
else checkpoint_file(run_dir, name)
)
return sha256_file(path) if path is not None else None
def validate_runtime_provenance(manifest: dict[str, Any]) -> dict[str, Any]:
contract = load_runtime_contract()
contract_hash = sha256_file(DEFAULT_CONTRACT_PATH)
if manifest.get("contract_sha256") != contract_hash:
raise InputError(
"checkpoint의 runtime contract가 run 시작 시 고정한 hash와 다릅니다."
)
rules = load_rules(DEFAULT_RULES_PATH)
rules_hash = sha256_file(DEFAULT_RULES_PATH)
if (
manifest.get("rules_version") != rules.get("rules_version")
or manifest.get("rules_sha256") != rules_hash
):
raise InputError(
"checkpoint의 quality rules version/hash가 run 시작 시 고정한 값과 다릅니다."
)
return contract
def validate_source_registry(
run_dir: Path, manifest: dict[str, Any], sources: dict[str, Any]
) -> None:
inputs = manifest["inputs"]
if sha256_file(checkpoint_file(run_dir, "01_input.md")) != inputs.get(
"input_sha256"
):
raise InputError("01_input.md SHA-256이 00_run.json의 고정값과 다릅니다.")
if sha256_text(json_text(sources)) != inputs.get("sources_manifest_sha256"):
raise InputError(
"01_sources.json 내용이 00_run.json.inputs.sources_manifest_sha256와 "
"다릅니다."
)
if (
inputs.get("brief") != sources.get("brief")
or inputs.get("draft") != sources.get("draft")
or inputs.get("source_count") != len(sources.get("sources", []))
):
raise InputError("01_sources.json inventory와 00_run.json.inputs가 다릅니다.")
brief = sources.get("brief")
draft = sources.get("draft")
if (
not isinstance(brief, dict)
or inputs.get("brief_sha256") != brief.get("sha256")
or inputs.get("draft_sha256")
!= (draft.get("sha256") if isinstance(draft, dict) else None)
):
raise InputError("brief/draft SHA-256 요약이 source registry와 다릅니다.")
inventories = [brief, draft, *sources.get("sources", [])]
for item in inventories:
if item is None:
continue
resolved_path = item.get("resolved_path")
if not isinstance(resolved_path, str):
raise InputError(f"source {item.get('id')!r}의 resolved_path가 없습니다.")
source_path = require_regular_nonsymlink(
Path(resolved_path), f"source {item.get('id')!r}"
)
current_hash = sha256_file(source_path)
if current_hash != item.get("sha256"):
raise InputError(
f"source {item.get('id')!r} SHA-256이 run 초기화 snapshot과 다릅니다."
)
if item.get("size_bytes") != source_path.stat().st_size:
raise InputError(
f"source {item.get('id')!r} size가 run 초기화 snapshot과 다릅니다."
)
def validate_evidence_checkpoint(run_dir: Path) -> None:
sources = checkpoint_json(run_dir, "01_sources.json")
evidence = checkpoint_json(run_dir, "03_evidence_map.json")
inventories = [
sources.get("brief"),
sources.get("draft"),
*sources.get("sources", []),
]
known_source_ids = {
item["id"]
for item in inventories
if isinstance(item, dict) and isinstance(item.get("id"), str)
}
claim_ids: set[str] = set()
premises: dict[str, list[str]] = {}
for claim in evidence["claims"]:
claim_id = claim["id"]
if claim_id in claim_ids:
raise InputError(f"03_evidence_map.json claim id가 중복됩니다: {claim_id}")
claim_ids.add(claim_id)
source_ids = set(claim["source_ids"])
location_pairs = [
(item["source_id"], item["locator"])
for item in claim["source_locations"]
]
if len(location_pairs) != len(set(location_pairs)):
raise InputError(
f"03_evidence_map.json claim {claim_id}의 source locator가 중복됩니다."
)
location_ids = {source_id for source_id, _ in location_pairs}
unknown = (source_ids | location_ids) - known_source_ids
if unknown:
raise InputError(
f"03_evidence_map.json claim {claim_id}가 모르는 source를 참조합니다: "
f"{sorted(unknown)}"
)
if source_ids != location_ids:
raise InputError(
f"03_evidence_map.json claim {claim_id}의 source_ids와 "
"source_locations가 일치하지 않습니다."
)
if claim["status"] == "derived":
premises[claim_id] = list(claim.get("premise_ids", []))
for claim_id, premise_ids in premises.items():
unknown = set(premise_ids) - claim_ids
if unknown or claim_id in premise_ids:
raise InputError(
f"03_evidence_map.json derived claim {claim_id}의 premise_ids가 "
f"잘못되었습니다: {sorted(unknown | ({claim_id} & set(premise_ids)))}"
)
visiting: set[str] = set()
visited: set[str] = set()
def visit(claim_id: str) -> None:
if claim_id in visiting:
raise InputError(
f"03_evidence_map.json derived claim premise cycle이 있습니다: {claim_id}"
)
if claim_id in visited:
return
visiting.add(claim_id)
for premise_id in premises.get(claim_id, []):
visit(premise_id)
visiting.remove(claim_id)
visited.add(claim_id)
for claim_id in claim_ids:
visit(claim_id)
def validate_plan_checkpoint(run_dir: Path, manifest: dict[str, Any]) -> None:
reader = checkpoint_json(run_dir, "02_reader_contract.json")
logic = checkpoint_json(run_dir, "04_logic_map.json")
terms = checkpoint_json(run_dir, "05_term_ledger.json")
configured_term_thresholds = load_rules(DEFAULT_RULES_PATH)["thresholds"]["term"]
expected_budgets = {
"per_sentence": configured_term_thresholds["max_new_terms_per_sentence"],
"per_paragraph": configured_term_thresholds["max_new_terms_per_paragraph"],
"per_section": configured_term_thresholds["max_new_terms_per_section"],
}
if terms.get("budgets") != expected_budgets:
raise InputError(
"05_term_ledger.json budgets가 quality rules와 다릅니다: "
f"recorded={terms.get('budgets')!r}, expected={expected_budgets!r}"
)
if not (
reader.get("document_kind")
== manifest.get("document_kind")
== logic.get("document_kind")
):
raise InputError(
"planned checkpoint의 run/reader/logic document_kind가 일치하지 않습니다."
)
requested_audience = manifest.get("audience")
if isinstance(requested_audience, str) and normalized_label(
requested_audience
) != normalized_label(reader.get("primary_audience", "")):
raise InputError(
"planned checkpoint의 요청 audience와 primary_audience가 일치하지 않습니다."
)
if set(reader["assumed_known"]) != set(terms["assumed_known"]):
raise InputError(
"planned checkpoint의 reader/term assumed_known이 일치하지 않습니다."
)
assumed = {normalized_label(item) for item in reader["assumed_known"]}
must_explain = {normalized_label(item) for item in reader["must_explain"]}
overlap = assumed & must_explain
if overlap:
raise InputError(
f"planned checkpoint에서 assumed_known와 must_explain이 겹칩니다: "
f"{sorted(overlap)}"
)
explainable: set[str] = set()
term_ids: set[str] = set()
expected_term_sections: dict[str, str] = {}
term_name_owners: dict[str, tuple[str, str]] = {}
for term in terms["terms"]:
term_id = term["id"]
if term_id in term_ids:
raise InputError(f"05_term_ledger.json term id가 중복됩니다: {term_id}")
term_ids.add(term_id)
expected_term_sections[term_id] = term["first_section"]
for field_name, name in (
("canonical", term.get("canonical")),
*(("alias", alias) for alias in term.get("aliases", [])),
("english", term.get("english")),
("abbreviation", term.get("abbreviation")),
):
if isinstance(name, str) and name.strip():
normalized = normalized_label(name)
previous = term_name_owners.get(normalized)
if previous is not None:
raise InputError(
f"05_term_ledger.json 용어 이름 {name!r}이 둘 이상에 "
f"배정되었습니다: {previous[0]}.{previous[1]}, "
f"{term_id}.{field_name}"
)
term_name_owners[normalized] = (term_id, field_name)
explainable.add(normalized)
missing_explanations = sorted(
item
for item in reader["must_explain"]
if normalized_label(item) not in explainable
)
if missing_explanations:
raise InputError(
"planned checkpoint의 must_explain 항목이 term ledger에 없습니다: "
f"{missing_explanations}"
)
sections = logic["sections"]
section_ids: set[str] = set()
actual_term_sections: dict[str, list[str]] = {}
for index, section in enumerate(sections):
section_id = section["id"]
if section_id in section_ids:
raise InputError(f"04_logic_map.json section id가 중복됩니다: {section_id}")
unknown_dependencies = set(section["depends_on"]) - section_ids
if unknown_dependencies:
raise InputError(
f"logic section {section_id}가 앞선 절이 아닌 depends_on을 참조합니다: "
f"{sorted(unknown_dependencies)}"
)
if index > 0 and not section["depends_on"]:
raise InputError(f"logic section {section_id}에는 depends_on이 필요합니다.")
expected_transition = sections[index + 1]["id"] if index + 1 < len(sections) else None
if section["transition_to"] != expected_transition:
raise InputError(
f"logic section {section_id}.transition_to는 다음 절 "
f"{expected_transition!r}이어야 합니다."
)
section_ids.add(section_id)
for term_id in section["new_terms"]:
actual_term_sections.setdefault(term_id, []).append(section_id)
unknown_terms = set(actual_term_sections) - term_ids
mismatched_terms = {
term_id: actual_term_sections.get(term_id, [])
for term_id, expected_section in expected_term_sections.items()
if actual_term_sections.get(term_id, []) != [expected_section]
}
if unknown_terms or mismatched_terms:
raise InputError(
"planned checkpoint의 logic/term 연결이 맞지 않습니다: "
f"unknown={sorted(unknown_terms)}, first_section={mismatched_terms}"
)
evidence_path = optional_checkpoint_file(run_dir, "03_evidence_map.json")
if evidence_path is not None:
evidence = checkpoint_json(run_dir, "03_evidence_map.json")
evidence_ids = {claim["id"] for claim in evidence["claims"]}
referenced_claims = {
claim_id for section in sections for claim_id in section["claim_ids"]
}
unknown_claims = referenced_claims - evidence_ids
if unknown_claims:
raise InputError(
"planned checkpoint의 logic map이 모르는 claim을 참조합니다: "
f"{sorted(unknown_claims)}"
)
def validate_review_semantics(review: dict[str, Any], name: str) -> None:
finding_ids: set[str] = set()
blocking = 0
for finding in review["findings"]:
finding_id = finding["id"]
if finding_id in finding_ids:
raise InputError(f"{name} finding id가 중복됩니다: {finding_id}")
finding_ids.add(finding_id)
if finding["severity"] in {"critical", "high"}:
blocking += 1
verdict = review["verdict"]
if verdict == "pass" and blocking:
raise InputError(f"{name} verdict=pass에는 blocking finding이 있을 수 없습니다.")
if verdict in {"revise", "hold_for_review"} and not blocking:
raise InputError(f"{name} verdict={verdict}에는 blocking finding이 필요합니다.")
def expected_review_inputs(run_dir: Path) -> dict[str, str | None]:
return {
"input_sha256": current_artifact_hash(run_dir, "01_input.md"),
"sources_sha256": current_artifact_hash(run_dir, "01_sources.json"),
"reader_contract_sha256": current_artifact_hash(
run_dir, "02_reader_contract.json"
),
"evidence_map_sha256": current_artifact_hash(
run_dir, "03_evidence_map.json", optional=True
),
"logic_map_sha256": current_artifact_hash(run_dir, "04_logic_map.json"),
"term_ledger_sha256": current_artifact_hash(run_dir, "05_term_ledger.json"),
}
def validate_reviews_checkpoint(run_dir: Path, manifest: dict[str, Any]) -> None:
draft_path = checkpoint_file(run_dir, "07_draft.md")
draft_hash = sha256_file(draft_path)
expected_inputs = expected_review_inputs(run_dir)
allowed_verdicts = (
{"pass", "revise"} if manifest["mode"] == "review" else {"pass"}
)
for name, expected_type in (
("08_logic_review.json", "logic"),
("08_reader_review.json", "reader"),
):
review = checkpoint_json(run_dir, name)
validate_review_semantics(review, name)
if review.get("review_type") != expected_type:
raise InputError(f"{name} review_type은 {expected_type!r}이어야 합니다.")
document = review.get("document", {})
recorded_path = document.get("path")
if not isinstance(recorded_path, str):
raise InputError(f"{name} document.path가 없습니다.")
candidate = Path(recorded_path).expanduser()
if not candidate.is_absolute():
candidate = run_dir / candidate
if not paths_alias(candidate, draft_path) or document.get("sha256") != draft_hash:
raise InputError(f"{name}이 현재 07_draft.md의 path/hash를 검토하지 않았습니다.")
if review.get("inputs") != expected_inputs:
raise InputError(f"{name}의 upstream artifact hash 묶음이 현재 값과 다릅니다.")
if review.get("verdict") not in allowed_verdicts:
raise InputError(
f"{name} verdict={review.get('verdict')!r}는 mode={manifest['mode']!r}의 "
"reviewed checkpoint를 통과할 수 없습니다."
)
def fidelity_record_matches(record: Any, path: Path) -> bool:
if not isinstance(record, dict) or not isinstance(record.get("path"), str):
return False
return paths_alias(Path(record["path"]), path) and record.get("sha256") == sha256_file(path)
def validate_lint_checkpoint(
run_dir: Path, manifest: dict[str, Any], *, publish: bool
) -> None:
lint_report = checkpoint_json(run_dir, "08_lint.json")
mode = manifest["mode"]
expected_document = checkpoint_file(
run_dir, "final.md" if publish else "07_draft.md"
)
expected_verdicts = {"pass"} if publish else {"pass", "fail"}
if lint_report.get("tool") != "lint_document" or lint_report.get(
"verdict"
) not in expected_verdicts:
raise InputError(
f"08_lint.json verdict={lint_report.get('verdict')!r}가 checkpoint 조건을 "
"만족하지 않습니다."
)
document = lint_report.get("document", {})
recorded_path = document.get("path")
if not isinstance(recorded_path, str) or not paths_alias(
Path(recorded_path), expected_document
) or document.get("sha256") != sha256_file(expected_document):
raise InputError("08_lint.json 대상 path/hash가 현재 문서와 다릅니다.")
rules = load_rules(DEFAULT_RULES_PATH)
rules_hash = sha256_file(DEFAULT_RULES_PATH)
if (
lint_report.get("rules_version") != rules.get("rules_version")
or lint_report.get("rules_sha256") != rules_hash
or manifest.get("rules_version") != rules.get("rules_version")
or manifest.get("rules_sha256") != rules_hash
):
raise InputError("08_lint.json quality rules version/hash가 현재 run과 다릅니다.")
for artifact_name, hash_field in (
("04_logic_map.json", "logic_map_sha256"),
("05_term_ledger.json", "term_ledger_sha256"),
("02_reader_contract.json", "reader_contract_sha256"),
):
if lint_report.get(hash_field) != current_artifact_hash(run_dir, artifact_name):
raise InputError(f"08_lint.json의 {artifact_name} hash가 현재 값과 다릅니다.")
fidelity = lint_report.get("fidelity", {})
draft_path = checkpoint_file(run_dir, "07_draft.md")
draft_record = fidelity.get("draft_baseline")
if mode in {"write", "revise"}:
if not fidelity_record_matches(draft_record, draft_path):
raise InputError("write/revise lint의 draft_baseline은 현재 07_draft.md여야 합니다.")
if document.get("sha256") != draft_record.get("sha256"):
raise InputError("final.md와 07_draft.md lint hash가 같아야 합니다.")
elif draft_record is not None:
raise InputError("review mode lint에는 draft_baseline을 사용할 수 없습니다.")
original_draft = manifest.get("inputs", {}).get("draft")
baseline_record = fidelity.get("baseline")
baseline_path: Path | None = None
if mode == "revise":
if not isinstance(original_draft, dict):
raise InputError("revise mode의 원본 draft inventory가 없습니다.")
baseline_path = checkpoint_file(
Path(original_draft["resolved_path"]).parent,
Path(original_draft["resolved_path"]).name,
)
original_hash = original_draft.get("sha256")
if sha256_file(baseline_path) != original_hash:
raise InputError(
"revise 원본 draft가 run 초기화 때 고정한 SHA-256과 다릅니다."
)
if (
not fidelity_record_matches(baseline_record, baseline_path)
or baseline_record.get("sha256") != original_hash
):
raise InputError("revise lint baseline은 run이 고정한 원본 draft여야 합니다.")
elif mode == "write" and baseline_record is not None:
raise InputError("write mode lint에는 원본 baseline을 사용할 수 없습니다.")
elif mode == "review" and baseline_record is not None:
if not isinstance(original_draft, dict):
raise InputError("review mode의 원본 draft inventory가 없습니다.")
baseline_path = checkpoint_file(
Path(original_draft["resolved_path"]).parent,
Path(original_draft["resolved_path"]).name,
)
original_hash = original_draft.get("sha256")
if sha256_file(baseline_path) != original_hash:
raise InputError(
"review 원본 draft가 run 초기화 때 고정한 SHA-256과 다릅니다."
)
if (
not fidelity_record_matches(baseline_record, baseline_path)
or baseline_record.get("sha256") != original_hash
):
raise InputError("review lint baseline은 run이 고정한 원본 draft여야 합니다.")
if manifest["route_hint"] == "deep" and lint_report.get("fail_on") != "warning":
raise InputError("deep route lint는 fail_on=warning이어야 합니다.")
replay_args = argparse.Namespace(
document=str(expected_document),
logic_map=str(run_dir / "04_logic_map.json"),
term_ledger=str(run_dir / "05_term_ledger.json"),
reader_contract=str(run_dir / "02_reader_contract.json"),
baseline=str(baseline_path) if baseline_path is not None else None,
draft_baseline=(
str(draft_path) if mode in {"write", "revise"} else None
),
fail_on=lint_report["fail_on"],
rules=str(DEFAULT_RULES_PATH),
)
replay_report, _ = evaluate_lint(replay_args)
comparable = {key: value for key, value in lint_report.items() if key != "generated_at"}
replay_comparable = {
key: value for key, value in replay_report.items() if key != "generated_at"
}
if comparable != replay_comparable:
raise InputError("08_lint.json이 현재 입력의 canonical lint 재실행 결과와 다릅니다.")
def validate_omissions_complete(
run_dir: Path, manifest: dict[str, Any], contract: dict[str, Any]
) -> None:
required = required_artifacts(contract, manifest, run_dir)
universe = artifact_universe(contract)
present = {
name
for name in universe
if (
(run_dir / name).is_file()
and not (run_dir / name).is_symlink()
and (run_dir / name).stat().st_size > 0
)
}
declared = {item["artifact"] for item in manifest["omissions"]}
undeclared = sorted((universe - required - present) - declared)
if undeclared:
raise InputError(
"checkpoint 전에 optional artifact omission 이유를 모두 기록해야 합니다: "
f"{undeclared}"
)
def validate_required_final_artifacts(
run_dir: Path, manifest: dict[str, Any], contract: dict[str, Any]
) -> None:
required = required_artifacts(contract, manifest, run_dir) - {
"00_run.json",
"09_final_report.json",
}
for name in sorted(required):
path = checkpoint_file(run_dir, name)
schema_name = CHECKPOINT_JSON_SCHEMAS.get(name)
if schema_name is not None:
value = load_json(path)
validate_with_schema(value, schema_name, str(path))
schema_version(value, path)
def validate_checkpoint_transition(
run_dir: Path,
manifest: dict[str, Any],
current: str,
target: str,
) -> None:
if target not in {"evidence_ready", "planned", "drafted", "reviewed", "finalized"}:
return
contract = validate_runtime_provenance(manifest)
sources = checkpoint_json(run_dir, "01_sources.json")
validate_source_registry(run_dir, manifest, sources)
validate_review_draft_immutability(run_dir, manifest)
needs_evidence = manifest["route_hint"] in {"standard", "deep"}
has_optional_evidence = os.path.lexists(run_dir / "03_evidence_map.json")
if target == "evidence_ready" or (
(needs_evidence or has_optional_evidence)
and target in {"planned", "drafted", "reviewed", "finalized"}
):
validate_evidence_checkpoint(run_dir)
if target in {"planned", "drafted", "reviewed", "finalized"}:
validate_plan_checkpoint(run_dir, manifest)
if target == "drafted":
validate_draft_checkpoint(run_dir)
elif target == "reviewed":
if manifest["mode"] != "review":
validate_draft_checkpoint(run_dir)
validate_reviews_checkpoint(run_dir, manifest)
validate_omissions_complete(run_dir, manifest, contract)
if manifest["mode"] == "review":
validate_required_final_artifacts(run_dir, manifest, contract)
validate_lint_checkpoint(run_dir, manifest, publish=False)
elif target == "finalized":
validate_draft_checkpoint(run_dir)
validate_omissions_complete(run_dir, manifest, contract)
validate_required_final_artifacts(run_dir, manifest, contract)
if current == "reviewed":
validate_reviews_checkpoint(run_dir, manifest)
elif any(
os.path.lexists(run_dir / name)
for name in ("08_logic_review.json", "08_reader_review.json")
):
raise InputError(
"review artifact가 있으면 reviewed checkpoint를 거친 뒤 finalized로 전이해야 합니다."
)
draft_path = checkpoint_file(run_dir, "07_draft.md")
final_path = checkpoint_file(run_dir, "final.md")
try:
identical = draft_path.read_bytes() == final_path.read_bytes()
except OSError as exc:
raise InputError(f"draft/final byte 비교에 실패했습니다: {exc}") from exc
if not identical:
raise InputError("final.md는 현재 07_draft.md와 byte-identical이어야 합니다.")
validate_lint_checkpoint(run_dir, manifest, publish=True)
def transition_manifest(
manifest_path: Path,
target: str,
*,
reason: str,
allow_same: bool = False,
error: dict[str, Any] | None = None,
) -> dict[str, Any]:
return update_manifest(
manifest_path,
manifest_path.parent,
target=target,
reason=reason,
omissions=[],
unomit=[],
allow_same=allow_same,
error=error,
)
def update_manifest(
manifest_path: Path,
run_dir: Path,
*,
target: str | None,
reason: str,
omissions: list[tuple[str, str]],
unomit: list[str],
allow_same: bool = False,
error: dict[str, Any] | None = None,
) -> dict[str, Any]:
if target is None and not omissions and not unomit:
raise InputError("--status, --omit, --unomit 중 하나 이상이 필요합니다.")
if target is not None and target not in STATUSES:
raise InputError(f"알 수 없는 status입니다: {target}")
if error is not None and target not in TERMINAL_FAILURES:
raise InputError("structured error는 terminal status 전이에만 사용할 수 있습니다.")
manifest = load_json(manifest_path)
validate_run_manifest(manifest, manifest_path)
if omissions or unomit:
contract = load_runtime_contract()
merged_omissions = update_omissions(
run_dir, manifest, omissions, unomit, contract
)
elif target in TERMINAL_FAILURES:
# A verifier must still be able to move a run with invalid business-level
# omission declarations to a terminal failure state. JSON Schema remains
# mandatory, while terminal failure recording must not be blocked by the
# defect it is trying to report.
merged_omissions = manifest["omissions"]
else:
contract = load_runtime_contract()
merged_omissions = update_omissions(run_dir, manifest, [], [], contract)
current = manifest.get("status")
if current not in STATUSES:
raise InputError(f"현재 status가 잘못되었습니다: {current!r}")
if target == current and allow_same and not omissions and not unomit and error is None:
return manifest
history = manifest.get("history")
if not isinstance(history, list):
raise InputError("00_run.json history는 배열이어야 합니다.")
if target is not None and target != current:
allowed = allowed_targets(manifest, current)
if target not in allowed:
raise InputError(f"허용되지 않은 status 전이입니다: {current} -> {target}")
elif target == current and not allow_same:
raise InputError(f"동일한 status로 전이할 수 없습니다: {current}")
if target is not None and (
not isinstance(reason, str) or not reason.strip()
):
raise InputError("status 전이 reason이 비어 있습니다.")
if target is not None and target != current:
checkpoint_manifest = dict(manifest)
checkpoint_manifest["omissions"] = merged_omissions
validate_checkpoint_transition(
run_dir,
checkpoint_manifest,
str(current),
target,
)
now = utc_now()
updated = dict(manifest)
updated["status"] = target if target is not None else current
updated["updated_at"] = now
updated["omissions"] = merged_omissions
next_error = manifest.get("error")
if target is not None and target != current:
next_error = (
terminal_error_record(current, target, reason, error)
if target in TERMINAL_FAILURES
else None
)
updated["error"] = next_error
if target is not None and target != current:
updated["history"] = [
*history,
{
"at": now,
"from": current,
"to": target,
"reason": reason,
"error": next_error,
},
]
validate_run_manifest(updated, manifest_path)
atomic_write_json(manifest_path, updated)
return updated
def update_run(
run_dir: Path,
target: str | None = None,
reason: str = "manual workflow transition",
*,
omissions: list[tuple[str, str]] | None = None,
unomit: list[str] | None = None,
error: dict[str, Any] | None = None,
) -> dict[str, Any]:
try:
resolved = run_dir.expanduser().resolve(strict=True)
except OSError as exc:
raise InputError(f"run directory가 없습니다: {run_dir}") from exc
if not resolved.is_dir():
raise InputError(f"run-dir은 디렉터리여야 합니다: {run_dir}")
manifest_path = require_regular_nonsymlink(
resolved / "00_run.json", "00_run.json"
)
if target == "verified":
raise InputError("verified 상태는 통과한 verify_run.py만 기록할 수 있습니다.")
with run_lock(resolved):
return update_manifest(
manifest_path,
resolved,
target=target,
reason=reason,
omissions=omissions or [],
unomit=unomit or [],
error=error,
)
def parser() -> argparse.ArgumentParser:
value = argparse.ArgumentParser(description=__doc__)
value.add_argument("--run-dir", required=True)
value.add_argument("--status", choices=STATUSES)
value.add_argument("--reason", default="manual workflow transition")
value.add_argument("--error-stage")
value.add_argument("--error-code")
value.add_argument("--error-message")
value.add_argument("--error-affected-artifact")
retryability = value.add_mutually_exclusive_group()
retryability.add_argument(
"--error-retryable",
dest="error_retryable",
action="store_true",
help="terminal error를 안전하게 재시도할 수 있다고 명시",
)
retryability.add_argument(
"--error-not-retryable",
dest="error_retryable",
action="store_false",
help="terminal error를 자동 재시도하면 안 된다고 명시",
)
value.set_defaults(error_retryable=None)
value.add_argument("--error-safe-next-action")
value.add_argument(
"--omit",
nargs=2,
action="append",
default=[],
metavar=("ARTIFACT", "REASON"),
help="생략할 정본 artifact와 구체적인 이유; 여러 번 지정 가능",
)
value.add_argument(
"--unomit",
action="append",
default=[],
metavar="ARTIFACT",
help="기존 omission을 철회; artifact를 만들기 전에 여러 번 지정 가능",
)
return value
def main(argv: list[str] | None = None) -> int:
args = parser().parse_args(argv)
error_values = {
"stage": args.error_stage,
"code": args.error_code,
"message": args.error_message,
"affected_artifact": args.error_affected_artifact,
"retryable": args.error_retryable,
"safe_next_action": args.error_safe_next_action,
}
error = (
{key: value for key, value in error_values.items() if value is not None}
if any(value is not None for value in error_values.values())
else None
)
try:
manifest = update_run(
Path(args.run_dir),
args.status,
args.reason,
omissions=[tuple(item) for item in args.omit],
unomit=args.unomit,
error=error,
)
except InputError as exc:
print(f"input error: {exc}", file=sys.stderr)
return 2
print(f"{manifest['run_id']}\tstatus={manifest['status']}")
return 0
if __name__ == "__main__":
raise SystemExit(main())