Files
document-haness/build/lib/claridoc/lint.py
T

487 lines
26 KiB
Python

from __future__ import annotations
import re
from collections import Counter
from dataclasses import dataclass
from claridoc.models import (
Brief,
DocumentType,
LintIssue,
LintReport,
Outline,
Severity,
SourcePack,
)
from claridoc.utils import line_number, normalize_heading, strip_code_blocks, word_count
GENERIC_HEADINGS = {
"introduction", "intro", "overview", "details", "misc", "other", "summary",
"소개", "개요", "내용", "상세", "기타", "요약",
}
DANGEROUS_PATTERNS = (
r"\brm\s+-rf\b",
r"\bDROP\s+(?:TABLE|DATABASE)\b",
r"\bkubectl\s+delete\b",
r"\bterraform\s+destroy\b",
r"\bgit\s+reset\s+--hard\b",
r"\btruncate\s+table\b",
r"\bDELETE\s+FROM\b",
)
META_LEAK_PATTERNS: tuple[tuple[str, str], ...] = (
(r"제공된\s+(?:근거|자료)(?:\s*팩)?", "Evidence-pack process language leaked into reader-facing prose."),
(r"확인\s*대상으로\s*제시", "Source-processing language leaked into reader-facing prose."),
(r"<\/?(?:BRIEF|SOURCE_PACK|OUTLINE|DETERMINISTIC_LINT|MODEL_REVIEWS)_JSON>", "Prompt tag leaked into the document."),
(r"\b(?:BRIEF|SOURCE_PACK|OUTLINE)_JSON\b", "Prompt artifact name leaked into the document."),
)
CANNED_META_PATTERNS: tuple[tuple[str, str], ...] = (
(r"이\s*절은.{0,100}답한다", "Section-planning narration is visible to the reader."),
(r"This section answers", "Section-planning narration is visible to the reader."),
(r"독자의 목표인", "Prompt-derived audience narration is visible to the reader."),
(r"다룰 핵심 항목은", "Prompt-derived outline narration is visible to the reader."),
)
CHOICE_PATTERN = re.compile(
r"(?:의도적으로|선택(?:했|하였다|한다|했다|하기로)|채택(?:했|하였다|한다|했다)|"
r"허용(?:했|하였다|한다|했다)|유지(?:했|하였다|한다|했다)|제외(?:했|하였다|한다|했다)|"
r"금지(?:했|하였다|한다|했다)|도입(?:했|하였다|한다|했다)|사용하기로|"
r"\b(?:intentionally|chose|chosen|selected|adopted|allowed|kept|rejected|forbids?|decided to)\b)",
re.IGNORECASE,
)
RATIONALE_PATTERN = re.compile(
r"(?:이유|때문|목적|위해|하려|피하|줄이|막기|보장|제약|따라서|왜냐|"
r"because|so that|in order to|to avoid|to reduce|constraint|rationale|reason)",
re.IGNORECASE,
)
TRADEOFF_PATTERN = re.compile(
r"(?:대안|대신|반면|비용|수용|포기|가드레일|경계|금지|한계|"
r"alternative|instead|whereas|cost|accepted|guardrail|boundary|limit|trade-?off|rejected)",
re.IGNORECASE,
)
ORDINAL_PARAGRAPH_OPENING = re.compile(
r"^(?:첫\s*번째|두\s*번째|세\s*번째|네\s*번째|다섯\s*번째|여섯\s*번째|일곱\s*번째|"
r"첫째|둘째|셋째|넷째|다섯째|여섯째|일곱째)"
r"(?:\s+[^.!?\n]{1,28}?)?(?:은|는|이|가)\s",
re.IGNORECASE,
)
@dataclass(slots=True)
class ParsedHeading:
line: int
level: int
title: str
index: int
def lint_document(text: str, brief: Brief, outline: Outline, sources: SourcePack) -> LintReport:
issues: list[LintIssue] = []
headings, fence_openings, fence_balanced = _parse_markdown(text)
def add(code: str, severity: Severity, message: str, *, line: int | None = None,
section: str = "", suggestion: str = "") -> None:
issues.append(LintIssue(code, severity, message, line, section, suggestion))
# Markdown integrity and headings.
if not fence_balanced:
add("MD001", Severity.BLOCKER, "Code fence is not closed.", suggestion="Close every fenced code block.")
for line_no, language in fence_openings:
if not language:
add("MD002", Severity.WARNING, "Code fence has no language tag.", line=line_no,
suggestion="Add a language such as ```python, ```bash, or ```text.")
h1s = [heading for heading in headings if heading.level == 1]
if len(h1s) != 1:
add("STR001", Severity.ERROR, f"Expected exactly one H1, found {len(h1s)}.",
suggestion=f"Use one H1 with the title: {brief.title}")
elif normalize_heading(h1s[0].title) != normalize_heading(brief.title):
add("STR002", Severity.ERROR, "H1 does not match the brief title.", line=h1s[0].line,
suggestion=f"Set the H1 to: {brief.title}")
previous_level = 0
for heading in headings:
if heading.level > brief.constraints.max_heading_depth:
add("STR003", Severity.WARNING,
f"Heading depth {heading.level} exceeds configured maximum {brief.constraints.max_heading_depth}.",
line=heading.line, section=heading.title)
if previous_level and heading.level > previous_level + 1:
add("STR004", Severity.ERROR, f"Heading level jumps from H{previous_level} to H{heading.level}.",
line=heading.line, section=heading.title, suggestion="Do not skip heading levels.")
previous_level = heading.level
normalized_titles = [normalize_heading(heading.title) for heading in headings]
duplicate_titles = {title for title, count in Counter(normalized_titles).items() if title and count > 1}
for duplicate in duplicate_titles:
first = next(heading for heading in headings if normalize_heading(heading.title) == duplicate)
add("STR005", Severity.WARNING, f"Heading is duplicated: {first.title}", line=first.line,
suggestion="Use unique headings that expose each section's distinct job.")
for heading in headings:
if heading.title.casefold().strip(" :") in GENERIC_HEADINGS:
add("STR006", Severity.WARNING, f"Heading is too generic: {heading.title}", line=heading.line,
suggestion="Name the reader question or conclusion handled by the section.")
h2_positions: dict[str, list[int]] = {}
for position, heading in enumerate(headings):
if heading.level == 2:
h2_positions.setdefault(normalize_heading(heading.title), []).append(position)
expected_positions: list[int] = []
for section in outline.sections:
key = normalize_heading(section.title)
if key not in h2_positions:
add("STR007", Severity.ERROR, f"Required H2 is missing: {section.title}", section=section.title,
suggestion="Use every outline H2 exactly once.")
else:
positions = h2_positions[key]
expected_positions.append(positions[0])
if len(positions) > 1:
add("STR009", Severity.ERROR, f"Required H2 appears {len(positions)} times: {section.title}",
section=section.title, suggestion="Use every outline H2 exactly once.")
if expected_positions and expected_positions != sorted(expected_positions):
add("STR008", Severity.ERROR, "Required H2 sections are out of contract order.",
suggestion="Restore the H2 order from outline.json.")
# Reader orientation.
lead = strip_code_blocks(text)[:1800]
lead_words = _content_words(lead)
goal_words = _content_words(brief.reader_goal)
message_words = _content_words(brief.core_message)
if goal_words and not goal_words.intersection(lead_words):
add("AUD001", Severity.WARNING, "The opening does not visibly connect to the reader goal.",
suggestion="State what the reader will be able to do or decide in the first section.")
if message_words and not message_words.intersection(lead_words):
add("AUD002", Severity.WARNING, "The core message is not visible near the start.",
suggestion="Front-load the answer before expanding the reasoning.")
if brief.non_scope and not _contains_any(lead, brief.non_scope):
add("AUD003", Severity.INFO, "Non-scope is not visible near the start.",
suggestion="Mention exclusions that the audience could reasonably expect.")
for pattern, message in META_LEAK_PATTERNS:
for match in re.finditer(pattern, text, flags=re.IGNORECASE | re.DOTALL):
add("META001", Severity.ERROR, message, line=line_number(text, match.start()),
suggestion="Remove authoring/evidence-process language and write the supported point directly.")
for pattern, message in CANNED_META_PATTERNS:
for match in re.finditer(pattern, text, flags=re.IGNORECASE | re.DOTALL):
add("META002", Severity.WARNING, message, line=line_number(text, match.start()),
suggestion="Replace the planning sentence with the actual claim, situation, or transition.")
opening_contract_terms = (
"이 글의 독자는", "읽고 나면", "범위는", "비범위", "적용 맥락",
"the intended readers", "after reading", "scope:", "non-scope:", "version/date context",
)
opening_contract_count = sum(term in lead.casefold() for term in opening_contract_terms)
if brief.document_type == DocumentType.TECHNICAL_BLOG and opening_contract_count >= 3:
add("OPEN001", Severity.ERROR, "The opening reads like a prompt contract rather than a technical story.",
suggestion="Open with a concrete situation, observable problem, cost, or decision tension.")
# Paragraph and sentence focus.
prose = strip_code_blocks(text)
paragraphs = _paragraphs(prose)
formulaic_ordinal_openings = [
(paragraph, start_index)
for paragraph, start_index in paragraphs
if ORDINAL_PARAGRAPH_OPENING.search(paragraph)
]
if brief.is_korean and brief.document_type == DocumentType.TECHNICAL_BLOG:
for index in range(max(0, len(formulaic_ordinal_openings) - 2)):
cluster = formulaic_ordinal_openings[index:index + 3]
if cluster[-1][1] - cluster[0][1] > 2400:
continue
add(
"STYLE001",
Severity.WARNING,
"Three nearby paragraphs use formulaic ordinal openings that expose the outline as prose.",
line=line_number(prose, cluster[0][1]),
suggestion=(
"State the concrete actor, state, change, consequence, or decision directly. "
"If the items are truly ordered or parallel, use a list or meaningful subheadings."
),
)
break
long_paragraph_count = 0
crowded_paragraph_count = 0
long_sentence_count = 0
for paragraph, start_index in paragraphs:
if len(paragraph) > 900 and long_paragraph_count < 5:
add("READ001", Severity.WARNING, f"Paragraph is long ({len(paragraph)} characters).",
line=line_number(prose, start_index), suggestion="Split at the change of idea or reasoning step.")
long_paragraph_count += 1
sentences = [item.strip() for item in re.split(r"(?<=[.!?。!?])\s+|(?<=다\.)\s*", paragraph) if item.strip()]
if len(sentences) > 6 and crowded_paragraph_count < 5:
add("READ002", Severity.WARNING, f"Paragraph contains {len(sentences)} sentences.",
line=line_number(prose, start_index), suggestion="Keep one central point per paragraph.")
crowded_paragraph_count += 1
for sentence in sentences:
if word_count(sentence) > 55 and long_sentence_count < 5:
add("READ003", Severity.WARNING, "Sentence is unusually long.",
line=line_number(prose, start_index), suggestion="Split the sentence at a logical dependency.")
long_sentence_count += 1
break
# Type-specific contract checks.
lowered = prose.casefold()
numbered_steps = bool(re.search(r"(?m)^\s*\d+[.)]\s+\S", prose))
has_code_or_example = "```" in text or bool(re.search(r"예시|example|worked example|사례", lowered))
has_verification = bool(re.search(r"검증|확인|성공 기준|expected (?:result|output)|verify|validation", lowered))
has_prerequisites = bool(re.search(r"사전|준비|prerequisite|before you begin|requirements", lowered))
has_tradeoffs = bool(re.search(r"트레이드오프|trade-?off|대안|alternative|한계|limit|실패 조건", lowered))
has_rollback = bool(re.search(r"롤백|원복|복구|rollback|revert|recovery", lowered))
if brief.document_type in {DocumentType.TUTORIAL, DocumentType.HOW_TO, DocumentType.TROUBLESHOOTING}:
if not numbered_steps:
add("TYPE001", Severity.ERROR, "Procedural document has no numbered steps.",
suggestion="Use ordered steps with one primary action per step.")
if not has_prerequisites:
add("TYPE002", Severity.ERROR, "Procedural document does not state prerequisites.")
if not has_verification:
add("TYPE003", Severity.ERROR, "Procedural document lacks an observable verification step.")
if brief.document_type in {DocumentType.HOW_TO, DocumentType.TROUBLESHOOTING, DocumentType.DESIGN_DOC} and not has_rollback:
add("TYPE004", Severity.ERROR, "Document type requires rollback or recovery guidance.")
if brief.document_type in {DocumentType.TECHNICAL_BLOG, DocumentType.TUTORIAL, DocumentType.EXPLANATION} and not has_code_or_example:
add("TYPE005", Severity.ERROR, "Document lacks a concrete or worked example.")
if brief.document_type in {DocumentType.TECHNICAL_BLOG, DocumentType.EXPLANATION, DocumentType.DESIGN_DOC} and not has_tradeoffs:
add("TYPE006", Severity.ERROR, "Document does not discuss alternatives, limits, or trade-offs.")
if brief.document_type == DocumentType.REFERENCE and "|" not in text:
add("TYPE007", Severity.WARNING, "Reference document has no table-like lookup surface.",
suggestion="Use a table for fields, parameters, defaults, or errors when appropriate.")
# Choice rationale and decision completeness.
if brief.document_type in {DocumentType.TECHNICAL_BLOG, DocumentType.EXPLANATION, DocumentType.DESIGN_DOC}:
for index, (paragraph, start_index) in enumerate(paragraphs):
if not CHOICE_PATTERN.search(paragraph):
continue
next_paragraph = paragraphs[index + 1][0] if index + 1 < len(paragraphs) else ""
context = f"{paragraph}\n{next_paragraph}"
if not RATIONALE_PATTERN.search(context):
add("RAT001", Severity.ERROR,
"A technical choice is declared without explaining why it was made.",
line=line_number(prose, start_index),
suggestion="State the relevant constraint and the reason in the same or next paragraph; otherwise remove or qualify the intentional-choice claim.")
if not TRADEOFF_PATTERN.search(context):
add("RAT002", Severity.WARNING,
"A technical choice does not expose an alternative, accepted cost, or guardrail.",
line=line_number(prose, start_index),
suggestion="Name the realistic alternative and the boundary or cost accepted with the choice.")
for section in outline.sections:
if section.decision_requirements and brief.constraints.require_citations and sources.sources and not section.evidence_ids:
add("RAT003", Severity.ERROR, f"Decision section has no allocated evidence: {section.title}",
section=section.title, suggestion="Retrieve a source that explicitly contains the decision rationale or record the evidence gap.")
# Evidence and claim hygiene.
known_marker_pattern = None
used_markers: set[str] = set()
if sources.ids:
alternatives = "|".join(re.escape(source_id) for source_id in sorted(sources.ids, key=len, reverse=True))
known_marker_pattern = re.compile(rf"\[({alternatives})\]")
used_markers = set(known_marker_pattern.findall(text))
source_like_pattern = re.compile(r"\[((?:SRC|S|L)[A-Za-z0-9_-]+)\]")
unknown_markers = sorted(set(source_like_pattern.findall(text)) - sources.ids)
for marker in unknown_markers:
add("EVD001", Severity.ERROR, f"Unknown source marker: [{marker}]",
suggestion="Use a valid public citation form or remove the unsupported marker.")
if brief.constraints.require_citations and not sources.sources:
add("EVD002", Severity.ERROR, "Evidence is required but the source pack is empty.",
suggestion="Provide a source pack or collect evidence from a local documentation corpus.")
citation_style = brief.constraints.citation_style
if citation_style == "source_id":
if brief.constraints.require_citations and sources.sources and not (used_markers & sources.ids):
add("EVD003", Severity.ERROR, "No source-pack citation markers are used.",
suggestion="Attach [SOURCE_ID] to each source-backed claim.")
uncited_numeric = 0
if brief.constraints.require_citations and sources.sources:
for paragraph, start_index in paragraphs:
if uncited_numeric >= 4:
break
if not re.search(r"\d", paragraph):
continue
if known_marker_pattern and known_marker_pattern.search(paragraph):
continue
if re.search(r"예시|가정|illustrative|example|단계|step|명령", paragraph.casefold()):
continue
add("EVD004", Severity.WARNING, "A numeric or version-like claim has no source marker.",
line=line_number(prose, start_index), suggestion="Cite it, qualify it, or mark it as illustrative.")
uncited_numeric += 1
unused_sources = sorted(sources.ids - used_markers)
if unused_sources:
add("EVD005", Severity.INFO, f"Source-pack entries not cited: {', '.join(unused_sources)}")
else:
for marker in sorted(used_markers):
match = re.search(rf"\[{re.escape(marker)}\]", text)
add("EVD007", Severity.ERROR, f"Internal source marker leaked into reader-facing prose: [{marker}]",
line=line_number(text, match.start()) if match else None,
suggestion="Remove the marker. Keep claim provenance in the generated evidence-map sidecar.")
if citation_style == "hidden":
for source in sources.sources:
if source.path and source.path in text:
match = re.search(re.escape(source.path), text)
add("META004", Severity.ERROR, f"Internal repository path leaked into the document: {source.path}",
line=line_number(text, match.start()) if match else None,
suggestion="Describe the supported technical point; keep the path in provenance.md.")
for forbidden in brief.forbidden_claims:
if forbidden.casefold() in lowered:
add("EVD006", Severity.BLOCKER, f"Forbidden claim appears in the document: {forbidden}",
suggestion="Remove the claim or change the brief deliberately.")
# Safety, unresolved placeholders, and version context.
for match in re.finditer(r"\b(?:TODO|TBD|FIXME)\b|\{\{[^}]+\}\}", text, flags=re.IGNORECASE):
add("FIN001", Severity.ERROR, f"Unresolved placeholder: {match.group(0)}", line=line_number(text, match.start()))
for pattern in DANGEROUS_PATTERNS:
for match in re.finditer(pattern, text, flags=re.IGNORECASE):
context = text[max(0, match.start() - 500): min(len(text), match.end() + 500)].casefold()
requirements = {
"impact warning": r"경고|주의|영향|위험|warning|caution|impact|risk",
"checkpoint or recovery": r"백업|체크포인트|스냅샷|롤백|원복|복구|backup|checkpoint|snapshot|rollback|revert|recovery",
"verification": r"검증|확인|예상 결과|성공 기준|verify|validation|expected (?:effect|result|output)|success criterion",
}
missing = [name for name, safety_pattern in requirements.items() if not re.search(safety_pattern, context)]
if missing:
add("SAFE001", Severity.BLOCKER,
f"Destructive command lacks nearby safety controls ({', '.join(missing)}): {match.group(0)}",
line=line_number(text, match.start()),
suggestion="Add impact warning, checkpoint/recovery path, expected effect, and verification.")
if (
brief.constraints.date_policy == "always"
and brief.constraints.version_context
and brief.constraints.version_context.casefold() not in lowered
):
add("VER001", Severity.WARNING, "Required material version/date context is not stated in the document.",
suggestion=f"State the applicable context naturally: {brief.constraints.version_context}")
date_boilerplate = re.compile(
r"(?:예시|문서|이\s*글|자료).{0,40}\b20\d{2}-\d{2}-\d{2}\b.{0,20}기준|"
r"(?:example|document|article).{0,40}\b20\d{2}-\d{2}-\d{2}\b.{0,25}(?:as of|checked)",
re.IGNORECASE | re.DOTALL,
)
for match in date_boilerplate.finditer(text):
add("DATE001", Severity.ERROR, "Access-date or example-date boilerplate leaked into the article.",
line=line_number(text, match.start()),
suggestion="Remove the date unless it materially changes behavior, compatibility, or reproducibility.")
if brief.constraints.date_policy != "always":
for source in sources.sources:
if source.accessed and source.accessed in text:
match = re.search(re.escape(source.accessed), text)
add("DATE002", Severity.WARNING, f"A source access date appears in reader-facing prose: {source.accessed}",
line=line_number(text, match.start()) if match else None,
suggestion="Keep access dates in provenance metadata, not in the article.")
total_words = word_count(text)
target = brief.constraints.target_words
if total_words < target * 0.45:
add("LEN001", Severity.ERROR, f"Document is substantially under target ({total_words}/{target} words).")
elif total_words < target * 0.65:
add("LEN002", Severity.WARNING, f"Document is under target ({total_words}/{target} words).")
elif total_words > target * 1.6:
add("LEN003", Severity.WARNING, f"Document is substantially over target ({total_words}/{target} words).")
penalties = {
Severity.BLOCKER: 25.0,
Severity.ERROR: 8.0,
Severity.WARNING: 2.5,
Severity.INFO: 0.5,
}
score = max(0.0, round(100.0 - sum(penalties[issue.severity] for issue in issues), 1))
severity_counts = Counter(issue.severity.value for issue in issues)
metrics = {
"heading_count": len(headings),
"h2_count": sum(heading.level == 2 for heading in headings),
"source_count": len(sources.sources),
"cited_source_count": len(used_markers & sources.ids),
"citation_style": brief.constraints.citation_style,
"decision_section_count": sum(bool(section.decision_requirements) for section in outline.sections),
"numbered_steps": numbered_steps,
"formulaic_ordinal_opening_count": len(formulaic_ordinal_openings),
"has_verification": has_verification,
"has_tradeoffs": has_tradeoffs,
"severity_counts": dict(severity_counts),
}
return LintReport(score=score, word_count=total_words, issues=issues, metrics=metrics)
def render_lint_markdown(report: LintReport) -> str:
lines = [
"# Deterministic lint report",
"",
f"- Score: **{report.score:.1f}/100**",
f"- Word count: **{report.word_count}**",
f"- Issues: **{len(report.issues)}**",
"",
]
if not report.issues:
lines.append("No issues found.\n")
return "\n".join(lines)
lines.extend(["| Severity | Code | Location | Finding | Suggested correction |", "|---|---|---|---|---|"])
for issue in report.issues:
location = f"line {issue.line}" if issue.line else (issue.section or "—")
message = issue.message.replace("|", "\\|")
suggestion = issue.suggestion.replace("|", "\\|") if issue.suggestion else "—"
lines.append(f"| {issue.severity.value} | `{issue.code}` | {location} | {message} | {suggestion} |")
lines.append("")
return "\n".join(lines)
def _parse_markdown(text: str) -> tuple[list[ParsedHeading], list[tuple[int, str]], bool]:
headings: list[ParsedHeading] = []
openings: list[tuple[int, str]] = []
in_fence = False
offset = 0
for line_no, raw_line in enumerate(text.splitlines(keepends=True), start=1):
line = raw_line.rstrip("\r\n")
fence = re.match(r"^\s*```\s*([^\s`]*)", line)
if fence:
if not in_fence:
openings.append((line_no, fence.group(1).strip()))
in_fence = not in_fence
offset += len(raw_line)
continue
if not in_fence:
match = re.match(r"^(#{1,6})\s+(.+?)\s*#*\s*$", line)
if match:
headings.append(ParsedHeading(line_no, len(match.group(1)), match.group(2).strip(), offset))
offset += len(raw_line)
return headings, openings, not in_fence
def _paragraphs(text: str) -> list[tuple[str, int]]:
result: list[tuple[str, int]] = []
cursor = 0
for match in re.finditer(r"(?:^|\n\s*\n)([^\n].*?)(?=\n\s*\n|\Z)", text, flags=re.DOTALL):
paragraph = match.group(1).strip()
if not paragraph:
continue
if paragraph.startswith("#") or re.match(r"^(?:[-*+] |\d+[.)] )", paragraph):
continue
if paragraph.startswith("|"):
continue
result.append((paragraph, match.start(1)))
cursor = match.end()
return result
def _content_words(text: str) -> set[str]:
stop = {
"그리고", "하지만", "대한", "통해", "위한", "에서", "으로", "하는", "한다", "문서", "독자", "이글",
"the", "and", "for", "with", "from", "that", "this", "what", "when", "into", "your", "document",
}
return {
word.casefold()
for word in re.findall(r"[0-9A-Za-z가-힣]+", text)
if len(word) >= 2 and word.casefold() not in stop
}
def _contains_any(text: str, phrases: list[str]) -> bool:
lowered = text.casefold()
for phrase in phrases:
tokens = _content_words(phrase)
if tokens and any(token in lowered for token in tokens):
return True
return False