from __future__ import annotations import hashlib import math import os import re from collections import Counter, defaultdict from dataclasses import dataclass from pathlib import Path from typing import Iterable, Sequence from claridoc.models import Brief, Source, SourcePack, ValidationError DEFAULT_INCLUDES: tuple[str, ...] = ( "wiki/projects", "wiki/concepts", "raw/branch-notes", "raw/official-docs", "raw/company-tech-blogs", ) ALLOWED_SUFFIXES = frozenset({".md", ".markdown", ".mdx", ".txt", ".rst", ".adoc", ".json", ".yaml", ".yml"}) SKIP_DIRS = frozenset({".git", ".hg", ".svn", "node_modules", ".venv", "venv", "dist", "build", "target", "__pycache__"}) MAX_FILE_BYTES = 2_000_000 MAX_CHUNK_CHARS = 4_000 _SOURCE_WEIGHTS = { "canonical-project": 2.6, "canonical-concept": 2.3, "branch-note": 2.15, "official-doc": 1.85, "company-tech-blog": 1.45, "local-document": 1.0, } _DECISION_TERMS = ( "결정", "선택", "이유", "근거", "대안", "트레이드오프", "trade-off", "tradeoff", "제약", "허용", "금지", "비용", "decision evidence map", "decision", "rationale", "alternative", "constraint", ) _TOKEN_RE = re.compile(r"[A-Za-z][A-Za-z0-9_.:/@-]*|[가-힣]{2,}|\d+(?:\.\d+)*") _HEADING_RE = re.compile(r"^(#{1,6})\s+(.+?)\s*#*\s*$") _FRONTMATTER_RE = re.compile(r"\A---\s*\n(.*?)\n---\s*(?:\n|\Z)", re.DOTALL) _CLAIM_RE = re.compile(r"\b(?:DEC-[A-Z0-9_-]+@\d+|[A-Z][A-Z0-9_-]+-C\d+|D\d{1,3})\b") @dataclass(frozen=True, slots=True) class CorpusChunk: path: str title: str heading: str line_start: int line_end: int text: str source_type: str status: str base_weight: float claim_ids: tuple[str, ...] decision_ids: tuple[str, ...] @dataclass(frozen=True, slots=True) class RankedChunk: chunk: CorpusChunk score: float def build_query_from_brief(brief: Brief) -> str: """Build a retrieval query that asks for both subject matter and decision rationale.""" parts = [ brief.title, brief.reader_goal, brief.core_message, *brief.scope, *brief.required_topics, ] if brief.document_type.value in {"technical_blog", "design_doc", "explanation"}: parts.extend(["선택 이유 근거 대안 트레이드오프 제약 비용 구현 검증", "decision rationale alternative trade-off"]) return "\n".join(item.strip() for item in parts if item and item.strip()) def collect_sources( root: str | Path, query: str, *, includes: Sequence[str] | None = None, top_k: int = 24, max_per_file: int = 3, ) -> SourcePack: """Read a local documentation repository and return ranked evidence chunks. The output is intentionally an internal evidence pack. Absolute paths are not placed in the pack; sources use stable repository-relative paths. """ root_path = Path(root).expanduser().resolve() if not root_path.is_dir(): raise ValidationError(f"source root is not a directory: {root_path}") if not query.strip(): raise ValidationError("corpus query must not be empty") if top_k < 1 or top_k > 500: raise ValidationError("source top_k must be between 1 and 500") if max_per_file < 1 or max_per_file > 20: raise ValidationError("source max_per_file must be between 1 and 20") include_paths = tuple(includes or DEFAULT_INCLUDES) files = list(_iter_files(root_path, include_paths)) chunks: list[CorpusChunk] = [] for path in files: chunks.extend(_read_chunks(root_path, path)) ranked = rank_chunks(chunks, query, top_k=top_k, max_per_file=max_per_file) return SourcePack(sources=[_ranked_to_source(item) for item in ranked]) def merge_source_packs(*packs: SourcePack) -> SourcePack: seen: set[str] = set() sources: list[Source] = [] for pack in packs: for source in pack.sources: candidate = source.id if candidate in seen: suffix = 2 while f"{candidate}_{suffix}" in seen: suffix += 1 data = pack_source_dict(source) data["id"] = f"{candidate}_{suffix}" source = Source.from_dict(data) seen.add(source.id) sources.append(source) return SourcePack(sources=sources) def pack_source_dict(source: Source) -> dict[str, object]: return { "id": source.id, "title": source.title, "url": source.url, "publisher": source.publisher, "accessed": source.accessed, "facts": list(source.facts), "notes": source.notes, "source_type": source.source_type, "status": source.status, "path": source.path, "heading": source.heading, "line_start": source.line_start, "line_end": source.line_end, "claim_ids": list(source.claim_ids), "decision_ids": list(source.decision_ids), "priority": source.priority, } def rank_chunks( chunks: Sequence[CorpusChunk], query: str, *, top_k: int, max_per_file: int, ) -> list[RankedChunk]: if not chunks: return [] query_tokens = _tokens(query) if not query_tokens: return [] docs = [Counter(_tokens(f"{chunk.title} {chunk.heading} {chunk.text}")) for chunk in chunks] document_frequency: Counter[str] = Counter() for doc in docs: document_frequency.update(doc.keys()) average_length = sum(sum(doc.values()) for doc in docs) / max(1, len(docs)) scored: list[RankedChunk] = [] for chunk, doc in zip(chunks, docs): length = max(1, sum(doc.values())) bm25 = 0.0 for token in query_tokens: tf = doc.get(token, 0) if not tf: continue df = document_frequency[token] idf = math.log(1 + (len(docs) - df + 0.5) / (df + 0.5)) denominator = tf + 1.5 * (1 - 0.75 + 0.75 * length / max(1.0, average_length)) bm25 += idf * (tf * 2.5 / denominator) normalized = f"{chunk.heading}\n{chunk.text}".casefold() phrase_bonus = sum(0.65 for term in _DECISION_TERMS if term in normalized) exact_bonus = sum(1.25 for phrase in _query_phrases(query) if phrase in normalized) status_bonus = _status_weight(chunk.status) score = (bm25 + phrase_bonus + exact_bonus + status_bonus) * chunk.base_weight if score > 0: scored.append(RankedChunk(chunk, round(score, 6))) scored.sort(key=lambda item: (-item.score, item.chunk.path, item.chunk.line_start)) per_file: defaultdict[str, int] = defaultdict(int) selected: list[RankedChunk] = [] for item in scored: if per_file[item.chunk.path] >= max_per_file: continue selected.append(item) per_file[item.chunk.path] += 1 if len(selected) >= top_k: break return selected def _iter_files(root: Path, includes: Sequence[str]) -> Iterable[Path]: seen_real: set[Path] = set() for include in includes: candidate = (root / include).resolve() if include not in {".", ""} else root if not candidate.exists(): continue if candidate.is_file(): paths = [candidate] else: paths = [] for current, dirs, filenames in os.walk(candidate, followlinks=True): dirs[:] = [name for name in dirs if name not in SKIP_DIRS] current_path = Path(current) real_current = current_path.resolve() if real_current in seen_real: dirs[:] = [] continue seen_real.add(real_current) paths.extend(current_path / name for name in filenames) for path in sorted(paths): if path.suffix.casefold() not in ALLOWED_SUFFIXES: continue try: if path.stat().st_size > MAX_FILE_BYTES: continue except OSError: continue yield path def _read_chunks(root: Path, path: Path) -> list[CorpusChunk]: try: text = path.read_text(encoding="utf-8") except (UnicodeDecodeError, OSError): return [] try: relative = path.relative_to(root).as_posix() except ValueError: relative = path.name metadata, body, frontmatter_lines = _split_frontmatter(text) status = metadata.get("status", "") or metadata.get("status_label", "") title = metadata.get("title", "") or path.stem.replace("-", " ") source_type = _classify_source(relative) base_weight = _SOURCE_WEIGHTS[source_type] lines = body.splitlines() chunks: list[CorpusChunk] = [] headings: list[tuple[int, int, str]] = [] for index, line in enumerate(lines): match = _HEADING_RE.match(line) if match: headings.append((index, len(match.group(1)), match.group(2).strip())) if not headings: headings = [(0, 1, title)] for position, (start, _level, heading) in enumerate(headings): end = headings[position + 1][0] if position + 1 < len(headings) else len(lines) raw = "\n".join(lines[start:end]).strip() if not raw: continue raw_lines = raw.splitlines() if len(raw_lines) == 1 and _HEADING_RE.match(raw_lines[0]): # A heading with no body is navigation, not evidence. Keeping it can # outrank a lower section merely because the title repeats query terms. continue for part_index, (offset_start, offset_end, part) in enumerate(_split_large_chunk(raw), start=1): absolute_start = frontmatter_lines + start + 1 + offset_start absolute_end = min(frontmatter_lines + end, absolute_start + offset_end - offset_start) effective_heading = heading if part_index == 1 else f"{heading} (part {part_index})" ids = sorted(set(_CLAIM_RE.findall(part))) decision_ids = tuple(item for item in ids if item.startswith("DEC-") or re.fullmatch(r"D\d{1,3}", item)) claim_ids = tuple(item for item in ids if item not in decision_ids) chunks.append( CorpusChunk( path=relative, title=title, heading=effective_heading, line_start=max(1, absolute_start), line_end=max(absolute_start, absolute_end), text=part.strip(), source_type=source_type, status=status, base_weight=base_weight, claim_ids=claim_ids, decision_ids=decision_ids, ) ) return chunks def _split_frontmatter(text: str) -> tuple[dict[str, str], str, int]: match = _FRONTMATTER_RE.match(text) if not match: return {}, text, 0 metadata: dict[str, str] = {} for line in match.group(1).splitlines(): if ":" not in line or line[:1].isspace(): continue key, value = line.split(":", 1) metadata[key.strip()] = value.strip().strip('"\'') consumed = text[: match.end()].count("\n") return metadata, text[match.end() :], consumed def _split_large_chunk(text: str) -> list[tuple[int, int, str]]: if len(text) <= MAX_CHUNK_CHARS: return [(0, text.count("\n") + 1, text)] lines = text.splitlines() result: list[tuple[int, int, str]] = [] start = 0 buffer: list[str] = [] chars = 0 for index, line in enumerate(lines): extra = len(line) + 1 if buffer and chars + extra > MAX_CHUNK_CHARS: result.append((start, index, "\n".join(buffer))) start = index buffer = [] chars = 0 buffer.append(line) chars += extra if buffer: result.append((start, len(lines), "\n".join(buffer))) return result def _classify_source(relative: str) -> str: normalized = relative.replace("\\", "/").casefold() if normalized.startswith("wiki/projects/"): return "canonical-project" if normalized.startswith("wiki/concepts/"): return "canonical-concept" if normalized.startswith("raw/branch-notes/"): return "branch-note" if normalized.startswith("raw/official-docs/"): return "official-doc" if normalized.startswith("raw/company-tech-blogs/"): return "company-tech-blog" return "local-document" def _status_weight(status: str) -> float: normalized = status.casefold() if any(term in normalized for term in ("verified", "reviewed", "published-ready", "actually-implemented", "locally-verified")): return 1.6 if any(term in normalized for term in ("planned", "documented-only", "needs-confirmation", "raw", "draft")): return -0.2 return 0.0 def _tokens(text: str) -> list[str]: return [token.casefold() for token in _TOKEN_RE.findall(text) if len(token) > 1] def _query_phrases(query: str) -> list[str]: phrases: list[str] = [] for line in query.splitlines(): phrase = re.sub(r"\s+", " ", line).strip().casefold() if 4 <= len(phrase) <= 140: phrases.append(phrase) return phrases[:12] def _ranked_to_source(item: RankedChunk) -> Source: chunk = item.chunk digest = hashlib.sha256(f"{chunk.path}:{chunk.line_start}:{chunk.heading}".encode("utf-8")).hexdigest()[:10] return Source( id=f"L{digest}", title=f"{chunk.title} — {chunk.heading}", url=f"repo:///{chunk.path}", publisher="local documentation corpus", facts=[chunk.text], notes=( "Internal retrieval excerpt. Preserve provenance in the sidecar evidence map; " "do not copy repository paths, source IDs, access dates, or process language into reader-facing prose." ), source_type=chunk.source_type, status=chunk.status, path=chunk.path, heading=chunk.heading, line_start=chunk.line_start, line_end=chunk.line_end, claim_ids=list(chunk.claim_ids), decision_ids=list(chunk.decision_ids), priority=item.score, )