#!/usr/bin/env python3 """Split UTF-8 Markdown at safe boundaries without losing a code point.""" from __future__ import annotations import argparse import ctypes import errno import os import re import shutil import sys import tempfile from pathlib import Path from typing import Any from harness_common import InputError, atomic_write_json, atomic_write_text, load_rules, sha256_bytes, sha256_text, utc_now from markdown_structure import ( advance_html_block, closing_fence, fence_container_continues, html_tag_spans, indented_code_container, inline_code_spans, inside_any_span, is_thematic_break, mask_closed_fence_candidates, list_continuation_indent, opening_fence, strip_blockquotes, ) H2_RE = re.compile(r"^ {0,3}##(?!#)(?:[ \t]+|$)") def parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser( description="Markdown를 H2·문단 경계에서 무손실 청크로 나눕니다." ) parser.add_argument("--document", required=True, type=Path) parser.add_argument("--output-dir", required=True, type=Path) parser.add_argument("--max-chars", type=int) parser.add_argument("--rules", type=Path) return parser.parse_args() def read_utf8_snapshot(path: Path) -> tuple[Path, bytes, str]: try: resolved = path.expanduser().resolve(strict=True) if not resolved.is_file(): raise InputError(f"문서는 일반 파일이어야 합니다: {path}") data = resolved.read_bytes() text = data.decode("utf-8") except (OSError, UnicodeError) as exc: raise InputError(f"UTF-8 문서를 읽을 수 없습니다: {path}: {exc}") from exc if not text: raise InputError(f"문서가 비어 있습니다: {path}") return resolved, data, text def safe_boundaries(text: str) -> dict[int, str]: """Return safe split offsets. Fenced-code interiors are never returned.""" boundaries: dict[int, str] = {len(text): "eof"} offset = 0 fence_char: str | None = None fence_len = 0 fence_quote_depth = 0 fence_close_mode = "top" fence_list_indent = 0 in_html_comment = False html_state: tuple[str, str] | None = None list_context_indent = 0 in_indented_code = False paragraph_open = False inline_spans = inline_code_spans(mask_closed_fence_candidates(text)) comment_exclusion_spans = list(inline_spans) comment_exclusion_spans.extend( (start, end, "") for start, end in html_tag_spans(text) ) for line in text.splitlines(keepends=True): line_start = offset offset += len(line) if in_html_comment: if "-->" in line: in_html_comment = False boundaries[offset] = "paragraph" continue indented = indented_code_container(line, list_context_indent) is not None if in_indented_code: if indented or not line.strip(" \t\r\n"): continue in_indented_code = False boundaries[line_start] = "paragraph" if fence_char is not None: if closing_fence( line, fence_char, fence_len, fence_quote_depth, fence_close_mode, fence_list_indent, ): fence_char = None fence_len = 0 fence_quote_depth = 0 fence_close_mode = "top" fence_list_indent = 0 boundaries[offset] = "paragraph" continue if fence_container_continues( line, fence_quote_depth, fence_close_mode, fence_list_indent, ): continue fence_char = None fence_len = 0 fence_quote_depth = 0 fence_close_mode = "top" fence_list_indent = 0 boundaries[line_start] = "paragraph" html_line, html_state = advance_html_block( line, html_state, paragraph_open=paragraph_open ) if html_line: paragraph_open = False continue opening = opening_fence(line, list_context_indent) if opening: ( fence_char, fence_len, fence_quote_depth, fence_close_mode, fence_list_indent, ) = opening continue if not paragraph_open and indented: in_indented_code = True continue comment_start = line.find("", comment_start + 4) < 0: in_html_comment = True continue heading_indent = len(line) - len(line.lstrip(" ")) nested_list_heading = bool( list_context_indent and heading_indent >= list_context_indent ) content = line.rstrip("\r\n") if line_start > 0 and H2_RE.match(line) and not nested_list_heading: boundaries[line_start] = "h2" structural, _ = strip_blockquotes(content) thematic_break = is_thematic_break(structural) if thematic_break: boundaries[offset] = "paragraph" paragraph_open = False if line.strip(" \t\r\n") == "": boundaries[offset] = "paragraph" list_context_indent = list_continuation_indent(line, list_context_indent) paragraph_open = bool(content.strip()) and not thematic_break and not re.match( r"^[ ]{0,3}(?:#{1,6}(?:[ \t]+|$)|(?:=+|-+)[ \t]*$)", content, ) # splitlines(keepends=True) omits no characters, including a final line # without a newline. The EOF boundary is always authoritative. boundaries[len(text)] = "eof" boundaries.pop(0, None) return boundaries def choose_chunks( text: str, max_chars: int, h2_fill_ratio: float, ) -> list[tuple[int, int, str]]: boundaries = safe_boundaries(text) offsets = sorted(boundaries) chunks: list[tuple[int, int, str]] = [] start = 0 while start < len(text): limit = start + max_chars within = [value for value in offsets if start < value <= min(limit, len(text))] preferred_h2 = [ value for value in within if boundaries[value] == "h2" and value - start >= max_chars * h2_fill_ratio ] if preferred_h2: end = preferred_h2[-1] reason = "h2" elif within: end = within[-1] reason = boundaries[end] else: after = [value for value in offsets if value > start] end = after[0] if after else len(text) reason = "oversize_atomic_block" if end - start > max_chars else boundaries[end] if end <= start: # Defensive guard against a malformed boundary scan. raise InputError(f"청크 경계를 전진시킬 수 없습니다: offset={start}") if end - start > max_chars: reason = "oversize_atomic_block" elif end == len(text): reason = "eof" chunks.append((start, end, reason)) start = end return chunks def build_split( *, original_path: Path, resolved_path: Path, source_bytes: bytes, text: str, target: Path, max_chars: int, h2_fill_ratio: float, ) -> dict[str, Any]: pieces = choose_chunks(text, max_chars, h2_fill_ratio) manifest_chunks: list[dict[str, Any]] = [] round_trip: list[str] = [] for index, (start, end, reason) in enumerate(pieces, start=1): content = text[start:end] input_name = f"chunk-{index:03d}.input.md" rewritten_name = f"chunk-{index:03d}.rewritten.md" atomic_write_text(target / input_name, content) round_trip.append(content) manifest_chunks.append( { "index": index, "input_file": input_name, "rewritten_file": rewritten_name, "start_offset": start, "end_offset": end, "char_count": len(content), "sha256": sha256_text(content), "boundary_reason": reason, } ) joined = "".join(round_trip) joined_hash = sha256_text(joined) source_hash = sha256_bytes(source_bytes) if joined != text or joined_hash != source_hash: raise InputError("내부 round-trip self-check가 실패했습니다.") return { "schema_version": "1.0", "tool": "split_document", "created_at": utc_now(), "source": { "path": str(original_path), "resolved_path": str(resolved_path), "sha256": source_hash, "size_bytes": len(source_bytes), "char_count": len(text), }, "max_chars": max_chars, "offset_unit": "unicode_codepoint", "chunks": manifest_chunks, "round_trip_sha256": joined_hash, "self_check": True, } def publish_directory_no_clobber(stage: Path, output_dir: Path) -> None: """Atomically rename a complete directory only if destination is absent.""" source_bytes = os.fsencode(stage) destination_bytes = os.fsencode(output_dir) if sys.platform.startswith("linux"): libc = ctypes.CDLL(None, use_errno=True) try: renameat2 = libc.renameat2 except AttributeError as exc: raise InputError("이 Linux libc는 원자적 no-clobber renameat2를 지원하지 않습니다.") from exc renameat2.argtypes = [ctypes.c_int, ctypes.c_char_p, ctypes.c_int, ctypes.c_char_p, ctypes.c_uint] renameat2.restype = ctypes.c_int result = renameat2(-100, source_bytes, -100, destination_bytes, 1) elif sys.platform == "darwin": libc = ctypes.CDLL(None, use_errno=True) try: renamex_np = libc.renamex_np except AttributeError as exc: raise InputError("이 macOS libc는 원자적 no-clobber renamex_np를 지원하지 않습니다.") from exc renamex_np.argtypes = [ctypes.c_char_p, ctypes.c_char_p, ctypes.c_uint] renamex_np.restype = ctypes.c_int result = renamex_np(source_bytes, destination_bytes, 0x00000004) # RENAME_EXCL elif os.name == "nt": try: os.rename(stage, output_dir) except FileExistsError as exc: raise InputError(f"출력 디렉터리가 이미 존재합니다: {output_dir}") from exc return else: raise InputError("이 플랫폼은 원자적 no-clobber 디렉터리 publish를 지원하지 않습니다.") if result == 0: return error_number = ctypes.get_errno() if error_number in {errno.EEXIST, errno.ENOTEMPTY}: raise InputError(f"출력 디렉터리가 이미 존재합니다: {output_dir}") raise OSError(error_number, os.strerror(error_number), str(output_dir)) def main() -> int: args = parse_args() temporary: Path | None = None try: rules = load_rules(args.rules) split_rules = rules["thresholds"]["split"] max_chars = args.max_chars or split_rules["default_max_chars"] fill_ratio = split_rules["minimum_h2_fill_ratio"] if not isinstance(max_chars, int) or isinstance(max_chars, bool) or max_chars < 1: raise InputError("--max-chars는 1 이상의 정수여야 합니다.") if not isinstance(fill_ratio, (int, float)) or not 0 <= fill_ratio <= 1: raise InputError("minimum_h2_fill_ratio는 0과 1 사이여야 합니다.") resolved, source_bytes, text = read_utf8_snapshot(args.document) raw_output = args.output_dir.expanduser() output_dir = Path(os.path.abspath(raw_output)) output_dir.parent.mkdir(parents=True, exist_ok=True) temporary = Path( tempfile.mkdtemp(prefix=f".{output_dir.name}.", dir=output_dir.parent) ) manifest = build_split( original_path=args.document, resolved_path=resolved, source_bytes=source_bytes, text=text, target=temporary, max_chars=max_chars, h2_fill_ratio=float(fill_ratio), ) atomic_write_json(temporary / "manifest.json", manifest) publish_directory_no_clobber(temporary, output_dir) temporary = None print(str(output_dir / "manifest.json")) return 0 except (InputError, KeyError, TypeError, OSError) as exc: print(f"input error: {exc}", file=sys.stderr) return 2 finally: if temporary is not None: shutil.rmtree(temporary, ignore_errors=True) if __name__ == "__main__": raise SystemExit(main())