init: document-haness 설계
This commit is contained in:
@@ -0,0 +1,365 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Split UTF-8 Markdown at safe boundaries without losing a code point."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import ctypes
|
||||
import errno
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import sys
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from harness_common import InputError, atomic_write_json, atomic_write_text, load_rules, sha256_bytes, sha256_text, utc_now
|
||||
from markdown_structure import (
|
||||
advance_html_block,
|
||||
closing_fence,
|
||||
fence_container_continues,
|
||||
html_tag_spans,
|
||||
indented_code_container,
|
||||
inline_code_spans,
|
||||
inside_any_span,
|
||||
is_thematic_break,
|
||||
mask_closed_fence_candidates,
|
||||
list_continuation_indent,
|
||||
opening_fence,
|
||||
strip_blockquotes,
|
||||
)
|
||||
|
||||
|
||||
H2_RE = re.compile(r"^ {0,3}##(?!#)(?:[ \t]+|$)")
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Markdown를 H2·문단 경계에서 무손실 청크로 나눕니다."
|
||||
)
|
||||
parser.add_argument("--document", required=True, type=Path)
|
||||
parser.add_argument("--output-dir", required=True, type=Path)
|
||||
parser.add_argument("--max-chars", type=int)
|
||||
parser.add_argument("--rules", type=Path)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def read_utf8_snapshot(path: Path) -> tuple[Path, bytes, str]:
|
||||
try:
|
||||
resolved = path.expanduser().resolve(strict=True)
|
||||
if not resolved.is_file():
|
||||
raise InputError(f"문서는 일반 파일이어야 합니다: {path}")
|
||||
data = resolved.read_bytes()
|
||||
text = data.decode("utf-8")
|
||||
except (OSError, UnicodeError) as exc:
|
||||
raise InputError(f"UTF-8 문서를 읽을 수 없습니다: {path}: {exc}") from exc
|
||||
if not text:
|
||||
raise InputError(f"문서가 비어 있습니다: {path}")
|
||||
return resolved, data, text
|
||||
|
||||
|
||||
def safe_boundaries(text: str) -> dict[int, str]:
|
||||
"""Return safe split offsets. Fenced-code interiors are never returned."""
|
||||
boundaries: dict[int, str] = {len(text): "eof"}
|
||||
offset = 0
|
||||
fence_char: str | None = None
|
||||
fence_len = 0
|
||||
fence_quote_depth = 0
|
||||
fence_close_mode = "top"
|
||||
fence_list_indent = 0
|
||||
in_html_comment = False
|
||||
html_state: tuple[str, str] | None = None
|
||||
list_context_indent = 0
|
||||
in_indented_code = False
|
||||
paragraph_open = False
|
||||
inline_spans = inline_code_spans(mask_closed_fence_candidates(text))
|
||||
comment_exclusion_spans = list(inline_spans)
|
||||
comment_exclusion_spans.extend(
|
||||
(start, end, "") for start, end in html_tag_spans(text)
|
||||
)
|
||||
|
||||
for line in text.splitlines(keepends=True):
|
||||
line_start = offset
|
||||
offset += len(line)
|
||||
if in_html_comment:
|
||||
if "-->" in line:
|
||||
in_html_comment = False
|
||||
boundaries[offset] = "paragraph"
|
||||
continue
|
||||
indented = indented_code_container(line, list_context_indent) is not None
|
||||
if in_indented_code:
|
||||
if indented or not line.strip(" \t\r\n"):
|
||||
continue
|
||||
in_indented_code = False
|
||||
boundaries[line_start] = "paragraph"
|
||||
if fence_char is not None:
|
||||
if closing_fence(
|
||||
line,
|
||||
fence_char,
|
||||
fence_len,
|
||||
fence_quote_depth,
|
||||
fence_close_mode,
|
||||
fence_list_indent,
|
||||
):
|
||||
fence_char = None
|
||||
fence_len = 0
|
||||
fence_quote_depth = 0
|
||||
fence_close_mode = "top"
|
||||
fence_list_indent = 0
|
||||
boundaries[offset] = "paragraph"
|
||||
continue
|
||||
if fence_container_continues(
|
||||
line,
|
||||
fence_quote_depth,
|
||||
fence_close_mode,
|
||||
fence_list_indent,
|
||||
):
|
||||
continue
|
||||
fence_char = None
|
||||
fence_len = 0
|
||||
fence_quote_depth = 0
|
||||
fence_close_mode = "top"
|
||||
fence_list_indent = 0
|
||||
boundaries[line_start] = "paragraph"
|
||||
|
||||
html_line, html_state = advance_html_block(
|
||||
line, html_state, paragraph_open=paragraph_open
|
||||
)
|
||||
if html_line:
|
||||
paragraph_open = False
|
||||
continue
|
||||
|
||||
opening = opening_fence(line, list_context_indent)
|
||||
if opening:
|
||||
(
|
||||
fence_char,
|
||||
fence_len,
|
||||
fence_quote_depth,
|
||||
fence_close_mode,
|
||||
fence_list_indent,
|
||||
) = opening
|
||||
continue
|
||||
|
||||
if not paragraph_open and indented:
|
||||
in_indented_code = True
|
||||
continue
|
||||
|
||||
comment_start = line.find("<!--")
|
||||
while comment_start >= 0 and inside_any_span(
|
||||
line_start + comment_start, comment_exclusion_spans
|
||||
):
|
||||
comment_start = line.find("<!--", comment_start + 4)
|
||||
if comment_start >= 0 and line.find("-->", comment_start + 4) < 0:
|
||||
in_html_comment = True
|
||||
continue
|
||||
|
||||
heading_indent = len(line) - len(line.lstrip(" "))
|
||||
nested_list_heading = bool(
|
||||
list_context_indent and heading_indent >= list_context_indent
|
||||
)
|
||||
content = line.rstrip("\r\n")
|
||||
if line_start > 0 and H2_RE.match(line) and not nested_list_heading:
|
||||
boundaries[line_start] = "h2"
|
||||
structural, _ = strip_blockquotes(content)
|
||||
thematic_break = is_thematic_break(structural)
|
||||
if thematic_break:
|
||||
boundaries[offset] = "paragraph"
|
||||
paragraph_open = False
|
||||
if line.strip(" \t\r\n") == "":
|
||||
boundaries[offset] = "paragraph"
|
||||
list_context_indent = list_continuation_indent(line, list_context_indent)
|
||||
paragraph_open = bool(content.strip()) and not thematic_break and not re.match(
|
||||
r"^[ ]{0,3}(?:#{1,6}(?:[ \t]+|$)|(?:=+|-+)[ \t]*$)",
|
||||
content,
|
||||
)
|
||||
|
||||
# splitlines(keepends=True) omits no characters, including a final line
|
||||
# without a newline. The EOF boundary is always authoritative.
|
||||
boundaries[len(text)] = "eof"
|
||||
boundaries.pop(0, None)
|
||||
return boundaries
|
||||
|
||||
|
||||
def choose_chunks(
|
||||
text: str,
|
||||
max_chars: int,
|
||||
h2_fill_ratio: float,
|
||||
) -> list[tuple[int, int, str]]:
|
||||
boundaries = safe_boundaries(text)
|
||||
offsets = sorted(boundaries)
|
||||
chunks: list[tuple[int, int, str]] = []
|
||||
start = 0
|
||||
|
||||
while start < len(text):
|
||||
limit = start + max_chars
|
||||
within = [value for value in offsets if start < value <= min(limit, len(text))]
|
||||
preferred_h2 = [
|
||||
value
|
||||
for value in within
|
||||
if boundaries[value] == "h2" and value - start >= max_chars * h2_fill_ratio
|
||||
]
|
||||
if preferred_h2:
|
||||
end = preferred_h2[-1]
|
||||
reason = "h2"
|
||||
elif within:
|
||||
end = within[-1]
|
||||
reason = boundaries[end]
|
||||
else:
|
||||
after = [value for value in offsets if value > start]
|
||||
end = after[0] if after else len(text)
|
||||
reason = "oversize_atomic_block" if end - start > max_chars else boundaries[end]
|
||||
|
||||
if end <= start: # Defensive guard against a malformed boundary scan.
|
||||
raise InputError(f"청크 경계를 전진시킬 수 없습니다: offset={start}")
|
||||
if end - start > max_chars:
|
||||
reason = "oversize_atomic_block"
|
||||
elif end == len(text):
|
||||
reason = "eof"
|
||||
chunks.append((start, end, reason))
|
||||
start = end
|
||||
|
||||
return chunks
|
||||
|
||||
|
||||
def build_split(
|
||||
*,
|
||||
original_path: Path,
|
||||
resolved_path: Path,
|
||||
source_bytes: bytes,
|
||||
text: str,
|
||||
target: Path,
|
||||
max_chars: int,
|
||||
h2_fill_ratio: float,
|
||||
) -> dict[str, Any]:
|
||||
pieces = choose_chunks(text, max_chars, h2_fill_ratio)
|
||||
manifest_chunks: list[dict[str, Any]] = []
|
||||
round_trip: list[str] = []
|
||||
|
||||
for index, (start, end, reason) in enumerate(pieces, start=1):
|
||||
content = text[start:end]
|
||||
input_name = f"chunk-{index:03d}.input.md"
|
||||
rewritten_name = f"chunk-{index:03d}.rewritten.md"
|
||||
atomic_write_text(target / input_name, content)
|
||||
round_trip.append(content)
|
||||
manifest_chunks.append(
|
||||
{
|
||||
"index": index,
|
||||
"input_file": input_name,
|
||||
"rewritten_file": rewritten_name,
|
||||
"start_offset": start,
|
||||
"end_offset": end,
|
||||
"char_count": len(content),
|
||||
"sha256": sha256_text(content),
|
||||
"boundary_reason": reason,
|
||||
}
|
||||
)
|
||||
|
||||
joined = "".join(round_trip)
|
||||
joined_hash = sha256_text(joined)
|
||||
source_hash = sha256_bytes(source_bytes)
|
||||
if joined != text or joined_hash != source_hash:
|
||||
raise InputError("내부 round-trip self-check가 실패했습니다.")
|
||||
|
||||
return {
|
||||
"schema_version": "1.0",
|
||||
"tool": "split_document",
|
||||
"created_at": utc_now(),
|
||||
"source": {
|
||||
"path": str(original_path),
|
||||
"resolved_path": str(resolved_path),
|
||||
"sha256": source_hash,
|
||||
"size_bytes": len(source_bytes),
|
||||
"char_count": len(text),
|
||||
},
|
||||
"max_chars": max_chars,
|
||||
"offset_unit": "unicode_codepoint",
|
||||
"chunks": manifest_chunks,
|
||||
"round_trip_sha256": joined_hash,
|
||||
"self_check": True,
|
||||
}
|
||||
|
||||
|
||||
def publish_directory_no_clobber(stage: Path, output_dir: Path) -> None:
|
||||
"""Atomically rename a complete directory only if destination is absent."""
|
||||
source_bytes = os.fsencode(stage)
|
||||
destination_bytes = os.fsencode(output_dir)
|
||||
if sys.platform.startswith("linux"):
|
||||
libc = ctypes.CDLL(None, use_errno=True)
|
||||
try:
|
||||
renameat2 = libc.renameat2
|
||||
except AttributeError as exc:
|
||||
raise InputError("이 Linux libc는 원자적 no-clobber renameat2를 지원하지 않습니다.") from exc
|
||||
renameat2.argtypes = [ctypes.c_int, ctypes.c_char_p, ctypes.c_int, ctypes.c_char_p, ctypes.c_uint]
|
||||
renameat2.restype = ctypes.c_int
|
||||
result = renameat2(-100, source_bytes, -100, destination_bytes, 1)
|
||||
elif sys.platform == "darwin":
|
||||
libc = ctypes.CDLL(None, use_errno=True)
|
||||
try:
|
||||
renamex_np = libc.renamex_np
|
||||
except AttributeError as exc:
|
||||
raise InputError("이 macOS libc는 원자적 no-clobber renamex_np를 지원하지 않습니다.") from exc
|
||||
renamex_np.argtypes = [ctypes.c_char_p, ctypes.c_char_p, ctypes.c_uint]
|
||||
renamex_np.restype = ctypes.c_int
|
||||
result = renamex_np(source_bytes, destination_bytes, 0x00000004) # RENAME_EXCL
|
||||
elif os.name == "nt":
|
||||
try:
|
||||
os.rename(stage, output_dir)
|
||||
except FileExistsError as exc:
|
||||
raise InputError(f"출력 디렉터리가 이미 존재합니다: {output_dir}") from exc
|
||||
return
|
||||
else:
|
||||
raise InputError("이 플랫폼은 원자적 no-clobber 디렉터리 publish를 지원하지 않습니다.")
|
||||
|
||||
if result == 0:
|
||||
return
|
||||
error_number = ctypes.get_errno()
|
||||
if error_number in {errno.EEXIST, errno.ENOTEMPTY}:
|
||||
raise InputError(f"출력 디렉터리가 이미 존재합니다: {output_dir}")
|
||||
raise OSError(error_number, os.strerror(error_number), str(output_dir))
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
temporary: Path | None = None
|
||||
try:
|
||||
rules = load_rules(args.rules)
|
||||
split_rules = rules["thresholds"]["split"]
|
||||
max_chars = args.max_chars or split_rules["default_max_chars"]
|
||||
fill_ratio = split_rules["minimum_h2_fill_ratio"]
|
||||
if not isinstance(max_chars, int) or isinstance(max_chars, bool) or max_chars < 1:
|
||||
raise InputError("--max-chars는 1 이상의 정수여야 합니다.")
|
||||
if not isinstance(fill_ratio, (int, float)) or not 0 <= fill_ratio <= 1:
|
||||
raise InputError("minimum_h2_fill_ratio는 0과 1 사이여야 합니다.")
|
||||
|
||||
resolved, source_bytes, text = read_utf8_snapshot(args.document)
|
||||
raw_output = args.output_dir.expanduser()
|
||||
output_dir = Path(os.path.abspath(raw_output))
|
||||
output_dir.parent.mkdir(parents=True, exist_ok=True)
|
||||
temporary = Path(
|
||||
tempfile.mkdtemp(prefix=f".{output_dir.name}.", dir=output_dir.parent)
|
||||
)
|
||||
manifest = build_split(
|
||||
original_path=args.document,
|
||||
resolved_path=resolved,
|
||||
source_bytes=source_bytes,
|
||||
text=text,
|
||||
target=temporary,
|
||||
max_chars=max_chars,
|
||||
h2_fill_ratio=float(fill_ratio),
|
||||
)
|
||||
atomic_write_json(temporary / "manifest.json", manifest)
|
||||
publish_directory_no_clobber(temporary, output_dir)
|
||||
temporary = None
|
||||
print(str(output_dir / "manifest.json"))
|
||||
return 0
|
||||
except (InputError, KeyError, TypeError, OSError) as exc:
|
||||
print(f"input error: {exc}", file=sys.stderr)
|
||||
return 2
|
||||
finally:
|
||||
if temporary is not None:
|
||||
shutil.rmtree(temporary, ignore_errors=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user