1807 lines
70 KiB
Python
1807 lines
70 KiB
Python
"""Deterministic hard-gate validators for generated resumes.
|
||
|
||
The validators in this module deliberately operate on the typed intermediate
|
||
representation instead of rendered Markdown. They are conservative where a
|
||
regular expression could otherwise confuse a technology term with personal
|
||
data, and they never include a detected PII value in a finding message.
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
from collections.abc import Iterable
|
||
from dataclasses import dataclass
|
||
from datetime import datetime, timezone
|
||
from decimal import Decimal, InvalidOperation
|
||
import re
|
||
import unicodedata
|
||
|
||
from .models import (
|
||
CandidateProfile,
|
||
ConstraintKind,
|
||
DraftClaim,
|
||
EvidenceItem,
|
||
GenerationConfig,
|
||
JobAnalysis,
|
||
PostingConstraint,
|
||
QualityCategory,
|
||
QualityFinding,
|
||
QualitySeverity,
|
||
ResumeDraft,
|
||
ResumeMode,
|
||
SectionType,
|
||
SensitiveDataCategory,
|
||
_claim_mentions_requirement,
|
||
)
|
||
|
||
|
||
_RESIDENT_ID_PATTERN = re.compile(
|
||
r"(?<!\d)\d{6}\s*[-\u2013\u2014]\s*[1-8]\d{6}(?!\d)"
|
||
)
|
||
_EMAIL_PATTERN = re.compile(
|
||
r"(?<![\w.+-])[A-Z0-9.!#$%&'*+/=?^_`{|}~-]+@"
|
||
r"(?:[A-Z0-9](?:[A-Z0-9-]{0,61}[A-Z0-9])?\.)+"
|
||
r"[A-Z]{2,63}(?![\w.-])",
|
||
re.IGNORECASE,
|
||
)
|
||
_PHONE_PATTERN = re.compile(
|
||
r"(?<!\d)"
|
||
r"(?:(?:\+|00)82[\s.-]?)?"
|
||
r"(?:0?(?:1[016789]|2|3[1-3]|4[1-4]|5[1-5]|6[1-4]|70))"
|
||
r"[\s.)-]?\d{3,4}[\s.-]?\d{4}"
|
||
r"(?!\d)"
|
||
)
|
||
_BANK_ACCOUNT_PATTERN = re.compile(
|
||
r"(?:계좌(?:\s*번호)?|은행\s*계좌)\s*[:\uff1a]?\s*\d[\d\s-]{7,}\d"
|
||
)
|
||
_PLACEHOLDER_PATTERNS = (
|
||
re.compile(r"(?<![A-Z])TBD(?![A-Z])", re.IGNORECASE),
|
||
re.compile(r"(?:[\[({<\u3010]\s*)?확인\s*필요(?:\s*[\])}>\u3011])?(?![가-힣])"),
|
||
)
|
||
|
||
# Numeric identifiers containing ASCII letters (B2B, OAuth2, HTTP/2, EC2,
|
||
# ISO-27001) are not quantitative claims. Keeping them out of the number
|
||
# validator avoids a common and costly false positive.
|
||
_NUMBER_PATTERN = re.compile(
|
||
r"(?<![A-Za-z0-9_])"
|
||
r"[+-]?(?:\d{1,3}(?:,\d{3})+|\d+)(?:\.\d+)?"
|
||
r"(?![0-9_])"
|
||
)
|
||
_IDENTIFIER_CHARS = frozenset(
|
||
"abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789_./:+-"
|
||
)
|
||
_NUMBER_UNIT_PATTERN = re.compile(
|
||
r"\s*(%|퍼센트|milliseconds?|ms|seconds?|secs?|s|명|건|개|회|배|"
|
||
r"개월|년|월|일|시간|분|초|억원|만원|원|TB|GB|MB|KB)",
|
||
re.IGNORECASE,
|
||
)
|
||
_ASCII_NUMBER_WITH_UNIT_PATTERN = re.compile(
|
||
r"^[+-]?[\d,.]+(?:milliseconds?|ms|seconds?|secs?|s|TB|GB|MB|KB)$",
|
||
re.IGNORECASE,
|
||
)
|
||
_RATIO_METRIC_KEY_PATTERN = re.compile(
|
||
r"(?:rate|ratio|percent|percentage|pct|reduction|increase|decrease|"
|
||
r"improvement|률|율|비율|증가|감소|개선)",
|
||
re.IGNORECASE,
|
||
)
|
||
_NUMERIC_DATE_RANGE_PATTERN = re.compile(
|
||
r"(?<!\d)(?P<sy>(?:19|20)\d{2})[./-](?P<sm>\d{1,2})"
|
||
r"(?:[./-](?P<sd>\d{1,2}))?\s*"
|
||
r"(?:~|\u2013|\u2014|\u301c|\uFF5E|부터)\s*"
|
||
r"(?P<ey>(?:19|20)\d{2})[./-](?P<em>\d{1,2})"
|
||
r"(?:[./-](?P<ed>\d{1,2}))?(?:까지)?(?!\d)"
|
||
)
|
||
|
||
|
||
# These patterns require an explicit label, value, or first-person construction.
|
||
# Bare words such as "사진" and "장애" are intentionally absent because they
|
||
# frequently describe legitimate engineering work.
|
||
_SENSITIVE_PATTERNS: tuple[
|
||
tuple[SensitiveDataCategory, tuple[re.Pattern[str], ...]], ...
|
||
] = (
|
||
(
|
||
SensitiveDataCategory.PHOTO,
|
||
(
|
||
re.compile(r"(?:증명|반명함|여권|프로필)\s*사진"),
|
||
re.compile(r"사진\s*(?:첨부|부착|제출)"),
|
||
),
|
||
),
|
||
(
|
||
SensitiveDataCategory.BIRTH_DATE,
|
||
(
|
||
re.compile(r"(?:생년월일|출생(?:일|연월일)?)\s*[:\uff1a]?"),
|
||
re.compile(
|
||
r"(?<!\d)(?:19|20)\d{2}\s*년\s*\d{1,2}\s*월\s*"
|
||
r"\d{1,2}\s*일\s*(?:생|출생)(?![가-힣])"
|
||
),
|
||
re.compile(r"(?<!\d)(?:19|20)\d{2}\s*년생(?![가-힣])"),
|
||
),
|
||
),
|
||
(
|
||
SensitiveDataCategory.GENDER,
|
||
(
|
||
re.compile(r"성별\s*[:\uff1a]\s*(?:남(?:성)?|여(?:성)?)(?![가-힣])"),
|
||
re.compile(r"저는\s*(?:남성|여성)(?:입니다|이고|이며)"),
|
||
),
|
||
),
|
||
(
|
||
SensitiveDataCategory.FULL_ADDRESS,
|
||
(
|
||
re.compile(r"(?:상세\s*)?주소\s*[:\uff1a]\s*\S+"),
|
||
re.compile(r"거주지\s*[:\uff1a]\s*\S+"),
|
||
),
|
||
),
|
||
(
|
||
SensitiveDataCategory.MARITAL_STATUS,
|
||
(
|
||
re.compile(r"혼인\s*여부"),
|
||
re.compile(r"결혼\s*여부"),
|
||
re.compile(r"(?:기혼|미혼)(?:입니다|\s|[,.)]|$)"),
|
||
),
|
||
),
|
||
(
|
||
SensitiveDataCategory.FAMILY_DETAILS,
|
||
(
|
||
re.compile(r"가족\s*관계"),
|
||
re.compile(r"(?:부친|모친|배우자|자녀)\s*[:\uff1a]"),
|
||
re.compile(r"(?:아버지|어머니|부모님)(?:의|은|는|이|가|께서)"),
|
||
),
|
||
),
|
||
(
|
||
SensitiveDataCategory.RELIGION,
|
||
(
|
||
re.compile(r"종교\s*[:\uff1a]\s*\S+"),
|
||
re.compile(r"저의?\s*신앙은?\s*\S+"),
|
||
),
|
||
),
|
||
(
|
||
SensitiveDataCategory.DISABILITY,
|
||
(
|
||
re.compile(r"장애\s*(?:여부|등급|유형)\s*[:\uff1a]?"),
|
||
re.compile(r"장애인\s*등록\s*(?:여부|번호)"),
|
||
),
|
||
),
|
||
(
|
||
SensitiveDataCategory.HEALTH,
|
||
(
|
||
re.compile(r"혈액형\s*[:\uff1a]\s*(?:A|B|AB|O)[+-]?\s*형?", re.IGNORECASE),
|
||
re.compile(r"(?:신장|키|체중)\s*[:\uff1a]\s*\d"),
|
||
re.compile(r"건강\s*상태\s*[:\uff1a]"),
|
||
),
|
||
),
|
||
(
|
||
SensitiveDataCategory.MILITARY_DETAILS,
|
||
(
|
||
re.compile(r"병역\s*(?:사항|구분|여부|상태)?\s*[:\uff1a]"),
|
||
re.compile(r"(?:군필|미필|병역\s*면제)(?:입니다|\s|[,.)]|$)"),
|
||
),
|
||
),
|
||
(
|
||
SensitiveDataCategory.COMPENSATION,
|
||
(
|
||
re.compile(r"(?:현재|희망)?\s*연봉\s*[:\uff1a]\s*\d"),
|
||
re.compile(r"(?:현재|희망)\s*급여\s*[:\uff1a]\s*\d"),
|
||
),
|
||
),
|
||
(
|
||
SensitiveDataCategory.POLITICAL_OPINION,
|
||
(
|
||
re.compile(
|
||
r"(?:정치적\s*견해|지지\s*정당|정당\s*가입\s*여부)\s*[::]?"
|
||
),
|
||
re.compile(r"저는\s*\S+당을\s*지지합니다"),
|
||
),
|
||
),
|
||
(
|
||
SensitiveDataCategory.PROPERTY,
|
||
(
|
||
re.compile(
|
||
r"(?:재산\s*(?:총액|보유액|내역)?|보유\s*부동산|자산\s*총액)"
|
||
r"\s*[::]\s*\S+"
|
||
),
|
||
),
|
||
),
|
||
(
|
||
SensitiveDataCategory.BANK_ACCOUNT,
|
||
(_BANK_ACCOUNT_PATTERN,),
|
||
),
|
||
)
|
||
|
||
_BLIND_SCHOOL_PATTERNS = (
|
||
re.compile(r"(?:학교명|출신\s*학교)\s*[:\uff1a]?"),
|
||
re.compile(
|
||
r"(?<![가-힣A-Za-z0-9])"
|
||
r"[가-힣A-Za-z0-9\u00b7]{2,}(?:대학교|고등학교|중학교)"
|
||
r"(?=에서|의|를|은|는|이|가|졸업|재학|수료|[\s,.)]|$)"
|
||
),
|
||
# Korean university abbreviations are routinely written as ``서울대`` or
|
||
# ``연세대``. Requiring education context avoids treating ordinary words
|
||
# such as ``최대`` as institutions while still catching natural prose such
|
||
# as ``서울대에서 전공`` and ``고대 출신``.
|
||
re.compile(
|
||
r"(?<![가-힣A-Za-z0-9])"
|
||
r"(?!최대|상대|절대|확대|세대|일대|휴대|군대|근대|현대)"
|
||
r"[가-힣]{1,8}대"
|
||
r"(?=에서|의|를|은|는|이|가|출신|졸업|재학|수료|전공|[\s,.)]|$)"
|
||
),
|
||
re.compile(
|
||
r"(?<![A-Za-z0-9])(?:SNU|KAIST|POSTECH|UNIST|GIST|DGIST)"
|
||
r"(?=에서|의|를|은|는|이|가|출신|졸업|재학|수료|전공|[\s,.)]|$)",
|
||
re.IGNORECASE,
|
||
),
|
||
)
|
||
_BLIND_ORIGIN_PATTERNS = (
|
||
# Explicit origin labels are sensitive even when the following region is
|
||
# not in a finite city-name dictionary. Requiring a label delimiter or a
|
||
# Korean topic/nominative particle avoids matching ordinary compounds such
|
||
# as ``고향사랑기부제`` and job-relevant phrases such as ``지역 고객``.
|
||
re.compile(
|
||
r"(?:출신\s*(?:지역|지)|고향)\s*"
|
||
r"(?:[:\uff1a=]|은|는|이|가)(?=\s*\S)"
|
||
),
|
||
# Colloquial profile text often omits both the label delimiter and topic
|
||
# particle (``고향 대전``). It is still an explicit place-of-origin
|
||
# disclosure. Exclude the common ``고향 사랑`` policy phrase so that a
|
||
# job-related public-service project does not become a false positive.
|
||
re.compile(
|
||
r"(?:출신\s*(?:지역|지)|고향)\s+"
|
||
r"(?!(?:사랑|기부|정책|문화|서비스)(?=[\s,.)]|$))"
|
||
r"[가-힣]{2,20}(?=[\s,.)]|$)"
|
||
),
|
||
re.compile(
|
||
r"(?:서울|부산|대구|인천|광주|대전|울산|세종|제주|"
|
||
r"[가-힣]{2,}(?:시|도|군|구))\s*출신"
|
||
r"(?:입니다|이고|이며|으로|자|[\s,.]|$)"
|
||
),
|
||
re.compile(
|
||
r"(?:서울|부산|대구|인천|광주|대전|울산|세종|제주|"
|
||
r"[가-힣]{2,}(?:시|도|군|구))에서\s*"
|
||
r"(?:태어나|태어났|출생|성장|자라|자랐)"
|
||
),
|
||
)
|
||
_BLIND_IDENTITY_PATTERN = re.compile(
|
||
r"(?:이름|성명)\s*[:\uff1a]\s*"
|
||
r"[가-힣A-Za-z](?:\s*[가-힣A-Za-z]){1,39}(?![가-힣A-Za-z])"
|
||
)
|
||
_TECH_TERM_PATTERN = re.compile(
|
||
r"(?<![A-Za-z0-9_])"
|
||
r"[A-Za-z][A-Za-z0-9]*(?:[.+#/-][A-Za-z0-9]+)*"
|
||
r"(?![A-Za-z0-9_])"
|
||
)
|
||
_TECH_TERM_STOPWORDS = frozenset(
|
||
{
|
||
"a",
|
||
"an",
|
||
"and",
|
||
"as",
|
||
"at",
|
||
"by",
|
||
"for",
|
||
"from",
|
||
"in",
|
||
"of",
|
||
"on",
|
||
"or",
|
||
"the",
|
||
"to",
|
||
"using",
|
||
"via",
|
||
"with",
|
||
}
|
||
)
|
||
_KOREAN_TECH_TERMS = (
|
||
"쿠버네티스",
|
||
"카프카",
|
||
"도커",
|
||
"레디스",
|
||
"리액트",
|
||
"스프링",
|
||
"장고",
|
||
"플라스크",
|
||
"젠킨스",
|
||
"테라폼",
|
||
"그라파나",
|
||
"프로메테우스",
|
||
)
|
||
_HIGH_RISK_KOREAN_CLAIM_TERMS = (
|
||
"대규모",
|
||
"글로벌",
|
||
"전사",
|
||
"총괄",
|
||
"아키텍처",
|
||
"분산 시스템",
|
||
"최초",
|
||
"유일",
|
||
"최고",
|
||
"전문가",
|
||
"책임자",
|
||
"팀장",
|
||
"수상",
|
||
"우승",
|
||
"최우수상",
|
||
"대통령상",
|
||
"승진",
|
||
"특허",
|
||
"논문",
|
||
"자격증",
|
||
"공인 인증",
|
||
"매출",
|
||
"투자 유치",
|
||
)
|
||
_GROUNDING_STOPWORDS = frozenset(
|
||
{
|
||
"경험",
|
||
"업무",
|
||
"담당",
|
||
"관련",
|
||
"기반",
|
||
"통해",
|
||
"위해",
|
||
"대한",
|
||
"사용",
|
||
"활용",
|
||
"수행",
|
||
"진행",
|
||
"참여",
|
||
"과정",
|
||
"결과",
|
||
"역할",
|
||
"있습니다",
|
||
"했습니다",
|
||
"합니다",
|
||
"하였다",
|
||
"했다",
|
||
"있는",
|
||
"그리고",
|
||
"또한",
|
||
"대한",
|
||
}
|
||
)
|
||
_KOREAN_GROUNDING_TOKEN_PATTERN = re.compile(r"[가-힣]{2,}")
|
||
_BLIND_AGE_PATTERNS = (
|
||
re.compile(r"나이\s*[:\uff1a]\s*(?:만\s*)?\d{1,3}\s*세?"),
|
||
re.compile(r"만\s*\d{1,3}\s*세(?:입니다|이고|이며|[\s,.]|$)"),
|
||
)
|
||
_EMPLOYER_NAME_PATTERNS = (
|
||
re.compile(
|
||
r"(?:회사명|기업명|기관명|근무\s*기관|근무처|직장명)"
|
||
r"\s*[:\uff1a]\s*\S+"
|
||
),
|
||
re.compile(
|
||
r"(?<![가-힣A-Za-z0-9])"
|
||
r"[가-힣A-Za-z0-9\u00b7]{2,}\s*"
|
||
r"(?:주식회사|유한회사|공사|공단|재단|병원)"
|
||
r"(?=에서|의|를|은|는|이|가|[\s,.)]|$)"
|
||
),
|
||
)
|
||
_EMPLOYER_FIELD_KEYWORDS = (
|
||
"회사명",
|
||
"기업명",
|
||
"기관명",
|
||
"근무기관",
|
||
"근무처",
|
||
"직장명",
|
||
)
|
||
_GENERIC_LOCATION_CONTEXT = (
|
||
"프로젝트",
|
||
"팀",
|
||
"환경",
|
||
"업무",
|
||
"과정",
|
||
"서비스",
|
||
"시스템",
|
||
"현장",
|
||
"조직",
|
||
"회사",
|
||
"기관",
|
||
"지역",
|
||
"애플리케이션",
|
||
"플랫폼",
|
||
"데이터베이스",
|
||
"클라우드",
|
||
"백엔드",
|
||
"프론트엔드",
|
||
"코드베이스",
|
||
"저장소",
|
||
"브라우저",
|
||
"서버",
|
||
"클라이언트",
|
||
"네트워크",
|
||
"컨테이너",
|
||
"파이프라인",
|
||
"Python",
|
||
"Java",
|
||
"Kotlin",
|
||
"AWS",
|
||
"GCP",
|
||
"Azure",
|
||
"Linux",
|
||
"Kubernetes",
|
||
"Docker",
|
||
"서울",
|
||
"부산",
|
||
"대전",
|
||
"대구",
|
||
"광주",
|
||
"인천",
|
||
)
|
||
|
||
_CONSTRAINT_CATEGORY_KEYWORDS: tuple[
|
||
tuple[SensitiveDataCategory, tuple[str, ...]], ...
|
||
] = (
|
||
(SensitiveDataCategory.PHOTO, ("사진", "외모")),
|
||
(SensitiveDataCategory.BIRTH_DATE, ("생년월일", "생일", "출생")),
|
||
(SensitiveDataCategory.GENDER, ("성별", "성별정보")),
|
||
(SensitiveDataCategory.FULL_ADDRESS, ("주소", "거주지")),
|
||
(SensitiveDataCategory.MARITAL_STATUS, ("혼인", "결혼")),
|
||
(SensitiveDataCategory.FAMILY_DETAILS, ("가족", "부모", "배우자", "자녀")),
|
||
(SensitiveDataCategory.RELIGION, ("종교", "신앙")),
|
||
(SensitiveDataCategory.DISABILITY, ("장애",)),
|
||
(SensitiveDataCategory.HEALTH, ("건강", "혈액형", "신장", "체중")),
|
||
(SensitiveDataCategory.MILITARY_DETAILS, ("병역", "군필", "미필")),
|
||
(SensitiveDataCategory.COMPENSATION, ("연봉", "급여")),
|
||
(
|
||
SensitiveDataCategory.POLITICAL_OPINION,
|
||
("정치적견해", "지지정당", "정당가입"),
|
||
),
|
||
(SensitiveDataCategory.PROPERTY, ("재산", "부동산", "자산총액")),
|
||
(SensitiveDataCategory.BANK_ACCOUNT, ("계좌",)),
|
||
)
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class _TextTarget:
|
||
text: str
|
||
location: str
|
||
claim: DraftClaim | None = None
|
||
|
||
@property
|
||
def claim_id(self) -> str | None:
|
||
return self.claim.claim_id if self.claim is not None else None
|
||
|
||
@property
|
||
def evidence_ids(self) -> list[str]:
|
||
return list(self.claim.evidence_ids) if self.claim is not None else []
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class _NumericToken:
|
||
value: Decimal
|
||
display: str
|
||
is_percent: bool
|
||
unit: str | None
|
||
|
||
|
||
class _FindingCollector:
|
||
def __init__(self) -> None:
|
||
self.findings: list[QualityFinding] = []
|
||
|
||
def add(
|
||
self,
|
||
*,
|
||
code: str,
|
||
severity: QualitySeverity,
|
||
category: QualityCategory,
|
||
message: str,
|
||
location: str | None = None,
|
||
claim_id: str | None = None,
|
||
evidence_ids: Iterable[str] = (),
|
||
suggestion: str | None = None,
|
||
) -> None:
|
||
unique_evidence = list(dict.fromkeys(evidence_ids))
|
||
self.findings.append(
|
||
QualityFinding(
|
||
finding_id=f"deterministic-{len(self.findings) + 1:04d}",
|
||
code=code,
|
||
severity=severity,
|
||
category=category,
|
||
message=message,
|
||
location=location,
|
||
claim_id=claim_id,
|
||
evidence_ids=unique_evidence,
|
||
suggestion=suggestion,
|
||
)
|
||
)
|
||
|
||
|
||
def _normalise_claim_text(text: str) -> str:
|
||
normalised = unicodedata.normalize("NFKC", text).casefold()
|
||
normalised = re.sub(r"\s+", " ", normalised).strip()
|
||
normalised = re.sub(r"^[\-*•·]\s*", "", normalised)
|
||
return normalised.rstrip(".!?。 ")
|
||
|
||
|
||
def _candidate_identity_patterns(profile: CandidateProfile) -> tuple[re.Pattern[str], ...]:
|
||
patterns: list[re.Pattern[str]] = [_BLIND_IDENTITY_PATTERN]
|
||
for name in (profile.name, profile.name_en):
|
||
if not name:
|
||
continue
|
||
escaped = re.escape(name)
|
||
if re.search(r"[가-힣]", name):
|
||
compact_name = re.sub(r"\s+", "", name)
|
||
flexible_name = r"\s*".join(
|
||
re.escape(character) for character in compact_name
|
||
)
|
||
patterns.append(
|
||
re.compile(
|
||
rf"(?<![가-힣A-Za-z0-9]){flexible_name}"
|
||
rf"(?=(?:은|는|이|가|을|를|의|에게|께서|으로|입니다|이라고|"
|
||
rf"[\s,.)]|$))",
|
||
re.IGNORECASE,
|
||
)
|
||
)
|
||
else:
|
||
patterns.append(
|
||
re.compile(rf"(?<!\w){escaped}(?!\w)", re.IGNORECASE)
|
||
)
|
||
return tuple(patterns)
|
||
|
||
|
||
def _unsupported_technical_terms(
|
||
claim: DraftClaim, supporting_facts: Iterable[EvidenceItem]
|
||
) -> list[str]:
|
||
facts = tuple(supporting_facts)
|
||
evidence_text = " ".join(
|
||
[
|
||
part
|
||
for fact in facts
|
||
for part in (
|
||
fact.content,
|
||
" ".join(fact.keywords),
|
||
" ".join(str(key) for key in fact.metrics),
|
||
" ".join(str(value) for value in fact.metrics.values()),
|
||
)
|
||
]
|
||
).casefold()
|
||
evidence_tokens = {
|
||
token.casefold() for token in _TECH_TERM_PATTERN.findall(evidence_text)
|
||
}
|
||
|
||
unsupported: list[str] = []
|
||
for token in _TECH_TERM_PATTERN.findall(claim.text):
|
||
folded = token.casefold()
|
||
if folded in _TECH_TERM_STOPWORDS:
|
||
continue
|
||
if folded in evidence_tokens:
|
||
continue
|
||
if folded == "api" and any(known.endswith("api") for known in evidence_tokens):
|
||
continue
|
||
if folded in {"ci", "cd"} and any(
|
||
known in {"ci/cd", "ci-cd"} for known in evidence_tokens
|
||
):
|
||
continue
|
||
unsupported.append(token)
|
||
for term in _KOREAN_TECH_TERMS:
|
||
if term in claim.text and term.casefold() not in evidence_text:
|
||
unsupported.append(term)
|
||
for term in _HIGH_RISK_KOREAN_CLAIM_TERMS:
|
||
if term in claim.text and term.casefold() not in evidence_text:
|
||
unsupported.append(term)
|
||
return list(dict.fromkeys(unsupported))
|
||
|
||
|
||
def _grounding_tokens(text: str) -> set[str]:
|
||
normalised = unicodedata.normalize("NFKC", text).casefold()
|
||
tokens = {
|
||
token.casefold()
|
||
for token in _TECH_TERM_PATTERN.findall(normalised)
|
||
if token.casefold() not in _TECH_TERM_STOPWORDS
|
||
}
|
||
tokens.update(
|
||
token
|
||
for token in _KOREAN_GROUNDING_TOKEN_PATTERN.findall(normalised)
|
||
if token not in _GROUNDING_STOPWORDS
|
||
)
|
||
return tokens
|
||
|
||
|
||
def _korean_common_prefix(left: str, right: str) -> int:
|
||
length = 0
|
||
for left_char, right_char in zip(left, right):
|
||
if left_char != right_char:
|
||
break
|
||
length += 1
|
||
return length
|
||
|
||
|
||
def _token_is_supported(token: str, evidence_tokens: set[str]) -> bool:
|
||
if token in evidence_tokens:
|
||
return True
|
||
if re.fullmatch(r"[가-힣]+", token):
|
||
return any(
|
||
re.fullmatch(r"[가-힣]+", candidate) is not None
|
||
and (
|
||
(min(len(token), len(candidate)) >= 3 and (
|
||
token in candidate or candidate in token
|
||
))
|
||
or _korean_common_prefix(token, candidate) >= 2
|
||
)
|
||
for candidate in evidence_tokens
|
||
)
|
||
return False
|
||
|
||
|
||
def _low_lexical_support(
|
||
claim: DraftClaim, supporting_facts: Iterable[EvidenceItem]
|
||
) -> tuple[bool, list[str]]:
|
||
claim_tokens = _grounding_tokens(claim.text)
|
||
if len(claim_tokens) < 2:
|
||
return False, []
|
||
evidence_text = " ".join(
|
||
part
|
||
for fact in supporting_facts
|
||
for part in (
|
||
fact.content,
|
||
" ".join(fact.keywords),
|
||
" ".join(str(key) for key in fact.metrics),
|
||
" ".join(str(value) for value in fact.metrics.values()),
|
||
)
|
||
)
|
||
evidence_tokens = _grounding_tokens(evidence_text)
|
||
unsupported = sorted(
|
||
token
|
||
for token in claim_tokens
|
||
if not _token_is_supported(token, evidence_tokens)
|
||
)
|
||
# A mostly grounded sentence can still append one wholly invented clause
|
||
# (for example an award or leadership result). Ratios therefore create a
|
||
# dilution bypass: enough copied evidence hides two unsupported concepts.
|
||
# In strict-evidence mode two unsupported semantic tokens are sufficient to
|
||
# require repair, regardless of how much grounded text surrounds them.
|
||
return len(unsupported) >= 2, unsupported
|
||
|
||
|
||
def _looks_like_hidden_fact_echo(claim_text: str, fact_text: str) -> bool:
|
||
claim_normalised = _normalise_claim_text(claim_text)
|
||
fact_normalised = _normalise_claim_text(fact_text)
|
||
if min(len(claim_normalised), len(fact_normalised)) >= 12 and (
|
||
claim_normalised in fact_normalised or fact_normalised in claim_normalised
|
||
):
|
||
return True
|
||
claim_tokens = _grounding_tokens(claim_text)
|
||
fact_tokens = _grounding_tokens(fact_text)
|
||
smaller = min(len(claim_tokens), len(fact_tokens))
|
||
return smaller >= 3 and len(claim_tokens & fact_tokens) / smaller >= 0.7
|
||
|
||
|
||
def _has_match(patterns: Iterable[re.Pattern[str]], text: str) -> bool:
|
||
return any(pattern.search(text) is not None for pattern in patterns)
|
||
|
||
|
||
def contains_public_blind_origin(text: str) -> bool:
|
||
"""Return whether *text* explicitly discloses a person's place of origin.
|
||
|
||
This predicate is intentionally shared by the pre-generation evidence
|
||
boundary and the final deterministic draft gate. Keeping one rule avoids
|
||
a phrase being withheld from validation yet still crossing the LLM
|
||
boundary (or the reverse).
|
||
"""
|
||
|
||
return _has_match(_BLIND_ORIGIN_PATTERNS, text)
|
||
|
||
|
||
def _posting_constraint_patterns(
|
||
constraint: PostingConstraint,
|
||
profile: CandidateProfile,
|
||
) -> tuple[re.Pattern[str], ...]:
|
||
if constraint.kind not in {ConstraintKind.BLIND_FIELD, ConstraintKind.REDACTION}:
|
||
return ()
|
||
|
||
descriptor = unicodedata.normalize(
|
||
"NFKC",
|
||
" ".join(
|
||
[*constraint.fields, constraint.description, constraint.source_quote]
|
||
),
|
||
).casefold()
|
||
compact_descriptor = re.sub(r"\s+", "", descriptor)
|
||
patterns: list[re.Pattern[str]] = []
|
||
|
||
if any(keyword in compact_descriptor for keyword in ("학교", "출신대", "학력")):
|
||
patterns.extend(_BLIND_SCHOOL_PATTERNS)
|
||
if any(keyword in compact_descriptor for keyword in ("출신지", "출신지역", "고향")):
|
||
patterns.extend(_BLIND_ORIGIN_PATTERNS)
|
||
if any(keyword in compact_descriptor for keyword in ("성명", "지원자이름", "본인이름")):
|
||
patterns.append(_BLIND_IDENTITY_PATTERN)
|
||
if profile.name:
|
||
escaped_name = re.escape(profile.name)
|
||
patterns.append(
|
||
re.compile(
|
||
rf"(?:이름|성명)\s*[:\uff1a]\s*{escaped_name}|"
|
||
rf"저는\s*{escaped_name}(?:입니다|이라고)"
|
||
)
|
||
)
|
||
if any(keyword in compact_descriptor for keyword in ("나이", "연령")):
|
||
patterns.extend(_BLIND_AGE_PATTERNS)
|
||
if any(keyword in compact_descriptor for keyword in _EMPLOYER_FIELD_KEYWORDS):
|
||
patterns.extend(_EMPLOYER_NAME_PATTERNS)
|
||
organisations = {
|
||
record.organization
|
||
for record in (
|
||
*profile.records.careers,
|
||
*profile.records.experiences,
|
||
)
|
||
if record.organization
|
||
}
|
||
for organisation in sorted(organisations):
|
||
patterns.append(
|
||
re.compile(
|
||
rf"(?<![가-힣A-Za-z0-9]){re.escape(organisation)}"
|
||
rf"(?=에서|의|를|은|는|이|가|[\s,.)]|$)"
|
||
)
|
||
)
|
||
# Structured career records enumerate known employers, but free-form
|
||
# project evidence can mention a different client or company. Always
|
||
# retain the conservative proper-name heuristic; limiting it to
|
||
# profiles without records lets such unknown names bypass both the
|
||
# pre-LLM boundary and the final gate.
|
||
generic = "|".join(
|
||
re.escape(value) for value in _GENERIC_LOCATION_CONTEXT
|
||
)
|
||
patterns.append(
|
||
re.compile(
|
||
rf"(?<![가-힣A-Za-z0-9])(?!(?:{generic})에서)"
|
||
rf"[가-힣A-Za-z][가-힣A-Za-z0-9·]{{1,30}}(?=에서)",
|
||
re.IGNORECASE,
|
||
)
|
||
)
|
||
if any(keyword in compact_descriptor for keyword in ("이메일", "전자우편", "메일주소")):
|
||
patterns.append(_EMAIL_PATTERN)
|
||
if any(keyword in compact_descriptor for keyword in ("전화번호", "휴대폰", "연락처")):
|
||
patterns.append(_PHONE_PATTERN)
|
||
|
||
for category, keywords in _CONSTRAINT_CATEGORY_KEYWORDS:
|
||
if not any(keyword in compact_descriptor for keyword in keywords):
|
||
continue
|
||
for mapped_category, category_patterns in _SENSITIVE_PATTERNS:
|
||
if mapped_category is category:
|
||
patterns.extend(category_patterns)
|
||
break
|
||
|
||
# An institution may use a field name that is not in our semantic map. We
|
||
# only search for an explicit ``field: value`` label in that case, rather
|
||
# than treating arbitrary words from the posting as forbidden prose.
|
||
for field in constraint.fields:
|
||
stripped = field.strip()
|
||
if len(stripped) < 2:
|
||
continue
|
||
patterns.append(
|
||
re.compile(rf"{re.escape(stripped)}\s*[:\uff1a]\s*\S+")
|
||
)
|
||
|
||
return tuple(patterns)
|
||
|
||
|
||
def contains_blocking_posting_field(
|
||
text: str,
|
||
profile: CandidateProfile,
|
||
analysis: JobAnalysis,
|
||
) -> bool:
|
||
"""Use the final-output field rules at the pre-LLM boundary as well."""
|
||
|
||
return any(
|
||
constraint.blocking
|
||
and _has_match(_posting_constraint_patterns(constraint, profile), text)
|
||
for constraint in analysis.constraints
|
||
)
|
||
|
||
|
||
def _is_ascii_identifier_number(text: str, start: int, end: int) -> bool:
|
||
left = start
|
||
right = end
|
||
while left > 0 and text[left - 1] in _IDENTIFIER_CHARS:
|
||
left -= 1
|
||
while right < len(text) and text[right] in _IDENTIFIER_CHARS:
|
||
right += 1
|
||
token = text[left:right]
|
||
if _ASCII_NUMBER_WITH_UNIT_PATTERN.fullmatch(token):
|
||
return False
|
||
return any(character.isascii() and character.isalpha() for character in token)
|
||
|
||
|
||
def _normalise_unit(raw: str | None) -> str | None:
|
||
if raw is None:
|
||
return None
|
||
folded = raw.strip().casefold()
|
||
aliases = {
|
||
"퍼센트": "%",
|
||
"millisecond": "ms",
|
||
"milliseconds": "ms",
|
||
"second": "s",
|
||
"seconds": "s",
|
||
"sec": "s",
|
||
"secs": "s",
|
||
"kb": "KB",
|
||
"mb": "MB",
|
||
"gb": "GB",
|
||
"tb": "TB",
|
||
}
|
||
return aliases.get(folded, folded)
|
||
|
||
|
||
def _metric_unit(key: str) -> str | None:
|
||
folded = key.casefold()
|
||
if _RATIO_METRIC_KEY_PATTERN.search(folded):
|
||
return "%"
|
||
if re.search(
|
||
r"(?:team|member|headcount|people|user|customer).*count|team_size",
|
||
folded,
|
||
):
|
||
return "명"
|
||
if re.search(r"(?:^|_)ms(?:$|_)|latency_ms|duration_ms", folded):
|
||
return "ms"
|
||
if re.search(r"(?:seconds?|secs?|duration_s)(?:$|_)", folded):
|
||
return "s"
|
||
if re.search(
|
||
r"(?:request|error|order|case|event|issue|ticket).*count|(?:^|_)count$",
|
||
folded,
|
||
):
|
||
return "건"
|
||
if re.search(r"months?|month_count", folded):
|
||
return "개월"
|
||
if re.search(r"years?|year_count", folded):
|
||
return "년"
|
||
return None
|
||
|
||
|
||
def _numeric_tokens(text: str) -> list[_NumericToken]:
|
||
tokens: list[_NumericToken] = []
|
||
for match in _NUMBER_PATTERN.finditer(text):
|
||
if _is_ascii_identifier_number(text, match.start(), match.end()):
|
||
continue
|
||
raw = match.group(0)
|
||
try:
|
||
value = Decimal(raw.replace(",", "").lstrip("+"))
|
||
except InvalidOperation:
|
||
continue
|
||
unit_match = _NUMBER_UNIT_PATTERN.match(text, match.end())
|
||
unit = _normalise_unit(unit_match.group(1) if unit_match else None)
|
||
tokens.append(
|
||
_NumericToken(
|
||
value=value,
|
||
display=raw,
|
||
is_percent=unit == "%",
|
||
unit=unit,
|
||
)
|
||
)
|
||
return tokens
|
||
|
||
|
||
def _without_pii(text: str) -> str:
|
||
for pattern in (
|
||
_RESIDENT_ID_PATTERN,
|
||
_EMAIL_PATTERN,
|
||
_PHONE_PATTERN,
|
||
_BANK_ACCOUNT_PATTERN,
|
||
):
|
||
text = pattern.sub(" ", text)
|
||
return text
|
||
|
||
|
||
def _supported_numbers(
|
||
facts: Iterable[EvidenceItem],
|
||
) -> tuple[set[tuple[Decimal, str | None]], set[Decimal]]:
|
||
values: set[tuple[Decimal, str | None]] = set()
|
||
derived_percentages: set[Decimal] = set()
|
||
for fact in facts:
|
||
values.update(
|
||
(token.value, token.unit)
|
||
for token in _numeric_tokens(_without_pii(fact.content))
|
||
)
|
||
for key, metric_value in fact.metrics.items():
|
||
metric_tokens = _numeric_tokens(str(metric_value))
|
||
inferred_unit = _metric_unit(str(key))
|
||
values.update(
|
||
(token.value, token.unit or inferred_unit)
|
||
for token in metric_tokens
|
||
)
|
||
if _RATIO_METRIC_KEY_PATTERN.search(str(key)):
|
||
for token in _numeric_tokens(str(metric_value)):
|
||
if not token.is_percent and abs(token.value) <= 1:
|
||
derived_percentages.add(token.value * 100)
|
||
|
||
if fact.date_range is not None:
|
||
for resume_date in (fact.date_range.start, fact.date_range.end):
|
||
if resume_date is None:
|
||
continue
|
||
values.add((Decimal(resume_date.year), None))
|
||
values.add((Decimal(resume_date.year), "년"))
|
||
if resume_date.month is not None:
|
||
values.add((Decimal(resume_date.month), "월"))
|
||
values.add(
|
||
(
|
||
Decimal(f"{resume_date.year}.{resume_date.month:02d}"),
|
||
None,
|
||
)
|
||
)
|
||
return values, derived_percentages
|
||
|
||
|
||
def _reversed_date_ranges(text: str) -> list[str]:
|
||
reversed_ranges: list[str] = []
|
||
for match in _NUMERIC_DATE_RANGE_PATTERN.finditer(text):
|
||
start = (
|
||
int(match.group("sy")),
|
||
int(match.group("sm")),
|
||
int(match.group("sd") or 1),
|
||
)
|
||
end = (
|
||
int(match.group("ey")),
|
||
int(match.group("em")),
|
||
int(match.group("ed") or 1),
|
||
)
|
||
if start > end:
|
||
reversed_ranges.append(match.group(0))
|
||
return reversed_ranges
|
||
|
||
|
||
def _text_targets(draft: ResumeDraft) -> list[_TextTarget]:
|
||
targets = [_TextTarget(draft.title, "title")]
|
||
for section_index, section in enumerate(draft.sections):
|
||
section_location = f"sections[{section_index}]"
|
||
targets.append(_TextTarget(section.heading, f"{section_location}.heading"))
|
||
for claim_index, claim in enumerate(section.claims):
|
||
targets.append(
|
||
_TextTarget(
|
||
claim.text,
|
||
f"{section_location}.claims[{claim_index}].text",
|
||
claim,
|
||
)
|
||
)
|
||
return targets
|
||
|
||
|
||
def _effective_policy_mode(draft: ResumeDraft, config: GenerationConfig) -> ResumeMode:
|
||
# When mode declarations disagree, applying the stricter blind policy keeps a
|
||
# configuration error from becoming a privacy bypass.
|
||
if ResumeMode.PUBLIC_BLIND in {draft.mode, config.resume_mode}:
|
||
return ResumeMode.PUBLIC_BLIND
|
||
return config.resume_mode
|
||
|
||
|
||
def _validate_structured_resume_completeness(
|
||
profile: CandidateProfile,
|
||
draft: ResumeDraft,
|
||
collector: _FindingCollector,
|
||
) -> None:
|
||
"""Reject skeletal drafts when structured resume records are available.
|
||
|
||
Evidence grounding answers whether a sentence is supportable; it does not
|
||
answer whether the resulting resume is professionally complete. This
|
||
gate uses only typed records and section structure, so a judge cannot hide
|
||
a one-line career or project behind inflated subjective scores.
|
||
"""
|
||
|
||
records = profile.records
|
||
if not records.all_records():
|
||
# Legacy/unstructured intake cannot be assessed by this deterministic
|
||
# rule. Deployments seeking a release-grade result should materialise
|
||
# career, experience, education, and certification records first.
|
||
return
|
||
|
||
sections_by_type: dict[SectionType, list] = {}
|
||
for section in draft.sections:
|
||
sections_by_type.setdefault(section.section_type, []).append(section)
|
||
|
||
def add_missing(section_type: SectionType, label: str) -> None:
|
||
collector.add(
|
||
code=f"CONTENT.MISSING_{section_type.value.upper()}_SECTION",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.COMPLETENESS,
|
||
message=f"구조화된 후보자 기록에 필요한 {label} 섹션이 없습니다.",
|
||
location="sections",
|
||
suggestion=f"근거가 연결된 {label} 섹션을 추가하세요.",
|
||
)
|
||
|
||
if (records.careers or records.experiences) and not sections_by_type.get(
|
||
SectionType.SUMMARY
|
||
):
|
||
add_missing(SectionType.SUMMARY, "핵심 요약")
|
||
if records.careers and not sections_by_type.get(SectionType.EXPERIENCE):
|
||
add_missing(SectionType.EXPERIENCE, "경력")
|
||
if records.experiences and not sections_by_type.get(SectionType.PROJECTS):
|
||
add_missing(SectionType.PROJECTS, "프로젝트/직무 경험")
|
||
if records.educations and not sections_by_type.get(SectionType.EDUCATION):
|
||
add_missing(SectionType.EDUCATION, "교육 및 학력")
|
||
if records.certifications and not sections_by_type.get(
|
||
SectionType.CERTIFICATIONS
|
||
):
|
||
add_missing(SectionType.CERTIFICATIONS, "자격")
|
||
|
||
summary_claims = [
|
||
claim
|
||
for section in sections_by_type.get(SectionType.SUMMARY, [])
|
||
for claim in section.claims
|
||
]
|
||
if (records.careers or records.experiences) and len(summary_claims) < 2:
|
||
collector.add(
|
||
code="CONTENT.THIN_SUMMARY",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.COMPLETENESS,
|
||
message="핵심 요약이 후보자의 역할과 대표 성과를 각각 보여 주지 못합니다.",
|
||
location="sections.summary",
|
||
suggestion="서로 다른 근거를 사용한 역할/전문성 요약과 대표 성과 요약을 2개 이상 작성하세요.",
|
||
)
|
||
|
||
visible_keywords = {
|
||
keyword.casefold()
|
||
for fact in profile.facts
|
||
if not fact.confidential and fact.sensitive_category is None
|
||
for keyword in fact.keywords
|
||
if keyword.strip()
|
||
}
|
||
competency_sections = [
|
||
*sections_by_type.get(SectionType.CORE_COMPETENCIES, []),
|
||
*sections_by_type.get(SectionType.SKILLS, []),
|
||
]
|
||
if len(visible_keywords) >= 4 and not competency_sections:
|
||
collector.add(
|
||
code="CONTENT.MISSING_COMPETENCIES_SECTION",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.COMPLETENESS,
|
||
message="근거로 확인된 기술이 충분하지만 핵심 역량/기술 섹션이 없습니다.",
|
||
location="sections",
|
||
suggestion="검증된 기술을 직무 기준으로 묶은 핵심 역량 섹션을 추가하세요.",
|
||
)
|
||
elif competency_sections:
|
||
competency_claims = sum(
|
||
len(section.claims) for section in competency_sections
|
||
)
|
||
minimum_competencies = 2 if len(visible_keywords) < 8 else 3
|
||
if competency_claims < minimum_competencies:
|
||
collector.add(
|
||
code="CONTENT.THIN_COMPETENCIES",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.COMPLETENESS,
|
||
message="핵심 역량/기술 섹션이 확인된 기술 범위를 충분히 구조화하지 못했습니다.",
|
||
location="sections.core_competencies",
|
||
suggestion=f"서로 다른 역량 묶음을 최소 {minimum_competencies}개 제시하세요.",
|
||
)
|
||
|
||
def claims_for_record(record: object, section_type: SectionType) -> list[DraftClaim]:
|
||
evidence_ids = set(getattr(record, "evidence_ids", []))
|
||
return [
|
||
claim
|
||
for section in sections_by_type.get(section_type, [])
|
||
for claim in section.claims
|
||
if evidence_ids & set(claim.evidence_ids)
|
||
]
|
||
|
||
for record in records.careers:
|
||
actual = len(claims_for_record(record, SectionType.EXPERIENCE))
|
||
minimum = min(5, max(3, len(record.evidence_ids) + 1))
|
||
if actual < minimum:
|
||
collector.add(
|
||
code="CONTENT.THIN_CAREER_RECORD",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.COMPLETENESS,
|
||
message="경력 기록이 역할과 복수의 행동·성과를 판단할 만큼 상세하지 않습니다.",
|
||
location=f"records.careers.{record.record_id}",
|
||
evidence_ids=record.evidence_ids,
|
||
suggestion=f"해당 경력에 근거가 연결된 역할/성과 문장을 최소 {minimum}개 구성하세요.",
|
||
)
|
||
|
||
for record in records.experiences:
|
||
actual = len(claims_for_record(record, SectionType.PROJECTS))
|
||
minimum = min(4, max(2, len(record.evidence_ids) + 1))
|
||
if actual < minimum:
|
||
collector.add(
|
||
code="CONTENT.THIN_EXPERIENCE_RECORD",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.COMPLETENESS,
|
||
message="프로젝트/직무 경험 기록이 역할, 구현 내용, 결과를 판단할 만큼 상세하지 않습니다.",
|
||
location=f"records.experiences.{record.record_id}",
|
||
evidence_ids=record.evidence_ids,
|
||
suggestion=f"해당 경험에 근거가 연결된 문장을 최소 {minimum}개 구성하세요.",
|
||
)
|
||
|
||
for record in records.educations:
|
||
if not claims_for_record(record, SectionType.EDUCATION):
|
||
collector.add(
|
||
code="CONTENT.UNMATERIALIZED_EDUCATION_RECORD",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.COMPLETENESS,
|
||
message="구조화된 교육/학력 기록이 초안에 반영되지 않았습니다.",
|
||
location=f"records.educations.{record.record_id}",
|
||
evidence_ids=record.evidence_ids,
|
||
suggestion="학교 정책을 적용한 뒤 전공·학위·직무 관련 교육을 근거와 함께 반영하세요.",
|
||
)
|
||
|
||
for record in records.certifications:
|
||
if not claims_for_record(record, SectionType.CERTIFICATIONS):
|
||
collector.add(
|
||
code="CONTENT.UNMATERIALIZED_CERTIFICATION_RECORD",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.COMPLETENESS,
|
||
message="구조화된 자격 기록이 초안에 반영되지 않았습니다.",
|
||
location=f"records.certifications.{record.record_id}",
|
||
evidence_ids=record.evidence_ids,
|
||
suggestion="직무 관련 자격명·발급기관·취득일을 근거와 함께 반영하세요.",
|
||
)
|
||
|
||
|
||
def validate_resume_draft(
|
||
profile: CandidateProfile,
|
||
draft: ResumeDraft,
|
||
config: GenerationConfig,
|
||
*,
|
||
analysis: JobAnalysis | None = None,
|
||
) -> list[QualityFinding]:
|
||
"""Return stable, deterministic findings for a typed resume draft.
|
||
|
||
The function does not raise for cross-model inconsistencies. This is
|
||
intentional: findings are repair-loop input, whereas Pydantic validation is
|
||
responsible for rejecting malformed individual objects at the intake edge.
|
||
"""
|
||
|
||
collector = _FindingCollector()
|
||
policy_mode = _effective_policy_mode(draft, config)
|
||
_validate_structured_resume_completeness(profile, draft, collector)
|
||
|
||
evidence_by_id: dict[str, EvidenceItem] = {}
|
||
duplicate_evidence_ids: list[str] = []
|
||
for fact in profile.facts:
|
||
if fact.evidence_id in evidence_by_id:
|
||
if fact.evidence_id not in duplicate_evidence_ids:
|
||
duplicate_evidence_ids.append(fact.evidence_id)
|
||
else:
|
||
evidence_by_id[fact.evidence_id] = fact
|
||
if duplicate_evidence_ids:
|
||
collector.add(
|
||
code="REFERENCE.DUPLICATE_EVIDENCE_ID",
|
||
severity=QualitySeverity.CRITICAL,
|
||
category=QualityCategory.EVIDENCE,
|
||
message="후보자 사실 원장에 중복 evidence_id가 있습니다.",
|
||
location="profile.facts",
|
||
evidence_ids=duplicate_evidence_ids,
|
||
suggestion="각 근거에 전역적으로 고유한 evidence_id를 부여하세요.",
|
||
)
|
||
hidden_facts = [
|
||
fact
|
||
for fact in profile.facts
|
||
if fact.confidential or fact.sensitive_category is not None
|
||
]
|
||
|
||
if draft.candidate_id != profile.candidate_id:
|
||
collector.add(
|
||
code="REFERENCE.CANDIDATE_MISMATCH",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.EVIDENCE,
|
||
message="초안의 candidate_id가 후보자 프로필과 일치하지 않습니다.",
|
||
location="candidate_id",
|
||
suggestion="동일 후보자의 프로필로 초안을 다시 생성하세요.",
|
||
)
|
||
|
||
if analysis is not None and draft.posting_id != analysis.posting_id:
|
||
collector.add(
|
||
code="REFERENCE.POSTING_MISMATCH",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.JOB_ALIGNMENT,
|
||
message="초안의 posting_id가 공고 분석과 일치하지 않습니다.",
|
||
location="posting_id",
|
||
suggestion="해당 공고에서 생성한 초안과 분석을 함께 사용하세요.",
|
||
)
|
||
|
||
if draft.mode != config.resume_mode:
|
||
collector.add(
|
||
code="CONFIG.MODE_MISMATCH",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.CONSISTENCY,
|
||
message="초안 모드와 생성 설정의 이력서 모드가 일치하지 않습니다.",
|
||
location="mode",
|
||
suggestion="한 정책 모드로 다시 생성하거나 설정을 일치시키세요.",
|
||
)
|
||
|
||
if config.include_photo and SensitiveDataCategory.PHOTO not in (
|
||
config.allowed_sensitive_categories
|
||
):
|
||
collector.add(
|
||
code="CONFIG.PHOTO_PERMISSION",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.PRIVACY,
|
||
message="사진 포함 설정에 필요한 민감정보 허용 범주가 없습니다.",
|
||
location="config.include_photo",
|
||
suggestion="사진을 제외하거나 명시적 동의가 연결된 사진 범주를 허용하세요.",
|
||
)
|
||
|
||
prohibited_config_categories = config.allowed_sensitive_categories & {
|
||
SensitiveDataCategory.NATIONAL_ID,
|
||
SensitiveDataCategory.BANK_ACCOUNT,
|
||
SensitiveDataCategory.HEALTH,
|
||
}
|
||
if prohibited_config_categories:
|
||
collector.add(
|
||
code="CONFIG.PROHIBITED_SENSITIVE_CATEGORY",
|
||
severity=QualitySeverity.CRITICAL,
|
||
category=QualityCategory.PRIVACY,
|
||
message="절대 금지된 민감정보 범주가 생성 설정에 포함되어 있습니다.",
|
||
location="config.allowed_sensitive_categories",
|
||
suggestion="건강정보, 주민등록번호, 계좌정보 허용을 제거하세요.",
|
||
)
|
||
|
||
unrequested_sensitive = config.allowed_sensitive_categories - getattr(
|
||
config, "employer_required_sensitive_categories", set()
|
||
)
|
||
if unrequested_sensitive:
|
||
collector.add(
|
||
code="CONFIG.SENSITIVE_NOT_EMPLOYER_REQUIRED",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.PRIVACY,
|
||
message="채용사 요구 근거가 없는 민감정보 범주가 활성화되어 있습니다.",
|
||
location="config.allowed_sensitive_categories",
|
||
suggestion="채용사 지정 요구를 기록하거나 해당 민감정보를 제외하세요.",
|
||
)
|
||
|
||
if policy_mode is ResumeMode.PUBLIC_BLIND and (
|
||
config.include_photo
|
||
or config.allowed_sensitive_categories
|
||
or getattr(config, "employer_required_sensitive_categories", set())
|
||
):
|
||
collector.add(
|
||
code="CONFIG.BLIND_SENSITIVE_ENABLED",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.PRIVACY,
|
||
message="공공 블라인드 모드에서 민감정보가 활성화되어 있습니다.",
|
||
location="config.allowed_sensitive_categories",
|
||
suggestion="사진과 모든 민감정보 허용 범주를 비활성화하세요.",
|
||
)
|
||
|
||
consent_instant = datetime.combine(
|
||
config.as_of_date, datetime.min.time(), tzinfo=timezone.utc
|
||
)
|
||
active_categories = {
|
||
consent.category
|
||
for consent in profile.consents
|
||
if consent.is_active_at(consent_instant)
|
||
}
|
||
missing_consent = config.allowed_sensitive_categories - active_categories
|
||
if missing_consent:
|
||
missing_labels = ", ".join(
|
||
sorted(category.value for category in missing_consent)
|
||
)
|
||
collector.add(
|
||
code="CONFIG.SENSITIVE_CONSENT",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.PRIVACY,
|
||
message=f"활성 동의가 없는 민감정보 범주가 있습니다: {missing_labels}.",
|
||
location="config.allowed_sensitive_categories",
|
||
suggestion="유효한 목적별 동의를 연결하거나 해당 범주를 제외하세요.",
|
||
)
|
||
|
||
claim_locations: dict[str, str] = {}
|
||
claims: list[DraftClaim] = []
|
||
known_requirement_ids = (
|
||
{requirement.requirement_id for requirement in analysis.requirements}
|
||
if analysis is not None
|
||
else None
|
||
)
|
||
for section_index, section in enumerate(draft.sections):
|
||
if (
|
||
policy_mode is ResumeMode.PUBLIC_BLIND
|
||
and section.section_type is SectionType.MILITARY_SERVICE
|
||
):
|
||
collector.add(
|
||
code="PRIVACY.BLIND_MILITARY_SECTION",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.PRIVACY,
|
||
message="공공 블라인드 본문에 병역 상세 섹션이 포함되어 있습니다.",
|
||
location=f"sections[{section_index}]",
|
||
suggestion="병역 상세 섹션을 제거하세요.",
|
||
)
|
||
|
||
for claim_index, claim in enumerate(section.claims):
|
||
location = f"sections[{section_index}].claims[{claim_index}]"
|
||
claims.append(claim)
|
||
claim_locations.setdefault(claim.claim_id, location)
|
||
|
||
if not claim.evidence_ids:
|
||
collector.add(
|
||
code="GROUNDING.MISSING_EVIDENCE",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.EVIDENCE,
|
||
message="claim에 연결된 근거가 없습니다.",
|
||
location=location,
|
||
claim_id=claim.claim_id,
|
||
suggestion="실제 후보자 근거를 연결하거나 claim을 제거하세요.",
|
||
)
|
||
|
||
unknown_ids = [
|
||
evidence_id
|
||
for evidence_id in claim.evidence_ids
|
||
if evidence_id not in evidence_by_id
|
||
]
|
||
if unknown_ids:
|
||
collector.add(
|
||
code="REFERENCE.UNKNOWN_EVIDENCE",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.EVIDENCE,
|
||
message="claim이 사실 원장에 없는 evidence_id를 참조합니다.",
|
||
location=f"{location}.evidence_ids",
|
||
claim_id=claim.claim_id,
|
||
evidence_ids=unknown_ids,
|
||
suggestion="존재하는 근거 ID로 교체하거나 claim을 제거하세요.",
|
||
)
|
||
|
||
if known_requirement_ids is not None:
|
||
unknown_requirement_ids = [
|
||
requirement_id
|
||
for requirement_id in claim.requirement_ids
|
||
if requirement_id not in known_requirement_ids
|
||
]
|
||
if unknown_requirement_ids:
|
||
collector.add(
|
||
code="REFERENCE.UNKNOWN_REQUIREMENT",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.JOB_ALIGNMENT,
|
||
message="claim이 공고 분석에 없는 requirement_id를 참조합니다.",
|
||
location=f"{location}.requirement_ids",
|
||
claim_id=claim.claim_id,
|
||
suggestion="공고 분석에 존재하는 요구사항 ID만 연결하세요.",
|
||
)
|
||
|
||
supporting_facts = [
|
||
evidence_by_id[evidence_id]
|
||
for evidence_id in claim.evidence_ids
|
||
if evidence_id in evidence_by_id
|
||
]
|
||
if any(
|
||
hidden.evidence_id not in claim.evidence_ids
|
||
and _looks_like_hidden_fact_echo(claim.text, hidden.content)
|
||
for hidden in hidden_facts
|
||
):
|
||
collector.add(
|
||
code="PRIVACY.HIDDEN_EVIDENCE_ECHO",
|
||
severity=QualitySeverity.CRITICAL,
|
||
category=QualityCategory.PRIVACY,
|
||
message=(
|
||
"claim이 공개 허용 근거를 참조하면서 비공개 또는 민감 "
|
||
"원장의 문구를 재현합니다."
|
||
),
|
||
location=f"{location}.text",
|
||
claim_id=claim.claim_id,
|
||
evidence_ids=claim.evidence_ids,
|
||
suggestion="해당 문구를 제거하고 공개 허용 근거만으로 다시 작성하세요.",
|
||
)
|
||
confidential_ids = [
|
||
fact.evidence_id for fact in supporting_facts if fact.confidential
|
||
]
|
||
if confidential_ids:
|
||
collector.add(
|
||
code="GROUNDING.CONFIDENTIAL_EVIDENCE",
|
||
severity=QualitySeverity.CRITICAL,
|
||
category=QualityCategory.PRIVACY,
|
||
message="claim이 외부 공개가 금지된 기밀 근거를 참조합니다.",
|
||
location=f"{location}.evidence_ids",
|
||
claim_id=claim.claim_id,
|
||
evidence_ids=confidential_ids,
|
||
suggestion="기밀 근거의 참조와 그로부터 파생된 문구를 모두 제거하세요.",
|
||
)
|
||
|
||
referenced_sensitive = {
|
||
fact.sensitive_category
|
||
for fact in supporting_facts
|
||
if fact.sensitive_category is not None
|
||
}
|
||
undeclared_sensitive = referenced_sensitive - claim.sensitive_categories
|
||
if undeclared_sensitive:
|
||
labels = ", ".join(
|
||
sorted(category.value for category in undeclared_sensitive)
|
||
)
|
||
collector.add(
|
||
code="REFERENCE.UNDECLARED_SENSITIVE_EVIDENCE",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.PRIVACY,
|
||
message=f"claim이 표시하지 않은 민감 근거 범주를 참조합니다: {labels}.",
|
||
location=f"{location}.evidence_ids",
|
||
claim_id=claim.claim_id,
|
||
evidence_ids=claim.evidence_ids,
|
||
suggestion="민감 근거를 제거하거나 허용·동의된 범주를 명시하세요.",
|
||
)
|
||
|
||
disallowed_referenced = (
|
||
referenced_sensitive
|
||
if policy_mode is ResumeMode.PUBLIC_BLIND
|
||
else referenced_sensitive - config.allowed_sensitive_categories
|
||
)
|
||
if disallowed_referenced:
|
||
labels = ", ".join(
|
||
sorted(category.value for category in disallowed_referenced)
|
||
)
|
||
collector.add(
|
||
code="PRIVACY.DISALLOWED_SENSITIVE_EVIDENCE",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.PRIVACY,
|
||
message=f"현재 모드에서 허용되지 않은 민감 근거를 참조합니다: {labels}.",
|
||
location=f"{location}.evidence_ids",
|
||
claim_id=claim.claim_id,
|
||
evidence_ids=claim.evidence_ids,
|
||
suggestion="해당 민감 근거와 그로부터 파생된 문구를 제거하세요.",
|
||
)
|
||
unsupported_sensitive = {
|
||
category
|
||
for category in claim.sensitive_categories
|
||
if not any(
|
||
fact.sensitive_category == category for fact in supporting_facts
|
||
)
|
||
}
|
||
if unsupported_sensitive:
|
||
labels = ", ".join(
|
||
sorted(category.value for category in unsupported_sensitive)
|
||
)
|
||
collector.add(
|
||
code="REFERENCE.UNSUPPORTED_SENSITIVE_CATEGORY",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.EVIDENCE,
|
||
message=f"근거가 뒷받침하지 않는 민감정보 범주가 표시되었습니다: {labels}.",
|
||
location=f"{location}.sensitive_categories",
|
||
claim_id=claim.claim_id,
|
||
evidence_ids=claim.evidence_ids,
|
||
suggestion="근거와 동의가 모두 있는 범주만 표시하세요.",
|
||
)
|
||
|
||
disallowed_declared = (
|
||
set(claim.sensitive_categories)
|
||
if policy_mode is ResumeMode.PUBLIC_BLIND
|
||
else claim.sensitive_categories - config.allowed_sensitive_categories
|
||
)
|
||
if disallowed_declared:
|
||
labels = ", ".join(
|
||
sorted(category.value for category in disallowed_declared)
|
||
)
|
||
code = (
|
||
"PRIVACY.BLIND_SENSITIVE_CATEGORY"
|
||
if policy_mode is ResumeMode.PUBLIC_BLIND
|
||
else "PRIVACY.DISALLOWED_SENSITIVE_CATEGORY"
|
||
)
|
||
collector.add(
|
||
code=code,
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.PRIVACY,
|
||
message=f"현재 모드에서 허용되지 않는 민감정보 범주입니다: {labels}.",
|
||
location=f"{location}.sensitive_categories",
|
||
claim_id=claim.claim_id,
|
||
evidence_ids=claim.evidence_ids,
|
||
suggestion="민감정보를 제거하거나 적법한 동의와 모드 정책을 확인하세요.",
|
||
)
|
||
|
||
targets = _text_targets(draft)
|
||
for target in targets:
|
||
if _has_match(_PLACEHOLDER_PATTERNS, target.text):
|
||
collector.add(
|
||
code="CONTENT.PLACEHOLDER",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.COMPLETENESS,
|
||
message="최종 문서에 편집용 placeholder가 남아 있습니다.",
|
||
location=target.location,
|
||
claim_id=target.claim_id,
|
||
evidence_ids=target.evidence_ids,
|
||
suggestion="확인된 내용으로 교체하거나 해당 문구를 제거하세요.",
|
||
)
|
||
|
||
if _RESIDENT_ID_PATTERN.search(target.text):
|
||
collector.add(
|
||
code="PRIVACY.RESIDENT_ID",
|
||
severity=QualitySeverity.CRITICAL,
|
||
category=QualityCategory.PRIVACY,
|
||
message="본문에 주민등록번호 형식의 값이 포함되어 있습니다.",
|
||
location=target.location,
|
||
claim_id=target.claim_id,
|
||
evidence_ids=target.evidence_ids,
|
||
suggestion="해당 값을 즉시 삭제하고 원본 및 로그의 잔존 여부도 확인하세요.",
|
||
)
|
||
if _EMAIL_PATTERN.search(target.text):
|
||
collector.add(
|
||
code="PRIVACY.EMAIL_IN_BODY",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.PRIVACY,
|
||
message="본문에 이메일 주소가 포함되어 있습니다.",
|
||
location=target.location,
|
||
claim_id=target.claim_id,
|
||
evidence_ids=target.evidence_ids,
|
||
suggestion="이메일은 본문에서 제거하고 렌더러의 신원 블록에만 삽입하세요.",
|
||
)
|
||
if _PHONE_PATTERN.search(target.text):
|
||
collector.add(
|
||
code="PRIVACY.PHONE_IN_BODY",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.PRIVACY,
|
||
message="본문에 전화번호 형식의 값이 포함되어 있습니다.",
|
||
location=target.location,
|
||
claim_id=target.claim_id,
|
||
evidence_ids=target.evidence_ids,
|
||
suggestion="전화번호는 본문에서 제거하고 렌더러의 신원 블록에만 삽입하세요.",
|
||
)
|
||
|
||
detected_categories = {
|
||
category
|
||
for category, patterns in _SENSITIVE_PATTERNS
|
||
if _has_match(patterns, target.text)
|
||
}
|
||
for category in sorted(detected_categories, key=lambda item: item.value):
|
||
declared = (
|
||
target.claim is not None
|
||
and category in target.claim.sensitive_categories
|
||
)
|
||
if category is SensitiveDataCategory.BANK_ACCOUNT:
|
||
# Account information is prohibited in every mode, even with consent.
|
||
collector.add(
|
||
code="PRIVACY.BANK_ACCOUNT",
|
||
severity=QualitySeverity.CRITICAL,
|
||
category=QualityCategory.PRIVACY,
|
||
message="본문에 계좌정보로 보이는 값이 포함되어 있습니다.",
|
||
location=target.location,
|
||
claim_id=target.claim_id,
|
||
evidence_ids=target.evidence_ids,
|
||
suggestion="계좌정보를 즉시 삭제하고 원본 및 로그의 잔존 여부도 확인하세요.",
|
||
)
|
||
elif policy_mode is ResumeMode.PUBLIC_BLIND:
|
||
if declared:
|
||
# The declared-category finding above already explains the same
|
||
# policy breach and is a better repair target.
|
||
continue
|
||
collector.add(
|
||
code="PRIVACY.BLIND_SENSITIVE_CONTENT",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.PRIVACY,
|
||
message=(
|
||
"공공 블라인드 본문에서 편견을 유발할 수 있는 "
|
||
f"민감정보 표현이 탐지되었습니다: {category.value}."
|
||
),
|
||
location=target.location,
|
||
claim_id=target.claim_id,
|
||
evidence_ids=target.evidence_ids,
|
||
suggestion="직무 수행 근거만 남기고 해당 개인정보 표현을 제거하세요.",
|
||
)
|
||
elif category not in config.allowed_sensitive_categories:
|
||
if declared:
|
||
continue
|
||
collector.add(
|
||
code="PRIVACY.DISALLOWED_SENSITIVE_CONTENT",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.PRIVACY,
|
||
message=(
|
||
"현재 모드에서 허용되지 않은 민감정보 표현이 "
|
||
f"탐지되었습니다: {category.value}."
|
||
),
|
||
location=target.location,
|
||
claim_id=target.claim_id,
|
||
evidence_ids=target.evidence_ids,
|
||
suggestion="해당 개인정보 표현을 제거하세요.",
|
||
)
|
||
elif target.claim is not None and not declared:
|
||
collector.add(
|
||
code="PRIVACY.UNDECLARED_SENSITIVE_CONTENT",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.PRIVACY,
|
||
message=(
|
||
"허용된 민감정보가 claim 메타데이터에 표시되지 "
|
||
f"않았습니다: {category.value}."
|
||
),
|
||
location=target.location,
|
||
claim_id=target.claim_id,
|
||
evidence_ids=target.evidence_ids,
|
||
suggestion="민감정보 범주와 이를 뒷받침하는 동의 근거를 명시하세요.",
|
||
)
|
||
|
||
if policy_mode is ResumeMode.PUBLIC_BLIND:
|
||
if _has_match(_BLIND_SCHOOL_PATTERNS, target.text):
|
||
collector.add(
|
||
code="PRIVACY.BLIND_SCHOOL",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.BIAS,
|
||
message="공공 블라인드 본문에 학교를 식별할 수 있는 표현이 있습니다.",
|
||
location=target.location,
|
||
claim_id=target.claim_id,
|
||
evidence_ids=target.evidence_ids,
|
||
suggestion="학교명은 제거하고 직무 관련 교육 내용만 남기세요.",
|
||
)
|
||
if contains_public_blind_origin(target.text):
|
||
collector.add(
|
||
code="PRIVACY.BLIND_ORIGIN",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.BIAS,
|
||
message="공공 블라인드 본문에 출신지를 드러내는 표현이 있습니다.",
|
||
location=target.location,
|
||
claim_id=target.claim_id,
|
||
evidence_ids=target.evidence_ids,
|
||
suggestion="출신지 표현을 제거하세요.",
|
||
)
|
||
if _has_match(_BLIND_AGE_PATTERNS, target.text):
|
||
collector.add(
|
||
code="PRIVACY.BLIND_AGE",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.BIAS,
|
||
message="공공 블라인드 본문에 연령을 드러내는 표현이 있습니다.",
|
||
location=target.location,
|
||
claim_id=target.claim_id,
|
||
evidence_ids=target.evidence_ids,
|
||
suggestion="연령 표현을 제거하세요.",
|
||
)
|
||
|
||
identity_found = _has_match(
|
||
_candidate_identity_patterns(profile), target.text
|
||
)
|
||
if identity_found:
|
||
collector.add(
|
||
code="PRIVACY.BLIND_IDENTITY",
|
||
severity=QualitySeverity.ERROR,
|
||
category=QualityCategory.BIAS,
|
||
message="공공 블라인드 본문에 지원자 이름을 드러내는 표현이 있습니다.",
|
||
location=target.location,
|
||
claim_id=target.claim_id,
|
||
evidence_ids=target.evidence_ids,
|
||
suggestion="이름은 심사용 본문에서 제거하고 본인확인 영역과 분리하세요.",
|
||
)
|
||
|
||
if analysis is not None:
|
||
for constraint in analysis.constraints:
|
||
patterns = _posting_constraint_patterns(constraint, profile)
|
||
if not patterns:
|
||
continue
|
||
for target in targets:
|
||
if not _has_match(patterns, target.text):
|
||
continue
|
||
field_labels = ", ".join(constraint.fields)
|
||
collector.add(
|
||
code="PRIVACY.POSTING_FIELD_LEAK",
|
||
severity=(
|
||
QualitySeverity.ERROR
|
||
if constraint.blocking
|
||
else QualitySeverity.WARNING
|
||
),
|
||
category=QualityCategory.BIAS,
|
||
message=(
|
||
"공고별 블라인드/삭제 제약에 지정된 필드가 본문에 "
|
||
f"노출되었습니다 ({constraint.constraint_id}: {field_labels})."
|
||
),
|
||
location=target.location,
|
||
claim_id=target.claim_id,
|
||
evidence_ids=target.evidence_ids,
|
||
suggestion="공고 원문의 해당 필드 규칙에 맞게 표현을 삭제하거나 비식별화하세요.",
|
||
)
|
||
|
||
first_claim_by_text: dict[str, DraftClaim] = {}
|
||
for claim in claims:
|
||
normalised = _normalise_claim_text(claim.text)
|
||
previous = first_claim_by_text.get(normalised)
|
||
if previous is None:
|
||
first_claim_by_text[normalised] = claim
|
||
continue
|
||
collector.add(
|
||
code="CONTENT.DUPLICATE_CLAIM",
|
||
severity=QualitySeverity.WARNING,
|
||
category=QualityCategory.CONSISTENCY,
|
||
message=f"동일한 claim 문구가 앞선 claim {previous.claim_id!r}과 중복됩니다.",
|
||
location=claim_locations.get(claim.claim_id),
|
||
claim_id=claim.claim_id,
|
||
evidence_ids=claim.evidence_ids,
|
||
suggestion="중복 문구를 제거하거나 서로 다른 근거와 기여를 명확히 구분하세요.",
|
||
)
|
||
|
||
number_severity = (
|
||
QualitySeverity.ERROR if config.strict_evidence else QualitySeverity.WARNING
|
||
)
|
||
requirement_by_id = (
|
||
{item.requirement_id: item for item in analysis.requirements}
|
||
if analysis is not None
|
||
else {}
|
||
)
|
||
for claim in claims:
|
||
for requirement_id in claim.requirement_ids:
|
||
requirement = requirement_by_id.get(requirement_id)
|
||
if requirement is not None and not _claim_mentions_requirement(
|
||
claim.text, requirement
|
||
):
|
||
collector.add(
|
||
code="ALIGNMENT.REQUIREMENT_MISMATCH",
|
||
severity=number_severity,
|
||
category=QualityCategory.JOB_ALIGNMENT,
|
||
message=(
|
||
"claim 문구에 연결된 직무 요건의 핵심 표현이 "
|
||
"확인되지 않습니다."
|
||
),
|
||
location=f"{claim_locations.get(claim.claim_id, 'claims')}.text",
|
||
claim_id=claim.claim_id,
|
||
evidence_ids=claim.evidence_ids,
|
||
suggestion=(
|
||
"요건과 직접 맞닿는 표현을 근거 범위 안에서 명시하거나 "
|
||
"잘못된 requirement ID 연결을 제거하세요."
|
||
),
|
||
)
|
||
supporting_facts = [
|
||
evidence_by_id[evidence_id]
|
||
for evidence_id in claim.evidence_ids
|
||
if evidence_id in evidence_by_id
|
||
]
|
||
if not supporting_facts:
|
||
# Missing/unknown evidence has already produced the primary repair
|
||
# finding; reporting every number as well would be redundant noise.
|
||
continue
|
||
reversed_ranges = _reversed_date_ranges(claim.text)
|
||
if reversed_ranges:
|
||
collector.add(
|
||
code="CHRONOLOGY.REVERSED_RANGE",
|
||
severity=number_severity,
|
||
category=QualityCategory.CHRONOLOGY,
|
||
message="claim의 시작일이 종료일보다 늦습니다.",
|
||
location=f"{claim_locations.get(claim.claim_id, 'claims')}.text",
|
||
claim_id=claim.claim_id,
|
||
evidence_ids=claim.evidence_ids,
|
||
suggestion="근거의 날짜 범위와 대조해 시작·종료 순서를 바로잡으세요.",
|
||
)
|
||
supported, derived_percentages = _supported_numbers(supporting_facts)
|
||
unsupported_displays: list[str] = []
|
||
seen_values: set[tuple[Decimal, str | None]] = set()
|
||
for token in _numeric_tokens(_without_pii(claim.text)):
|
||
key = (token.value, token.unit)
|
||
if key in seen_values:
|
||
continue
|
||
seen_values.add(key)
|
||
if (token.value, token.unit) in supported:
|
||
continue
|
||
if token.is_percent and token.value in derived_percentages:
|
||
continue
|
||
unsupported_displays.append(token.display)
|
||
if unsupported_displays:
|
||
values = ", ".join(unsupported_displays)
|
||
collector.add(
|
||
code="GROUNDING.UNSUPPORTED_NUMBER",
|
||
severity=number_severity,
|
||
category=QualityCategory.EVIDENCE,
|
||
message=f"claim의 숫자가 연결 근거의 content 또는 metrics에 없습니다: {values}.",
|
||
location=f"{claim_locations.get(claim.claim_id, 'claims')}.text",
|
||
claim_id=claim.claim_id,
|
||
evidence_ids=claim.evidence_ids,
|
||
suggestion="근거에 있는 정확한 숫자로 교체하거나 숫자 표현을 제거하세요.",
|
||
)
|
||
|
||
unsupported_terms = _unsupported_technical_terms(claim, supporting_facts)
|
||
if unsupported_terms:
|
||
collector.add(
|
||
code="GROUNDING.UNSUPPORTED_TECH_TERM",
|
||
severity=number_severity,
|
||
category=QualityCategory.EVIDENCE,
|
||
message=(
|
||
"claim의 기술 용어가 연결 근거의 content, keywords 또는 "
|
||
"metrics에 없습니다: " + ", ".join(unsupported_terms) + "."
|
||
),
|
||
location=f"{claim_locations.get(claim.claim_id, 'claims')}.text",
|
||
claim_id=claim.claim_id,
|
||
evidence_ids=claim.evidence_ids,
|
||
suggestion="근거에 있는 기술 용어로 교체하거나 해당 표현을 제거하세요.",
|
||
)
|
||
|
||
low_support, unsupported_words = _low_lexical_support(
|
||
claim, supporting_facts
|
||
)
|
||
if low_support:
|
||
preview = ", ".join(unsupported_words[:8])
|
||
collector.add(
|
||
code="GROUNDING.LOW_LEXICAL_SUPPORT",
|
||
severity=number_severity,
|
||
category=QualityCategory.EVIDENCE,
|
||
message=(
|
||
"claim의 핵심 표현 다수가 연결 근거에서 확인되지 않습니다: "
|
||
f"{preview}."
|
||
),
|
||
location=f"{claim_locations.get(claim.claim_id, 'claims')}.text",
|
||
claim_id=claim.claim_id,
|
||
evidence_ids=claim.evidence_ids,
|
||
suggestion=(
|
||
"연결 근거에 명시된 맥락·행동·결과만 사용하거나 추가 "
|
||
"근거를 제공하세요."
|
||
),
|
||
)
|
||
|
||
return collector.findings
|
||
|
||
|
||
# Short compatibility name for callers that already operate on ResumeDraft.
|
||
validate_draft = validate_resume_draft
|
||
|
||
|
||
__all__ = [
|
||
"contains_blocking_posting_field",
|
||
"contains_public_blind_origin",
|
||
"validate_draft",
|
||
"validate_resume_draft",
|
||
]
|