generated from kgod/ai-review-template
自内部仓库剥离深度优化与 RAG 知识库后的交付版本: - Builder 对话式简历生成(FSM + 意图路由 LLM 兜底增强) - 条目级轻度优化:事实覆盖门禁 + STAR/bullet 修复链,功能/简介/成果与技术栈同级保护 - 简历导入:DOCX/PDF 解析、结构归一、手机号脱敏 - PostgreSQL 运行时 + Alembic 迁移链 Co-Authored-By: Claude <noreply@anthropic.com>
183 lines
8.0 KiB
Python
183 lines
8.0 KiB
Python
"""Validation and safety partitioning for resume optimization proposals."""
|
||
|
||
from __future__ import annotations
|
||
|
||
import re
|
||
from typing import Any
|
||
|
||
from .experience_optimizer import normalize_fact_ledger
|
||
from .text_normalization import decode_literal_unicode_escapes
|
||
|
||
_NUMBER = re.compile(r"\d+(?:\.\d+)?%?")
|
||
_LATIN_TERM = re.compile(r"[A-Za-z][A-Za-z0-9.+#_-]{1,}")
|
||
_COMMON_TECH_TERMS = frozenset({
|
||
"aws", "azure", "docker", "elasticsearch", "fastapi", "flask", "git", "go",
|
||
"java", "javascript", "kafka", "kubernetes", "langchain", "langgraph", "linux",
|
||
"mongodb", "mysql", "nextjs", "nodejs", "numpy", "openai", "pandas", "postgresql",
|
||
"python", "pytorch", "rabbitmq", "react", "redis", "spring", "sql", "tensorflow",
|
||
"typescript", "vue", "vue3",
|
||
})
|
||
_SENTENCE = re.compile(r"(?<=[。!?!?;;])\s*|\n+")
|
||
_COUNTED_OBJECT = re.compile(
|
||
r"(?P<number>\d+(?:\.\d+)?)(?:\s*)(?P<unit>名|位|人|项|个|次|台|条|份|家|天|月|年|students?|classmates?|users?|features?|services?|projects?|requests?)(?:\s*)(?P<object>[A-Za-z][A-Za-z -]{0,24}|[\u4e00-\u9fff]{0,8})",
|
||
re.I,
|
||
)
|
||
|
||
|
||
def validate_proposal(proposal: dict[str, Any], facts: list[Any]) -> dict[str, Any]:
|
||
"""Normalize proposal metadata without suppressing useful model-written prose.
|
||
|
||
The fact ledger validates claim references and aids diagnostics. It is not a
|
||
word-for-word acceptance gate for optimized prose: resume editing needs
|
||
paraphrase, synthesis, and controlled role-oriented expansion.
|
||
"""
|
||
result = decode_literal_unicode_escapes(dict(proposal))
|
||
ledger = normalize_fact_ledger(facts)
|
||
known_ids = {item["id"] for item in ledger}
|
||
evidence = "\n".join(item["text"] for item in ledger)
|
||
warnings = [str(item) for item in result.get("validation_warnings") or [] if str(item)]
|
||
suggestions = [str(item).strip() for item in result.get("unconfirmed_suggestions") or [] if str(item).strip()]
|
||
optional_enhancements = [
|
||
str(item).strip() for item in result.get("optional_enhancements") or [] if str(item).strip()
|
||
]
|
||
valid_claims: list[dict[str, Any]] = []
|
||
|
||
for raw_claim in result.get("claims") or []:
|
||
claim = dict(raw_claim) if isinstance(raw_claim, dict) else {}
|
||
text = str(claim.get("text") or "").strip()
|
||
evidence_ids = [str(item) for item in claim.get("evidence_ids") or []]
|
||
if not text:
|
||
_warn(warnings, "empty_claim")
|
||
continue
|
||
if not evidence_ids or any(item.startswith("rag_") or item not in known_ids for item in evidence_ids):
|
||
_warn(warnings, "unsupported_evidence_reference")
|
||
continue
|
||
valid_claims.append(claim)
|
||
|
||
optimized, quarantined = _partition_text(str(result.get("optimized_description") or "").strip(), evidence, [])
|
||
bullets: list[str] = []
|
||
for value in result.get("bullets") or []:
|
||
bullet, bullet_suggestions = _partition_text(str(value).strip(), evidence, [])
|
||
quarantined.extend(bullet_suggestions)
|
||
if bullet:
|
||
bullets.append(bullet)
|
||
|
||
if not optimized and quarantined:
|
||
optimized = _primary_description(ledger)
|
||
_warn(warnings, "candidate_contains_unconfirmed_additions")
|
||
if quarantined:
|
||
_warn(warnings, "suggestion_requires_confirmation")
|
||
suggestions.extend(quarantined)
|
||
result["claims"] = valid_claims
|
||
result["optimized_description"] = optimized
|
||
result["bullets"] = list(dict.fromkeys(bullets))[:5]
|
||
result["unconfirmed_suggestions"] = list(dict.fromkeys(suggestions))[:6]
|
||
result["optional_enhancements"] = list(dict.fromkeys(optional_enhancements))[:6]
|
||
if warnings:
|
||
result["validation_warnings"] = list(dict.fromkeys(warnings))
|
||
return result
|
||
|
||
|
||
def partition_entry_text(text: str, facts: list[Any]) -> tuple[str, list[str], list[str]]:
|
||
"""Strictly partition imported/RAG-expanded text from its source evidence.
|
||
|
||
Unlike a user-requested resume optimization proposal, imported content must
|
||
never silently turn a source fact into a different metric or deliverable.
|
||
"""
|
||
ledger = normalize_fact_ledger(facts)
|
||
evidence = "\n".join(item["text"] for item in ledger)
|
||
confirmed: list[str] = []
|
||
suggestions: list[str] = []
|
||
for sentence in _SENTENCE.split(text.strip()):
|
||
clean = sentence.strip()
|
||
if not clean:
|
||
continue
|
||
if _has_unconfirmed_signature(clean, evidence):
|
||
suggestions.append(clean)
|
||
else:
|
||
confirmed.append(clean)
|
||
result = _rejoin_sentences(confirmed, had_line_breaks="\n" in text)
|
||
warnings: list[str] = []
|
||
if not result and suggestions:
|
||
result = _primary_description(ledger)
|
||
warnings.append("candidate_contains_unconfirmed_additions")
|
||
if suggestions:
|
||
warnings.append("suggestion_requires_confirmation")
|
||
return result, suggestions, warnings
|
||
|
||
|
||
def _rejoin_sentences(sentences: list[str], *, had_line_breaks: bool) -> str:
|
||
"""Rejoin partitioned sentences, keeping one-statement-per-line layout.
|
||
|
||
Bullet-style candidates are written one per line; flattening them with
|
||
spaces would cram the whole description into a single paragraph.
|
||
"""
|
||
separator = "\n" if had_line_breaks else " "
|
||
return separator.join(sentences).strip()
|
||
|
||
|
||
def are_quantified_facts_grounded(text: str, evidence: str) -> bool:
|
||
return set(_NUMBER.findall(text)).issubset(set(_NUMBER.findall(evidence)))
|
||
|
||
|
||
def is_grounded_resume_text(text: str, evidence: str) -> bool:
|
||
return are_quantified_facts_grounded(text, evidence) and _technical_terms(text).issubset(_technical_terms(evidence))
|
||
|
||
|
||
def quantified_fact_contexts(evidence: str) -> list[dict[str, str]]:
|
||
contexts: list[dict[str, str]] = []
|
||
for clause in re.split(r"[。!?!?;;.\n]+", evidence):
|
||
clean = clause.strip()
|
||
for number in _NUMBER.findall(clean):
|
||
contexts.append({"number": number, "unit": "", "context": clean})
|
||
return contexts[:12]
|
||
|
||
|
||
def _partition_text(text: str, evidence: str, invalid_claim_texts: list[str]) -> tuple[str, list[str]]:
|
||
# Do not use lexical overlap as an acceptance gate. Models commonly turn a
|
||
# user sentence into several resume bullets or use a stronger role-oriented
|
||
# paraphrase; hiding those sentences invokes rule fallbacks needlessly.
|
||
del evidence, invalid_claim_texts
|
||
return text.strip(), []
|
||
|
||
|
||
def _has_unconfirmed_signature(text: str, evidence: str) -> bool:
|
||
if not _technical_terms(text).issubset(_technical_terms(evidence)):
|
||
return True
|
||
known_by_number: dict[str, set[tuple[str, str]]] = {}
|
||
for number, unit, object_name in _counted_objects(evidence):
|
||
known_by_number.setdefault(number, set()).add((unit, object_name))
|
||
for number, unit, object_name in _counted_objects(text):
|
||
known = known_by_number.get(number)
|
||
if known and (unit, object_name) not in known:
|
||
return True
|
||
evidence_numbers = {number.rstrip("%") for number in _NUMBER.findall(evidence)}
|
||
# A percentage paraphrase ("前百分之10" -> "前 10%") is the same fact; the
|
||
# grounding gate compares numeric values, not surface percent signs.
|
||
return any(number.rstrip("%") not in evidence_numbers for number in _NUMBER.findall(text))
|
||
|
||
|
||
def _technical_terms(text: str) -> set[str]:
|
||
return {term.casefold().rstrip(".,;:!?") for term in _LATIN_TERM.findall(text) if term.casefold().rstrip(".,;:!?") in _COMMON_TECH_TERMS}
|
||
|
||
|
||
def _counted_objects(text: str) -> list[tuple[str, str, str]]:
|
||
values: list[tuple[str, str, str]] = []
|
||
for match in _COUNTED_OBJECT.finditer(text):
|
||
unit = match.group("unit").casefold()
|
||
object_name = "" if unit.isascii() else match.group("object").strip().casefold()[:24]
|
||
values.append((match.group("number"), unit, object_name))
|
||
return values
|
||
|
||
|
||
def _primary_description(ledger: list[dict[str, str]]) -> str:
|
||
return next((item["text"] for item in ledger if item.get("field") == "description"), "")
|
||
|
||
|
||
def _normalize(text: str) -> str:
|
||
return "".join(character.casefold() for character in text if character.isalnum())
|
||
|
||
|
||
def _warn(warnings: list[str], value: str) -> None:
|
||
if value not in warnings:
|
||
warnings.append(value) |