feat: initialize resume agent with OfferPai sync

This commit is contained in:
Codex
2026-08-05 20:20:18 +08:00
commit 61ec750031
197 changed files with 33291 additions and 0 deletions
+183
View File
@@ -0,0 +1,183 @@
"""Validation and safety partitioning for resume optimization proposals."""
from __future__ import annotations
import re
from typing import Any
from .experience_optimizer import normalize_fact_ledger
from .text_normalization import decode_literal_unicode_escapes
_NUMBER = re.compile(r"\d+(?:\.\d+)?%?")
_LATIN_TERM = re.compile(r"[A-Za-z][A-Za-z0-9.+#_-]{1,}")
_COMMON_TECH_TERMS = frozenset({
"aws", "azure", "docker", "elasticsearch", "fastapi", "flask", "git", "go",
"java", "javascript", "kafka", "kubernetes", "langchain", "langgraph", "linux",
"mongodb", "mysql", "nextjs", "nodejs", "numpy", "openai", "pandas", "postgresql",
"python", "pytorch", "rabbitmq", "react", "redis", "spring", "sql", "tensorflow",
"typescript", "vue", "vue3",
})
_SENTENCE = re.compile(r"(?<=[。!?!?;])\s*|\n+")
_COUNTED_OBJECT = re.compile(
r"(?P<number>\d+(?:\.\d+)?)(?:\s*)(?P<unit>名|位|人|项|个|次|台|条|份|家|天|月|年|students?|classmates?|users?|features?|services?|projects?|requests?)(?:\s*)(?P<object>[A-Za-z][A-Za-z -]{0,24}|[\u4e00-\u9fff]{0,8})",
re.I,
)
def validate_proposal(proposal: dict[str, Any], facts: list[Any]) -> dict[str, Any]:
"""Normalize proposal metadata without suppressing useful model-written prose.
The fact ledger validates claim references and aids diagnostics. It is not a
word-for-word acceptance gate for optimized prose: resume editing needs
paraphrase, synthesis, and controlled role-oriented expansion.
"""
result = decode_literal_unicode_escapes(dict(proposal))
ledger = normalize_fact_ledger(facts)
known_ids = {item["id"] for item in ledger}
evidence = "\n".join(item["text"] for item in ledger)
warnings = [str(item) for item in result.get("validation_warnings") or [] if str(item)]
suggestions = [str(item).strip() for item in result.get("unconfirmed_suggestions") or [] if str(item).strip()]
optional_enhancements = [
str(item).strip() for item in result.get("optional_enhancements") or [] if str(item).strip()
]
valid_claims: list[dict[str, Any]] = []
for raw_claim in result.get("claims") or []:
claim = dict(raw_claim) if isinstance(raw_claim, dict) else {}
text = str(claim.get("text") or "").strip()
evidence_ids = [str(item) for item in claim.get("evidence_ids") or []]
if not text:
_warn(warnings, "empty_claim")
continue
if not evidence_ids or any(item.startswith("rag_") or item not in known_ids for item in evidence_ids):
_warn(warnings, "unsupported_evidence_reference")
continue
valid_claims.append(claim)
optimized, quarantined = _partition_text(str(result.get("optimized_description") or "").strip(), evidence, [])
bullets: list[str] = []
for value in result.get("bullets") or []:
bullet, bullet_suggestions = _partition_text(str(value).strip(), evidence, [])
quarantined.extend(bullet_suggestions)
if bullet:
bullets.append(bullet)
if not optimized and quarantined:
optimized = _primary_description(ledger)
_warn(warnings, "candidate_contains_unconfirmed_additions")
if quarantined:
_warn(warnings, "suggestion_requires_confirmation")
suggestions.extend(quarantined)
result["claims"] = valid_claims
result["optimized_description"] = optimized
result["bullets"] = list(dict.fromkeys(bullets))[:5]
result["unconfirmed_suggestions"] = list(dict.fromkeys(suggestions))[:6]
result["optional_enhancements"] = list(dict.fromkeys(optional_enhancements))[:6]
if warnings:
result["validation_warnings"] = list(dict.fromkeys(warnings))
return result
def partition_entry_text(text: str, facts: list[Any]) -> tuple[str, list[str], list[str]]:
"""Strictly partition imported/RAG-expanded text from its source evidence.
Unlike a user-requested resume optimization proposal, imported content must
never silently turn a source fact into a different metric or deliverable.
"""
ledger = normalize_fact_ledger(facts)
evidence = "\n".join(item["text"] for item in ledger)
confirmed: list[str] = []
suggestions: list[str] = []
for sentence in _SENTENCE.split(text.strip()):
clean = sentence.strip()
if not clean:
continue
if _has_unconfirmed_signature(clean, evidence):
suggestions.append(clean)
else:
confirmed.append(clean)
result = _rejoin_sentences(confirmed, had_line_breaks="\n" in text)
warnings: list[str] = []
if not result and suggestions:
result = _primary_description(ledger)
warnings.append("candidate_contains_unconfirmed_additions")
if suggestions:
warnings.append("suggestion_requires_confirmation")
return result, suggestions, warnings
def _rejoin_sentences(sentences: list[str], *, had_line_breaks: bool) -> str:
"""Rejoin partitioned sentences, keeping one-statement-per-line layout.
Bullet-style candidates are written one per line; flattening them with
spaces would cram the whole description into a single paragraph.
"""
separator = "\n" if had_line_breaks else " "
return separator.join(sentences).strip()
def are_quantified_facts_grounded(text: str, evidence: str) -> bool:
return set(_NUMBER.findall(text)).issubset(set(_NUMBER.findall(evidence)))
def is_grounded_resume_text(text: str, evidence: str) -> bool:
return are_quantified_facts_grounded(text, evidence) and _technical_terms(text).issubset(_technical_terms(evidence))
def quantified_fact_contexts(evidence: str) -> list[dict[str, str]]:
contexts: list[dict[str, str]] = []
for clause in re.split(r"[。!?!?;.\n]+", evidence):
clean = clause.strip()
for number in _NUMBER.findall(clean):
contexts.append({"number": number, "unit": "", "context": clean})
return contexts[:12]
def _partition_text(text: str, evidence: str, invalid_claim_texts: list[str]) -> tuple[str, list[str]]:
# Do not use lexical overlap as an acceptance gate. Models commonly turn a
# user sentence into several resume bullets or use a stronger role-oriented
# paraphrase; hiding those sentences invokes rule fallbacks needlessly.
del evidence, invalid_claim_texts
return text.strip(), []
def _has_unconfirmed_signature(text: str, evidence: str) -> bool:
if not _technical_terms(text).issubset(_technical_terms(evidence)):
return True
known_by_number: dict[str, set[tuple[str, str]]] = {}
for number, unit, object_name in _counted_objects(evidence):
known_by_number.setdefault(number, set()).add((unit, object_name))
for number, unit, object_name in _counted_objects(text):
known = known_by_number.get(number)
if known and (unit, object_name) not in known:
return True
evidence_numbers = {number.rstrip("%") for number in _NUMBER.findall(evidence)}
# A percentage paraphrase ("前百分之10" -> "前 10%") is the same fact; the
# grounding gate compares numeric values, not surface percent signs.
return any(number.rstrip("%") not in evidence_numbers for number in _NUMBER.findall(text))
def _technical_terms(text: str) -> set[str]:
return {term.casefold().rstrip(".,;:!?") for term in _LATIN_TERM.findall(text) if term.casefold().rstrip(".,;:!?") in _COMMON_TECH_TERMS}
def _counted_objects(text: str) -> list[tuple[str, str, str]]:
values: list[tuple[str, str, str]] = []
for match in _COUNTED_OBJECT.finditer(text):
unit = match.group("unit").casefold()
object_name = "" if unit.isascii() else match.group("object").strip().casefold()[:24]
values.append((match.group("number"), unit, object_name))
return values
def _primary_description(ledger: list[dict[str, str]]) -> str:
return next((item["text"] for item in ledger if item.get("field") == "description"), "")
def _normalize(text: str) -> str:
return "".join(character.casefold() for character in text if character.isalnum())
def _warn(warnings: list[str], value: str) -> None:
if value not in warnings:
warnings.append(value)