Files
resume-agent/backend/app/experience_optimizer.py
T

551 lines
23 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Grounded experience optimization backed by structured model output."""
from __future__ import annotations
import re
from typing import Any, Protocol
from .experience_optimizer_models import ExperienceOptimizationOutput
from .llm_services import LLMServiceError, OpenAICompatibleStructuredClient, log_ai_event
from .settings import Settings
Fact = dict[str, str]
class ExperienceOptimizer(Protocol):
def optimize(
self, entry: dict[str, Any], *, context: dict[str, Any], facts: list[Any]
) -> dict[str, Any]: ...
class OpenAIExperienceOptimizer:
"""Build a grounded proposal from user facts; RAG is style-only context."""
def __init__(self, completion: Any, retriever: Any = None, embedder: Any = None) -> None:
self.completion = completion
self.retriever = retriever
self.embedder = embedder
def optimize(
self, entry: dict[str, Any], *, context: dict[str, Any], facts: list[Any]
) -> dict[str, Any]:
ledger = normalize_fact_ledger(facts)
output: ExperienceOptimizationOutput = self.completion.complete(
schema=ExperienceOptimizationOutput,
schema_name="experience_optimization",
system_prompt=_OPTIMIZATION_PROMPT,
payload={
"user_fact_ledger": ledger,
"primary_narrative": str(entry.get("description") or "").strip(),
"resume_context": {
"target_position": context.get("target_position"),
"major": context.get("major"),
"entry_type": context.get("entry_type"),
"optimization_mode": context.get("optimization_mode", "light"),
"user_instruction": context.get("instruction"),
},
"deep_interview": {
"completion": context.get("interview_completion") or {},
"completed_dimensions": context.get("completed_dimensions") or [],
"question_history": context.get("question_history") or [],
},
"style_references": self._retrieve(ledger, context),
},
)
proposal = self._with_fact_coverage(
self._grounded_proposal(output.model_dump(), ledger), ledger
)
reason = self._quality_reason(proposal, entry, ledger)
if reason:
proposal = self._repair(entry, context, ledger, proposal, reason)
proposal["source"] = "ai_expanded"
return proposal
def _repair(
self,
entry: dict[str, Any],
context: dict[str, Any],
ledger: list[Fact],
proposal: dict[str, Any],
reason: str,
) -> dict[str, Any]:
log_ai_event(
"experience_optimization_repair_started",
reason_code=reason,
entry_type=str(context.get("entry_type") or ""),
optimization_mode=str(context.get("optimization_mode") or "light"),
required_fact_count=len(self._required_fact_ids(ledger)),
omitted_fact_count=len(proposal.get("omitted_fact_ids") or []),
)
try:
repaired: ExperienceOptimizationOutput = self.completion.complete(
schema=ExperienceOptimizationOutput,
schema_name="experience_optimization_repair",
system_prompt=_OPTIMIZATION_REPAIR_PROMPT,
payload={
"user_fact_ledger": ledger,
"primary_narrative": str(entry.get("description") or "").strip(),
"canonical_fact_draft": _canonical_fact_draft(ledger),
"rejected_candidate": proposal,
"rejected_reason": reason,
"required_fact_ids": self._required_fact_ids(ledger),
"omitted_fact_ids": proposal.get("omitted_fact_ids") or [],
"entry_type": context.get("entry_type"),
"resume_context": {
"target_position": context.get("target_position"),
"major": context.get("major"),
"optimization_mode": context.get("optimization_mode", "light"),
},
"deep_interview": {
"completion": context.get("interview_completion") or {},
"completed_dimensions": context.get("completed_dimensions") or [],
"question_history": context.get("question_history") or [],
},
"validation_requirements": [
"optimized_description must not be empty",
"every claim must cite only user_fact_ledger IDs",
"retain every material confirmed fact; only merge genuinely duplicate wording",
"put non-blocking improvement ideas in optional_enhancements",
"do not put unconfirmed identity facts into optimized_description",
"do not return punctuation-only text or a raw field dump",
],
},
)
except LLMServiceError:
raise
except Exception as exc:
raise LLMServiceError(
"Experience optimization repair failed",
reason_code="repair_failed",
stage="experience_repair",
safe_summary=type(exc).__name__,
) from exc
repaired_proposal = self._with_fact_coverage(
self._grounded_proposal(repaired.model_dump(), ledger), ledger
)
repaired_reason = self._quality_reason(repaired_proposal, entry, ledger)
if repaired_reason == "empty_result":
log_ai_event(
"experience_optimization_repair_rejected",
reason_code=repaired_reason,
entry_type=str(context.get("entry_type") or ""),
optimization_mode=str(context.get("optimization_mode") or "light"),
)
raise LLMServiceError(
"Repaired model output was empty",
reason_code="empty_result",
stage="experience_repair_validation",
safe_summary=repaired_reason,
)
if repaired_reason:
repaired_proposal.setdefault("validation_warnings", []).append(
"material_fact_omitted_after_repair"
if repaired_reason == "material_fact_omitted"
else repaired_reason
)
log_ai_event(
"experience_optimization_repair_relaxed",
reason_code=repaired_reason,
entry_type=str(context.get("entry_type") or ""),
optimization_mode=str(context.get("optimization_mode") or "light"),
)
return repaired_proposal
@staticmethod
def _grounded_proposal(output: dict[str, Any], ledger: list[Fact]) -> dict[str, Any]:
from .claim_validator import validate_proposal
return validate_proposal(output, ledger)
@staticmethod
def _quality_reason(
proposal: dict[str, Any], entry: dict[str, Any], ledger: list[Fact]
) -> str | None:
# A candidate that drops confirmed material facts (feature lists, product
# intro, outcomes) gets exactly one repair pass. If the repair still omits
# them, _repair relaxes with a warning — omissions never veto the draft.
optimized = str(proposal.get("optimized_description") or "").strip()
if not optimized or not any(character.isalnum() for character in optimized):
return "empty_result"
if proposal.get("omitted_fact_ids"):
return "material_fact_omitted"
return None
@staticmethod
def _required_fact_ids(ledger: list[Fact]) -> list[str]:
return required_material_fact_ids(ledger)
@classmethod
def _with_fact_coverage(cls, proposal: dict[str, Any], ledger: list[Fact]) -> dict[str, Any]:
required = cls._required_fact_ids(ledger)
narrative = " ".join(
[
str(proposal.get("optimized_description") or ""),
*[str(item) for item in proposal.get("bullets") or []],
]
)
covered = [
fact_id
for fact_id in required
if _fact_text_is_preserved(fact_id, ledger, narrative)
]
proposal["covered_fact_ids"] = covered
proposal["omitted_fact_ids"] = [
fact_id for fact_id in required if fact_id not in covered
]
return proposal
def style_references(self, facts: list[Any], context: dict[str, Any]) -> list[Any]:
"""Expose non-blocking RAG style examples to the gap-analysis role."""
try:
return self._retrieve(normalize_fact_ledger(facts), context)
except Exception:
return []
def _retrieve(
self, ledger: list[Fact], context: dict[str, Any]
) -> list[dict[str, str]]:
query = " ".join(item["text"] for item in ledger)
if context.get("target_position"):
query = f"{context['target_position']} {query}".strip()
try:
results = self.retriever.retrieve(
query_text=query or "resume experience optimization",
embedder=self.embedder,
position_category=context.get("target_position") or None,
exp_type=context.get("entry_type") or None,
k=3,
)
except Exception:
return []
return [
{
"id": str(item.get("id") or f"rag_{index}"),
"title": str(item.get("title") or item.get("title_path") or "reference"),
"content": str(item.get("content") or item.get("optimized") or ""),
"writing_points": str(item.get("points") or ""),
}
for index, item in enumerate(results[:3], start=1)
]
class FallbackExperienceOptimizer:
def __init__(self, primary: ExperienceOptimizer, fallback: ExperienceOptimizer) -> None:
self.primary = primary
self.fallback = fallback
def optimize(
self, entry: dict[str, Any], *, context: dict[str, Any], facts: list[Any]
) -> dict[str, Any]:
try:
proposal = self.primary.optimize(entry, context=context, facts=facts)
proposal["generation_source"] = "llm"
return proposal
except Exception as exc:
reason = _fallback_reason(exc)
log_ai_event(
"experience_optimization_failed",
reason_code=reason,
trace_id=getattr(exc, "trace_id", None),
stage=getattr(exc, "stage", "experience_optimization"),
entry_type=str(context.get("entry_type") or ""),
optimization_mode=str(context.get("optimization_mode") or "light"),
exception=type(exc).__name__,
)
if isinstance(exc, LLMServiceError):
raise
proposal = self.fallback.optimize(entry, context=context, facts=facts)
proposal["generation_source"] = "rule_fallback"
proposal["fallback_reason"] = reason
return proposal
def style_references(self, facts: list[Any], context: dict[str, Any]) -> list[Any]:
provider = getattr(self.primary, "style_references", None)
if provider is None:
return []
try:
return list(provider(facts, context) or [])
except Exception:
return []
class RuleStructuredExperienceOptimizer:
"""Offline generator for tests and explicit no-model fallback mode."""
def optimize(
self, entry: dict[str, Any], *, context: dict[str, Any], facts: list[Any]
) -> dict[str, Any]:
ledger = normalize_fact_ledger(facts)
description = str(entry.get("description") or "").strip()
material = "; ".join(dict.fromkeys(item["text"] for item in ledger))
if not material:
return {
"optimized_description": "",
"changes": [],
"missing_facts": ["specific action", "method or tool", "verifiable result"],
"star": {},
"claims": [],
"bullets": [],
"source": "rule_structured",
"generation_source": "rule",
"fallback_reason": "insufficient_user_facts",
}
optimized = material
return {
"optimized_description": optimized,
"bullets": [optimized],
"changes": ["reorganized confirmed actions and facts"],
"missing_facts": _missing_facts(material),
"star": {
"situation": description or None,
"task": str(entry.get("title") or entry.get("position") or "") or None,
"action": material,
"result": _known_result(material),
},
"claims": [],
"source": "rule_structured",
"generation_source": "rule",
}
def _fallback_reason(exc: Exception) -> str:
if isinstance(exc, LLMServiceError):
return exc.reason_code
return type(exc).__name__.lower()[:48]
def build_experience_optimizer(
settings: Settings, client: Any | None = None
) -> ExperienceOptimizer:
"""Light optimizer: pure LLM on user facts (RAG knowledge base removed)."""
rules = RuleStructuredExperienceOptimizer()
if not settings.use_openai:
return rules
completion = OpenAICompatibleStructuredClient(settings, client)
primary: ExperienceOptimizer = OpenAIExperienceOptimizer(completion)
return FallbackExperienceOptimizer(primary, rules) if settings.fallback_to_rules else primary
def normalize_fact_ledger(facts: list[Any]) -> list[Fact]:
ledger: list[Fact] = []
for index, value in enumerate(facts, start=1):
if isinstance(value, dict):
text = str(value.get("text") or "").strip()
fact = {
"id": str(value.get("id") or f"fact_{index}"),
"source": str(value.get("source") or "user_form"),
"field": str(value.get("field") or "unknown"),
"text": text,
}
else:
text = str(value).strip()
fact = {
"id": f"fact_{index}",
"source": "user_form",
"field": "unknown",
"text": text,
}
if text:
ledger.append(fact)
return _append_description_parts(ledger)
_DESCRIPTION_ITEM_MARKER = re.compile(r"^\s*\d+\s*[.、)]\s*")
_DESCRIPTION_LINE_LABEL = re.compile(r"^[\u4e00-\u9fff]{2,8}[:]\s*")
def split_description_parts(text: str) -> list[str]:
"""Split a structured description into independently checkable fragments.
A long multi-line description judged as one fact lets dropped features hide
behind the overall n-gram coverage of the kept tech stack. Line/clause
fragments make each feature, intro, or outcome its own gate entry. Short
single-sentence descriptions stay unsplit (one fragment -> caller keeps the
parent fact).
"""
parts: list[str] = []
for raw_line in str(text).splitlines():
line = raw_line.strip()
if not line:
continue
segments = re.split(r"[。;;]", line) if len(line) > 40 else [line]
for segment in segments:
part = _DESCRIPTION_LINE_LABEL.sub("", _DESCRIPTION_ITEM_MARKER.sub("", segment.strip())).strip()
if len(part) >= 4:
parts.append(part)
return parts
def _append_description_parts(ledger: list[Fact]) -> list[Fact]:
expanded: list[Fact] = []
for fact in ledger:
expanded.append(fact)
if fact.get("field") != "description":
continue
parts = split_description_parts(fact["text"])
if len(parts) < 2:
continue
for part_index, part in enumerate(parts, start=1):
expanded.append(
{
"id": f"{fact['id']}_part_{part_index}",
"source": fact.get("source") or "user_form",
"field": "description_part",
"text": part,
}
)
return expanded
def required_material_fact_ids(ledger: list[Fact]) -> list[str]:
"""Ids of facts that must survive in the narrative.
When a description was split into fragments, the fragments stand in for the
parent so coverage is judged per fragment, not per whole entry.
"""
split_parents = {
fact["id"].rsplit("_part_", 1)[0]
for fact in ledger
if fact.get("field") == "description_part"
}
return [
fact["id"]
for fact in ledger
if _is_material_resume_fact(fact) and fact["id"] not in split_parents
]
def _is_material_resume_fact(fact: Fact) -> bool:
"""Facts that belong in the narrative rather than only in card metadata."""
if fact.get("field") in {
"degree", "start_date", "end_date_or_present", "date", "title", "name",
"company", "organization", "school", "project_name", "position", "role",
"project_role", "major", "award",
}:
return False
return bool(str(fact.get("text") or "").strip()) and (
fact.get("field") in {"description", "description_part"}
or fact.get("source") == "user_answer"
)
def _fact_text_is_preserved(fact_id: str, ledger: list[Fact], narrative: str) -> bool:
fact = next((item for item in ledger if item["id"] == fact_id), None)
if fact is None:
return False
source = _normalized_text(fact["text"])
target = _normalized_text(narrative)
if not source or not target:
return False
if source in target:
return True
latin_terms = [
_latin_stem(term)
for term in re.findall(r"[A-Za-z][A-Za-z0-9+#._-]{1,}", fact["text"])
if term.casefold() not in _LATIN_STOPWORDS
]
target_latin_terms = {
_latin_stem(term)
for term in re.findall(r"[A-Za-z][A-Za-z0-9+#._-]{1,}", narrative)
if term.casefold() not in _LATIN_STOPWORDS
}
latin_covered = not latin_terms or sum(
term in target_latin_terms for term in latin_terms
) >= max(1, int(len(latin_terms) * 0.6 + 0.999))
if not latin_covered:
return False
chinese_segments = re.findall(r"[\u4e00-\u9fff]{2,}", fact["text"])
chinese_ngrams = {
segment[index:index + size]
for segment in chinese_segments
for size in (2, 3, 4)
for index in range(max(0, len(segment) - size + 1))
}
matched_ngrams = sum(
_normalized_text(token) in target for token in chinese_ngrams
)
chinese_covered = not chinese_ngrams or matched_ngrams >= max(
1, int(len(chinese_ngrams) * 0.65)
)
number_tokens = re.findall(r"\d+(?:\.\d+)?%?", fact["text"])
numbers_covered = all(token in narrative for token in number_tokens)
return chinese_covered and numbers_covered and bool(
latin_terms or chinese_ngrams or number_tokens
)
_LATIN_STOPWORDS = frozenset({
"and", "are", "for", "from", "into", "its", "that", "the", "this", "through", "using", "with",
})
def _latin_stem(term: str) -> str:
normalized = term.casefold().rstrip(".,;:!?")
if normalized == "built":
return "build"
if normalized == "ran":
return "run"
for suffix in ("ing", "ed", "es", "s"):
if normalized.endswith(suffix) and len(normalized) - len(suffix) >= 4:
return normalized[:-len(suffix)]
return normalized
def _normalized_text(text: str) -> str:
return "".join(character.casefold() for character in text if character.isalnum())
def _missing_facts(text: str) -> list[str]:
missing: list[str] = []
if not any(token in text for token in ("%", "result", "impact", "users")):
missing.append("verifiable result or impact")
if not any(token in text.casefold() for token in ("using", "with", "through", "via")):
missing.append("method, tool, or collaboration approach")
return missing or ["scope of responsibility"]
def _known_result(text: str) -> str | None:
markers = ("%", "improved", "reduced", "completed", "launched", "users")
return text if any(marker in text.casefold() for marker in markers) else None
def _canonical_fact_draft(ledger: list[Fact]) -> str:
return "; ".join(dict.fromkeys(item["text"] for item in ledger if item.get("text")))
_OPTIMIZATION_PROMPT = """
你是一名中文简历经历编辑。只返回符合 output_json_schema 的 JSON。
请将用户提供的经历改写为专业、可直接用于简历的中文正文,并采用自然的 STAR
结构。完整性优先于篇幅:保留用户已确认的职责、动作、方法、工具、协作、范围、
交付物和结果。可以重排和合并真正重复的措辞,但不得为了缩短文本删除有意义的事实;
必要时可使用多句或多条要点。项目或产品的功能模块、平台定位/简介与量化成果,
与技术栈同等重要:不得只保留技术栈而省略功能点、平台简介或成果描述。
deep_interview.completion.is_sufficient 为 true 时,表示 LangGraph 已确认当前候选稿
所需信息足够。此时不得把任何已回答维度重新列为 missing_facts,也不要把泛泛的
“补充技术栈、量化结果或职责”当作当前候选稿的阻塞条件。若存在不影响当前候选稿的
提升方向,只能放入 optional_enhancements,且要明确是可选增强。
style_references 只用于学习表达方式,不是用户个人事实。不得虚构公司、学校、
奖项、证书、日期或归属。可以基于用户的经历语义做自然的职业化改写、结构化归纳和适度的
岗位导向扩展;不要因为原文没有逐字写出某个方法或影响就机械省略整段内容。量化表达
应优先使用用户确认的数据;未确认时可以使用不带精确数字的合理影响描述。每条 claim
必须只引用 user_fact_ledger 中的 evidence_ids,绝不能引用 rag_ IDs。
""".strip()
_OPTIMIZATION_REPAIR_PROMPT = """
你负责修复一份未通过确定性校验的中文简历候选稿。只返回符合 output_json_schema 的
JSON。rejected_candidate 和 rejected_reason 只说明缺陷,不是新的事实来源。
输出非空、专业、可直接用于简历的中文叙述,采用自然的 STAR 结构。修复的首要目标是保留 required_fact_ids
对应的全部重要事实,包括原描述和深度追问答案中的动作、方法、工具、范围、协作、
交付物和结果。不要为了简洁而压缩掉这些信息;可使用多句或多条要点,只合并语义重复
的表达。允许自然的同义改写、结构化归纳和岗位导向扩展,不要求逐字复述每项事实。
不得悄然编造公司、学校、奖项、证书、日期或归属。非阻塞的后续提升方向放入
optional_enhancements。每条 claim 必须引用已有 user_fact_ledger evidence_ids。不要
返回只有标点的文本或原始表单字段拼接。
""".strip()