"""Grounded experience optimization backed by structured model output.""" from __future__ import annotations import re from typing import Any, Protocol from .experience_optimizer_models import ExperienceOptimizationOutput from .llm_services import LLMServiceError, OpenAICompatibleStructuredClient, log_ai_event from .settings import Settings Fact = dict[str, str] class ExperienceOptimizer(Protocol): def optimize( self, entry: dict[str, Any], *, context: dict[str, Any], facts: list[Any] ) -> dict[str, Any]: ... class OpenAIExperienceOptimizer: """Build a grounded proposal from user facts; RAG is style-only context.""" def __init__(self, completion: Any, retriever: Any = None, embedder: Any = None) -> None: self.completion = completion self.retriever = retriever self.embedder = embedder def optimize( self, entry: dict[str, Any], *, context: dict[str, Any], facts: list[Any] ) -> dict[str, Any]: ledger = normalize_fact_ledger(facts) output: ExperienceOptimizationOutput = self.completion.complete( schema=ExperienceOptimizationOutput, schema_name="experience_optimization", system_prompt=_OPTIMIZATION_PROMPT, payload={ "user_fact_ledger": ledger, "primary_narrative": str(entry.get("description") or "").strip(), "resume_context": { "target_position": context.get("target_position"), "major": context.get("major"), "entry_type": context.get("entry_type"), "optimization_mode": context.get("optimization_mode", "light"), "user_instruction": context.get("instruction"), }, "deep_interview": { "completion": context.get("interview_completion") or {}, "completed_dimensions": context.get("completed_dimensions") or [], "question_history": context.get("question_history") or [], }, "style_references": self._retrieve(ledger, context), }, ) proposal = self._with_fact_coverage( self._grounded_proposal(output.model_dump(), ledger), ledger ) reason = self._quality_reason(proposal, entry, ledger) if reason: proposal = self._repair(entry, context, ledger, proposal, reason) proposal["source"] = "ai_expanded" return proposal def _repair( self, entry: dict[str, Any], context: dict[str, Any], ledger: list[Fact], proposal: dict[str, Any], reason: str, ) -> dict[str, Any]: log_ai_event( "experience_optimization_repair_started", reason_code=reason, entry_type=str(context.get("entry_type") or ""), optimization_mode=str(context.get("optimization_mode") or "light"), required_fact_count=len(self._required_fact_ids(ledger)), omitted_fact_count=len(proposal.get("omitted_fact_ids") or []), ) try: repaired: ExperienceOptimizationOutput = self.completion.complete( schema=ExperienceOptimizationOutput, schema_name="experience_optimization_repair", system_prompt=_OPTIMIZATION_REPAIR_PROMPT, payload={ "user_fact_ledger": ledger, "primary_narrative": str(entry.get("description") or "").strip(), "canonical_fact_draft": _canonical_fact_draft(ledger), "rejected_candidate": proposal, "rejected_reason": reason, "required_fact_ids": self._required_fact_ids(ledger), "omitted_fact_ids": proposal.get("omitted_fact_ids") or [], "entry_type": context.get("entry_type"), "resume_context": { "target_position": context.get("target_position"), "major": context.get("major"), "optimization_mode": context.get("optimization_mode", "light"), }, "deep_interview": { "completion": context.get("interview_completion") or {}, "completed_dimensions": context.get("completed_dimensions") or [], "question_history": context.get("question_history") or [], }, "validation_requirements": [ "optimized_description must not be empty", "every claim must cite only user_fact_ledger IDs", "retain every material confirmed fact; only merge genuinely duplicate wording", "put non-blocking improvement ideas in optional_enhancements", "do not put unconfirmed identity facts into optimized_description", "do not return punctuation-only text or a raw field dump", ], }, ) except LLMServiceError: raise except Exception as exc: raise LLMServiceError( "Experience optimization repair failed", reason_code="repair_failed", stage="experience_repair", safe_summary=type(exc).__name__, ) from exc repaired_proposal = self._with_fact_coverage( self._grounded_proposal(repaired.model_dump(), ledger), ledger ) repaired_reason = self._quality_reason(repaired_proposal, entry, ledger) if repaired_reason == "empty_result": log_ai_event( "experience_optimization_repair_rejected", reason_code=repaired_reason, entry_type=str(context.get("entry_type") or ""), optimization_mode=str(context.get("optimization_mode") or "light"), ) raise LLMServiceError( "Repaired model output was empty", reason_code="empty_result", stage="experience_repair_validation", safe_summary=repaired_reason, ) if repaired_reason: repaired_proposal.setdefault("validation_warnings", []).append( "material_fact_omitted_after_repair" if repaired_reason == "material_fact_omitted" else repaired_reason ) log_ai_event( "experience_optimization_repair_relaxed", reason_code=repaired_reason, entry_type=str(context.get("entry_type") or ""), optimization_mode=str(context.get("optimization_mode") or "light"), ) return repaired_proposal @staticmethod def _grounded_proposal(output: dict[str, Any], ledger: list[Fact]) -> dict[str, Any]: from .claim_validator import validate_proposal return validate_proposal(output, ledger) @staticmethod def _quality_reason( proposal: dict[str, Any], entry: dict[str, Any], ledger: list[Fact] ) -> str | None: # A candidate that drops confirmed material facts (feature lists, product # intro, outcomes) gets exactly one repair pass. If the repair still omits # them, _repair relaxes with a warning — omissions never veto the draft. optimized = str(proposal.get("optimized_description") or "").strip() if not optimized or not any(character.isalnum() for character in optimized): return "empty_result" if proposal.get("omitted_fact_ids"): return "material_fact_omitted" return None @staticmethod def _required_fact_ids(ledger: list[Fact]) -> list[str]: return required_material_fact_ids(ledger) @classmethod def _with_fact_coverage(cls, proposal: dict[str, Any], ledger: list[Fact]) -> dict[str, Any]: required = cls._required_fact_ids(ledger) narrative = " ".join( [ str(proposal.get("optimized_description") or ""), *[str(item) for item in proposal.get("bullets") or []], ] ) covered = [ fact_id for fact_id in required if _fact_text_is_preserved(fact_id, ledger, narrative) ] proposal["covered_fact_ids"] = covered proposal["omitted_fact_ids"] = [ fact_id for fact_id in required if fact_id not in covered ] return proposal def style_references(self, facts: list[Any], context: dict[str, Any]) -> list[Any]: """Expose non-blocking RAG style examples to the gap-analysis role.""" try: return self._retrieve(normalize_fact_ledger(facts), context) except Exception: return [] def _retrieve( self, ledger: list[Fact], context: dict[str, Any] ) -> list[dict[str, str]]: query = " ".join(item["text"] for item in ledger) if context.get("target_position"): query = f"{context['target_position']} {query}".strip() try: results = self.retriever.retrieve( query_text=query or "resume experience optimization", embedder=self.embedder, position_category=context.get("target_position") or None, exp_type=context.get("entry_type") or None, k=3, ) except Exception: return [] return [ { "id": str(item.get("id") or f"rag_{index}"), "title": str(item.get("title") or item.get("title_path") or "reference"), "content": str(item.get("content") or item.get("optimized") or ""), "writing_points": str(item.get("points") or ""), } for index, item in enumerate(results[:3], start=1) ] class FallbackExperienceOptimizer: def __init__(self, primary: ExperienceOptimizer, fallback: ExperienceOptimizer) -> None: self.primary = primary self.fallback = fallback def optimize( self, entry: dict[str, Any], *, context: dict[str, Any], facts: list[Any] ) -> dict[str, Any]: try: proposal = self.primary.optimize(entry, context=context, facts=facts) proposal["generation_source"] = "llm" return proposal except Exception as exc: reason = _fallback_reason(exc) log_ai_event( "experience_optimization_failed", reason_code=reason, trace_id=getattr(exc, "trace_id", None), stage=getattr(exc, "stage", "experience_optimization"), entry_type=str(context.get("entry_type") or ""), optimization_mode=str(context.get("optimization_mode") or "light"), exception=type(exc).__name__, ) if isinstance(exc, LLMServiceError): raise proposal = self.fallback.optimize(entry, context=context, facts=facts) proposal["generation_source"] = "rule_fallback" proposal["fallback_reason"] = reason return proposal def style_references(self, facts: list[Any], context: dict[str, Any]) -> list[Any]: provider = getattr(self.primary, "style_references", None) if provider is None: return [] try: return list(provider(facts, context) or []) except Exception: return [] class RuleStructuredExperienceOptimizer: """Offline generator for tests and explicit no-model fallback mode.""" def optimize( self, entry: dict[str, Any], *, context: dict[str, Any], facts: list[Any] ) -> dict[str, Any]: ledger = normalize_fact_ledger(facts) description = str(entry.get("description") or "").strip() material = "; ".join(dict.fromkeys(item["text"] for item in ledger)) if not material: return { "optimized_description": "", "changes": [], "missing_facts": ["specific action", "method or tool", "verifiable result"], "star": {}, "claims": [], "bullets": [], "source": "rule_structured", "generation_source": "rule", "fallback_reason": "insufficient_user_facts", } optimized = material return { "optimized_description": optimized, "bullets": [optimized], "changes": ["reorganized confirmed actions and facts"], "missing_facts": _missing_facts(material), "star": { "situation": description or None, "task": str(entry.get("title") or entry.get("position") or "") or None, "action": material, "result": _known_result(material), }, "claims": [], "source": "rule_structured", "generation_source": "rule", } def _fallback_reason(exc: Exception) -> str: if isinstance(exc, LLMServiceError): return exc.reason_code return type(exc).__name__.lower()[:48] def build_experience_optimizer( settings: Settings, client: Any | None = None ) -> ExperienceOptimizer: """Light optimizer: pure LLM on user facts (RAG knowledge base removed).""" rules = RuleStructuredExperienceOptimizer() if not settings.use_openai: return rules completion = OpenAICompatibleStructuredClient(settings, client) primary: ExperienceOptimizer = OpenAIExperienceOptimizer(completion) return FallbackExperienceOptimizer(primary, rules) if settings.fallback_to_rules else primary def normalize_fact_ledger(facts: list[Any]) -> list[Fact]: ledger: list[Fact] = [] for index, value in enumerate(facts, start=1): if isinstance(value, dict): text = str(value.get("text") or "").strip() fact = { "id": str(value.get("id") or f"fact_{index}"), "source": str(value.get("source") or "user_form"), "field": str(value.get("field") or "unknown"), "text": text, } else: text = str(value).strip() fact = { "id": f"fact_{index}", "source": "user_form", "field": "unknown", "text": text, } if text: ledger.append(fact) return _append_description_parts(ledger) _DESCRIPTION_ITEM_MARKER = re.compile(r"^\s*\d+\s*[.、))]\s*") _DESCRIPTION_LINE_LABEL = re.compile(r"^[\u4e00-\u9fff]{2,8}[::]\s*") def split_description_parts(text: str) -> list[str]: """Split a structured description into independently checkable fragments. A long multi-line description judged as one fact lets dropped features hide behind the overall n-gram coverage of the kept tech stack. Line/clause fragments make each feature, intro, or outcome its own gate entry. Short single-sentence descriptions stay unsplit (one fragment -> caller keeps the parent fact). """ parts: list[str] = [] for raw_line in str(text).splitlines(): line = raw_line.strip() if not line: continue segments = re.split(r"[。;;]", line) if len(line) > 40 else [line] for segment in segments: part = _DESCRIPTION_LINE_LABEL.sub("", _DESCRIPTION_ITEM_MARKER.sub("", segment.strip())).strip() if len(part) >= 4: parts.append(part) return parts def _append_description_parts(ledger: list[Fact]) -> list[Fact]: expanded: list[Fact] = [] for fact in ledger: expanded.append(fact) if fact.get("field") != "description": continue parts = split_description_parts(fact["text"]) if len(parts) < 2: continue for part_index, part in enumerate(parts, start=1): expanded.append( { "id": f"{fact['id']}_part_{part_index}", "source": fact.get("source") or "user_form", "field": "description_part", "text": part, } ) return expanded def required_material_fact_ids(ledger: list[Fact]) -> list[str]: """Ids of facts that must survive in the narrative. When a description was split into fragments, the fragments stand in for the parent so coverage is judged per fragment, not per whole entry. """ split_parents = { fact["id"].rsplit("_part_", 1)[0] for fact in ledger if fact.get("field") == "description_part" } return [ fact["id"] for fact in ledger if _is_material_resume_fact(fact) and fact["id"] not in split_parents ] def _is_material_resume_fact(fact: Fact) -> bool: """Facts that belong in the narrative rather than only in card metadata.""" if fact.get("field") in { "degree", "start_date", "end_date_or_present", "date", "title", "name", "company", "organization", "school", "project_name", "position", "role", "project_role", "major", "award", }: return False return bool(str(fact.get("text") or "").strip()) and ( fact.get("field") in {"description", "description_part"} or fact.get("source") == "user_answer" ) def _fact_text_is_preserved(fact_id: str, ledger: list[Fact], narrative: str) -> bool: fact = next((item for item in ledger if item["id"] == fact_id), None) if fact is None: return False source = _normalized_text(fact["text"]) target = _normalized_text(narrative) if not source or not target: return False if source in target: return True latin_terms = [ _latin_stem(term) for term in re.findall(r"[A-Za-z][A-Za-z0-9+#._-]{1,}", fact["text"]) if term.casefold() not in _LATIN_STOPWORDS ] target_latin_terms = { _latin_stem(term) for term in re.findall(r"[A-Za-z][A-Za-z0-9+#._-]{1,}", narrative) if term.casefold() not in _LATIN_STOPWORDS } latin_covered = not latin_terms or sum( term in target_latin_terms for term in latin_terms ) >= max(1, int(len(latin_terms) * 0.6 + 0.999)) if not latin_covered: return False chinese_segments = re.findall(r"[\u4e00-\u9fff]{2,}", fact["text"]) chinese_ngrams = { segment[index:index + size] for segment in chinese_segments for size in (2, 3, 4) for index in range(max(0, len(segment) - size + 1)) } matched_ngrams = sum( _normalized_text(token) in target for token in chinese_ngrams ) chinese_covered = not chinese_ngrams or matched_ngrams >= max( 1, int(len(chinese_ngrams) * 0.65) ) number_tokens = re.findall(r"\d+(?:\.\d+)?%?", fact["text"]) numbers_covered = all(token in narrative for token in number_tokens) return chinese_covered and numbers_covered and bool( latin_terms or chinese_ngrams or number_tokens ) _LATIN_STOPWORDS = frozenset({ "and", "are", "for", "from", "into", "its", "that", "the", "this", "through", "using", "with", }) def _latin_stem(term: str) -> str: normalized = term.casefold().rstrip(".,;:!?") if normalized == "built": return "build" if normalized == "ran": return "run" for suffix in ("ing", "ed", "es", "s"): if normalized.endswith(suffix) and len(normalized) - len(suffix) >= 4: return normalized[:-len(suffix)] return normalized def _normalized_text(text: str) -> str: return "".join(character.casefold() for character in text if character.isalnum()) def _missing_facts(text: str) -> list[str]: missing: list[str] = [] if not any(token in text for token in ("%", "result", "impact", "users")): missing.append("verifiable result or impact") if not any(token in text.casefold() for token in ("using", "with", "through", "via")): missing.append("method, tool, or collaboration approach") return missing or ["scope of responsibility"] def _known_result(text: str) -> str | None: markers = ("%", "improved", "reduced", "completed", "launched", "users") return text if any(marker in text.casefold() for marker in markers) else None def _canonical_fact_draft(ledger: list[Fact]) -> str: return "; ".join(dict.fromkeys(item["text"] for item in ledger if item.get("text"))) _OPTIMIZATION_PROMPT = """ 你是一名中文简历经历编辑。只返回符合 output_json_schema 的 JSON。 请将用户提供的经历改写为专业、可直接用于简历的中文正文,并采用自然的 STAR 结构。完整性优先于篇幅:保留用户已确认的职责、动作、方法、工具、协作、范围、 交付物和结果。可以重排和合并真正重复的措辞,但不得为了缩短文本删除有意义的事实; 必要时可使用多句或多条要点。项目或产品的功能模块、平台定位/简介与量化成果, 与技术栈同等重要:不得只保留技术栈而省略功能点、平台简介或成果描述。 deep_interview.completion.is_sufficient 为 true 时,表示 LangGraph 已确认当前候选稿 所需信息足够。此时不得把任何已回答维度重新列为 missing_facts,也不要把泛泛的 “补充技术栈、量化结果或职责”当作当前候选稿的阻塞条件。若存在不影响当前候选稿的 提升方向,只能放入 optional_enhancements,且要明确是可选增强。 style_references 只用于学习表达方式,不是用户个人事实。不得虚构公司、学校、 奖项、证书、日期或归属。可以基于用户的经历语义做自然的职业化改写、结构化归纳和适度的 岗位导向扩展;不要因为原文没有逐字写出某个方法或影响就机械省略整段内容。量化表达 应优先使用用户确认的数据;未确认时可以使用不带精确数字的合理影响描述。每条 claim 必须只引用 user_fact_ledger 中的 evidence_ids,绝不能引用 rag_ IDs。 """.strip() _OPTIMIZATION_REPAIR_PROMPT = """ 你负责修复一份未通过确定性校验的中文简历候选稿。只返回符合 output_json_schema 的 JSON。rejected_candidate 和 rejected_reason 只说明缺陷,不是新的事实来源。 输出非空、专业、可直接用于简历的中文叙述,采用自然的 STAR 结构。修复的首要目标是保留 required_fact_ids 对应的全部重要事实,包括原描述和深度追问答案中的动作、方法、工具、范围、协作、 交付物和结果。不要为了简洁而压缩掉这些信息;可使用多句或多条要点,只合并语义重复 的表达。允许自然的同义改写、结构化归纳和岗位导向扩展,不要求逐字复述每项事实。 不得悄然编造公司、学校、奖项、证书、日期或归属。非阻塞的后续提升方向放入 optional_enhancements。每条 claim 必须引用已有 user_fact_ledger evidence_ids。不要 返回只有标点的文本或原始表单字段拼接。 """.strip()