Files
resume-agent/backend/app/entry_expander.py
T
hypandClaude ae2d9b128d feat: builder 简历生成 + 轻度优化 + 简历导入交付副本
自内部仓库剥离深度优化与 RAG 知识库后的交付版本:
- Builder 对话式简历生成(FSM + 意图路由 LLM 兜底增强)
- 条目级轻度优化:事实覆盖门禁 + STAR/bullet 修复链,功能/简介/成果与技术栈同级保护
- 简历导入:DOCX/PDF 解析、结构归一、手机号脱敏
- PostgreSQL 运行时 + Alembic 迁移链

Co-Authored-By: Claude <noreply@anthropic.com>
2026-08-05 11:08:30 +08:00

87 lines
3.4 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Injectable entry expansion protocol and deterministic P0 implementation."""
from __future__ import annotations
import re
from typing import Any, Protocol
class EntryExpander(Protocol):
"""Produce an optimization proposal without mutating the source entry."""
def expand(self, entry: dict[str, Any], *, context: dict[str, Any]) -> dict[str, Any]: ...
class RuleBasedEntryExpander:
"""Conservative local fallback used when no model is configured or available."""
def expand(self, entry: dict[str, Any], *, context: dict[str, Any]) -> dict[str, Any]:
description = str(entry.get("description") or "").strip()
highlights = [
str(value).strip()
for value in entry.get("highlights") or []
if str(value).strip()
]
material = description or "".join(highlights)
if material:
optimized = _polish_text(material)
else:
optimized = _description_from_structured_facts(entry, str(context.get("entry_type") or ""))
if not optimized:
return {"optimized_description": "", "changes": [], "source": "rule_polish"}
changes = ["统一为简洁、正式的简历表达"]
if not description and not highlights:
changes = ["根据已填写的结构化事实补充经历描述"]
return {
"optimized_description": optimized,
"changes": changes,
"source": "rule_polish",
}
def _polish_text(text: str) -> str:
replacements = (
(r"^做过", "完成"),
(r"^做了", "完成"),
(r"^拿了奖(?:项)?", "获得奖项"),
(r"^参与了", "参与"),
(r"^使用了?\s*(?=[A-Za-z0-9])", "基于 "),
(r"^帮忙", "协助"),
(r"^负责", "承担"),
(r"^参加", "参与"),
(r",将", ",推动"),
(r",获得", ",并获得"),
(r"降低了", "降低"),
(r"提升了", "提升"),
(r"优化了", "优化"),
)
parts: list[str] = []
for raw in re.split(r"[。;;\n]+", text):
part = raw.strip(" ,。;;")
if not part:
continue
for pattern, replacement in replacements:
part = re.sub(pattern, replacement, part)
parts.append(part)
return "".join(parts[:5]) + ("。" if parts else "")
def _description_from_structured_facts(entry: dict[str, Any], entry_type: str) -> str:
if entry_type in {"work_experience", "internship_experience"}:
company = str(entry.get("company") or "").strip()
position = str(entry.get("position") or "").strip()
return f"在{company}担任{position}。" if company and position else ""
if entry_type == "project_experience":
name = str(entry.get("project_name") or "").strip()
role = str(entry.get("project_role") or "").strip()
return f"参与{name},担任{role}。" if name and role else ""
if entry_type == "competition":
name = str(entry.get("name") or "").strip()
award = str(entry.get("award") or "").strip()
return f"参加{name}并获得{award}。" if name and award else ""
if entry_type == "campus_experience":
organization = str(entry.get("organization") or "").strip()
role = str(entry.get("role") or "").strip()
return f"在{organization}担任{role}。" if organization and role else ""
return ""