generated from kgod/ai-review-template
自内部仓库剥离深度优化与 RAG 知识库后的交付版本: - Builder 对话式简历生成(FSM + 意图路由 LLM 兜底增强) - 条目级轻度优化:事实覆盖门禁 + STAR/bullet 修复链,功能/简介/成果与技术栈同级保护 - 简历导入:DOCX/PDF 解析、结构归一、手机号脱敏 - PostgreSQL 运行时 + Alembic 迁移链 Co-Authored-By: Claude <noreply@anthropic.com>
286 lines
10 KiB
Python
286 lines
10 KiB
Python
from __future__ import annotations
|
||
|
||
import re
|
||
from dataclasses import asdict, dataclass
|
||
from typing import Any, Protocol
|
||
|
||
from .entry_expander import EntryExpander, RuleBasedEntryExpander
|
||
from .skill_classifier import classify_skills
|
||
|
||
|
||
@dataclass(slots=True)
|
||
class ExtractedExperience:
|
||
raw_text: str
|
||
title: str
|
||
organization: str | None
|
||
role: str | None
|
||
highlights: list[str]
|
||
metrics: list[str]
|
||
confidence: float
|
||
|
||
def to_dict(self) -> dict[str, Any]:
|
||
return asdict(self)
|
||
|
||
|
||
class ExperienceExtractor(Protocol):
|
||
"""Replacement seam for an LLM or another structured extractor."""
|
||
|
||
def extract(self, text: str) -> ExtractedExperience: ...
|
||
|
||
def extract_anchor(
|
||
self,
|
||
text: str,
|
||
anchor_type: str,
|
||
missing_fields: list[str],
|
||
) -> dict[str, str]: ...
|
||
|
||
|
||
class ResumeRewriter(Protocol):
|
||
"""Replacement seam for an LLM-backed resume renderer."""
|
||
|
||
def rewrite(self, profile: dict[str, Any]) -> dict[str, Any]: ...
|
||
|
||
|
||
class RuleBasedExperienceExtractor:
|
||
_metric_pattern = re.compile(
|
||
r"(?:\d+(?:\.\d+)?\s*(?:%|倍|万|千|人|项|个|天|小时|ms|s))",
|
||
re.IGNORECASE,
|
||
)
|
||
_organization_patterns = (
|
||
re.compile(
|
||
r"(?:在|就职于|任职于)\s*([\w\u4e00-\u9fff·.-]{2,30}?)(?=担任|,|,|。|$)"
|
||
),
|
||
re.compile(r"(?:at|for)\s+([A-Z][\w& .-]{1,40})", re.IGNORECASE),
|
||
)
|
||
_role_patterns = (
|
||
re.compile(r"(?:担任|职位是|任)\s*([\w\u4e00-\u9fff·.-]{2,24})"),
|
||
re.compile(r"(?:as|role:?\s*)\s+(?:an?\s+)?([\w /-]{2,32})", re.IGNORECASE),
|
||
)
|
||
_month_pattern = re.compile(
|
||
r"(?P<year>(?:19|20)\d{2})[年./-](?P<month>1[0-2]|0?[1-9])月?"
|
||
)
|
||
|
||
def extract(self, text: str) -> ExtractedExperience:
|
||
normalized = " ".join(text.split())
|
||
organization = self._first_match(self._organization_patterns, normalized)
|
||
role = self._first_match(self._role_patterns, normalized)
|
||
metrics = list(dict.fromkeys(self._metric_pattern.findall(normalized)))
|
||
highlights = [
|
||
part.strip(" ,,。.;;")
|
||
for part in re.split(r"[。;;\n]+", normalized)
|
||
if part.strip(" ,,。.;;")
|
||
][:5]
|
||
title = role or organization or (highlights[0][:32] if highlights else "补充经历")
|
||
evidence = sum(bool(value) for value in (organization, role, metrics, highlights))
|
||
confidence = min(0.95, 0.35 + evidence * 0.15)
|
||
return ExtractedExperience(
|
||
raw_text=normalized,
|
||
title=title,
|
||
organization=organization,
|
||
role=role,
|
||
highlights=highlights,
|
||
metrics=metrics,
|
||
confidence=round(confidence, 2),
|
||
)
|
||
|
||
def extract_anchor(
|
||
self,
|
||
text: str,
|
||
anchor_type: str,
|
||
missing_fields: list[str],
|
||
) -> dict[str, str]:
|
||
"""Extract only facts explicitly present in the current user message.
|
||
|
||
This deterministic implementation keeps the local MVP runnable. A model-backed
|
||
adapter can replace it without changing the FSM or gate rules.
|
||
"""
|
||
normalized = " ".join(text.split())
|
||
patch: dict[str, str] = {}
|
||
|
||
if anchor_type == "education":
|
||
self._assign_match(
|
||
patch,
|
||
"school",
|
||
normalized,
|
||
(
|
||
re.compile(r"(?:就读于|毕业于|学校(?:是|为|[::])?)\s*([^,,。;;\s]{2,40})"),
|
||
re.compile(r"([\w\u4e00-\u9fff·.-]{2,32}(?:大学|学院|学校))"),
|
||
),
|
||
)
|
||
self._assign_match(
|
||
patch,
|
||
"major",
|
||
normalized,
|
||
(
|
||
re.compile(r"(?:主修|专业(?:是|为|[::])?)\s*([^,,。;;\s]{2,32}?)(?:专业)?(?=[,,。;;\s]|$)"),
|
||
),
|
||
)
|
||
for degree in ("博士", "硕士", "本科", "大专", "专科", "高中"):
|
||
if degree in normalized:
|
||
patch["degree"] = "大专" if degree == "专科" else degree
|
||
break
|
||
elif anchor_type in {"work_experience", "internship_experience"}:
|
||
self._assign_match(
|
||
patch,
|
||
"company",
|
||
normalized,
|
||
(
|
||
re.compile(r"(?:就职于|任职于|公司(?:是|为|[::])?)\s*([^,,。;;\s]{2,40})"),
|
||
re.compile(r"(?:在)\s*([^,,。;;]{2,40}?(?:公司|集团|科技|银行|事务所))"),
|
||
),
|
||
)
|
||
self._assign_match(
|
||
patch,
|
||
"position",
|
||
normalized,
|
||
(
|
||
re.compile(r"(?:担任|职位(?:是|为|[::])?|任职为)\s*([^,,。;;\s]{2,32})"),
|
||
),
|
||
)
|
||
elif anchor_type == "project_experience":
|
||
self._assign_match(
|
||
patch,
|
||
"project_name",
|
||
normalized,
|
||
(
|
||
re.compile(r"(?:项目名(?:是|为|[::])?|参与(?:了)?)\s*([^,,。;;\s]{2,40}?)(?:项目)?(?=[,,。;;\s]|$)"),
|
||
),
|
||
)
|
||
self._assign_match(
|
||
patch,
|
||
"project_role",
|
||
normalized,
|
||
(
|
||
re.compile(r"(?:项目角色(?:是|为|[::])?|担任)\s*([^,,。;;\s]{2,32})"),
|
||
),
|
||
)
|
||
|
||
months = [
|
||
f"{match.group('year')}-{int(match.group('month')):02d}"
|
||
for match in self._month_pattern.finditer(normalized)
|
||
]
|
||
if months:
|
||
patch["start_date"] = months[0]
|
||
if len(months) > 1:
|
||
patch["end_date_or_present"] = months[1]
|
||
elif "至今" in normalized or "现在" in normalized:
|
||
patch["end_date_or_present"] = "present"
|
||
|
||
# Short direct replies are useful after a targeted question. Do not treat a
|
||
# full narrative as a field value when no explicit pattern matched.
|
||
if not patch and len(normalized) <= 40 and not re.search(r"[,,。;;]", normalized):
|
||
target = next(
|
||
(
|
||
field
|
||
for field in missing_fields
|
||
if field not in {"degree", "start_date", "end_date_or_present"}
|
||
),
|
||
None,
|
||
)
|
||
if target:
|
||
patch[target] = normalized
|
||
return patch
|
||
|
||
@staticmethod
|
||
def _assign_match(
|
||
patch: dict[str, str],
|
||
field: str,
|
||
text: str,
|
||
patterns: tuple[re.Pattern[str], ...],
|
||
) -> None:
|
||
value = RuleBasedExperienceExtractor._first_match(patterns, text)
|
||
if value:
|
||
patch[field] = value
|
||
|
||
@staticmethod
|
||
def _first_match(patterns: tuple[re.Pattern[str], ...], text: str) -> str | None:
|
||
for pattern in patterns:
|
||
match = pattern.search(text)
|
||
if match:
|
||
return match.group(1).strip()
|
||
return None
|
||
|
||
|
||
class RuleBasedResumeRewriter:
|
||
def rewrite(self, profile: dict[str, Any]) -> dict[str, Any]:
|
||
phone = profile.get("phone")
|
||
masked_phone = f"{phone[:3]}****{phone[-4:]}" if phone else None
|
||
anchor = profile.get("anchor", {})
|
||
anchor_type = profile.get("anchor_type")
|
||
sections: list[dict[str, Any]] = []
|
||
if anchor:
|
||
sections.append(
|
||
{
|
||
"kind": anchor_type,
|
||
"heading": self._heading(anchor_type),
|
||
"items": [anchor],
|
||
}
|
||
)
|
||
experiences = profile.get("experiences", [])
|
||
if experiences:
|
||
sections.append(
|
||
{
|
||
"kind": "additional_experience",
|
||
"heading": "补充经历",
|
||
"items": experiences,
|
||
}
|
||
)
|
||
records = profile.get("records") or {}
|
||
for kind in (
|
||
"work_experience",
|
||
"internship_experience",
|
||
"project_experience",
|
||
"education",
|
||
"campus_experience",
|
||
"competition",
|
||
):
|
||
items = [r for r in records.get(kind, []) if r.get("rewrite_confirmed")]
|
||
if items:
|
||
sections.append(
|
||
{"kind": kind, "heading": self._heading(kind), "items": items}
|
||
)
|
||
tags = profile.get("tags") or {}
|
||
skills = [str(value).strip() for value in tags.get("skills") or [] if str(value).strip()]
|
||
skill_groups = classify_skills(skills)
|
||
certificates = tags.get("certificates") or []
|
||
if certificates:
|
||
sections.append(
|
||
{
|
||
"kind": "certificates",
|
||
"heading": self._heading("certificates"),
|
||
"items": [{"value": value} for value in certificates],
|
||
}
|
||
)
|
||
basics = {
|
||
"name": profile.get("name"),
|
||
"masked_phone": masked_phone,
|
||
"phone_source": profile.get("phone_source"),
|
||
"email": profile.get("email"),
|
||
"city": profile.get("city"),
|
||
"portfolio_url": profile.get("portfolio_url"),
|
||
}
|
||
basics = {key: value for key, value in basics.items() if value is not None}
|
||
return {
|
||
"schema_version": 3,
|
||
"basics": basics,
|
||
"target": {
|
||
"job_type": profile.get("job_type"),
|
||
"position": profile.get("target_position"),
|
||
},
|
||
"sections": sections,
|
||
"skill_groups": skill_groups,
|
||
}
|
||
|
||
@staticmethod
|
||
def _heading(anchor_type: str | None) -> str:
|
||
return {
|
||
"education": "教育经历",
|
||
"work_experience": "工作经历",
|
||
"internship_experience": "实习经历",
|
||
"project_experience": "项目经历",
|
||
"campus_experience": "校园经历",
|
||
"competition": "竞赛获奖",
|
||
"skills": "技能",
|
||
"certificates": "证书",
|
||
}.get(anchor_type, "核心经历")
|