Files
resume-agent/backend/app/services.py
T
hypandClaude ae2d9b128d feat: builder 简历生成 + 轻度优化 + 简历导入交付副本
自内部仓库剥离深度优化与 RAG 知识库后的交付版本:
- Builder 对话式简历生成(FSM + 意图路由 LLM 兜底增强)
- 条目级轻度优化:事实覆盖门禁 + STAR/bullet 修复链,功能/简介/成果与技术栈同级保护
- 简历导入:DOCX/PDF 解析、结构归一、手机号脱敏
- PostgreSQL 运行时 + Alembic 迁移链

Co-Authored-By: Claude <noreply@anthropic.com>
2026-08-05 11:08:30 +08:00

286 lines
10 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
from __future__ import annotations
import re
from dataclasses import asdict, dataclass
from typing import Any, Protocol
from .entry_expander import EntryExpander, RuleBasedEntryExpander
from .skill_classifier import classify_skills
@dataclass(slots=True)
class ExtractedExperience:
raw_text: str
title: str
organization: str | None
role: str | None
highlights: list[str]
metrics: list[str]
confidence: float
def to_dict(self) -> dict[str, Any]:
return asdict(self)
class ExperienceExtractor(Protocol):
"""Replacement seam for an LLM or another structured extractor."""
def extract(self, text: str) -> ExtractedExperience: ...
def extract_anchor(
self,
text: str,
anchor_type: str,
missing_fields: list[str],
) -> dict[str, str]: ...
class ResumeRewriter(Protocol):
"""Replacement seam for an LLM-backed resume renderer."""
def rewrite(self, profile: dict[str, Any]) -> dict[str, Any]: ...
class RuleBasedExperienceExtractor:
_metric_pattern = re.compile(
r"(?:\d+(?:\.\d+)?\s*(?:%|倍|万|千|人|项|个|天|小时|ms|s))",
re.IGNORECASE,
)
_organization_patterns = (
re.compile(
r"(?:在|就职于|任职于)\s*([\w\u4e00-\u9fff·.-]{2,30}?)(?=担任||,|。|$)"
),
re.compile(r"(?:at|for)\s+([A-Z][\w& .-]{1,40})", re.IGNORECASE),
)
_role_patterns = (
re.compile(r"(?:担任|职位是|任)\s*([\w\u4e00-\u9fff·.-]{2,24})"),
re.compile(r"(?:as|role:?\s*)\s+(?:an?\s+)?([\w /-]{2,32})", re.IGNORECASE),
)
_month_pattern = re.compile(
r"(?P<year>(?:19|20)\d{2})[年./-](?P<month>1[0-2]|0?[1-9])月?"
)
def extract(self, text: str) -> ExtractedExperience:
normalized = " ".join(text.split())
organization = self._first_match(self._organization_patterns, normalized)
role = self._first_match(self._role_patterns, normalized)
metrics = list(dict.fromkeys(self._metric_pattern.findall(normalized)))
highlights = [
part.strip(" ,。.;")
for part in re.split(r"[。;;\n]+", normalized)
if part.strip(" ,。.;")
][:5]
title = role or organization or (highlights[0][:32] if highlights else "补充经历")
evidence = sum(bool(value) for value in (organization, role, metrics, highlights))
confidence = min(0.95, 0.35 + evidence * 0.15)
return ExtractedExperience(
raw_text=normalized,
title=title,
organization=organization,
role=role,
highlights=highlights,
metrics=metrics,
confidence=round(confidence, 2),
)
def extract_anchor(
self,
text: str,
anchor_type: str,
missing_fields: list[str],
) -> dict[str, str]:
"""Extract only facts explicitly present in the current user message.
This deterministic implementation keeps the local MVP runnable. A model-backed
adapter can replace it without changing the FSM or gate rules.
"""
normalized = " ".join(text.split())
patch: dict[str, str] = {}
if anchor_type == "education":
self._assign_match(
patch,
"school",
normalized,
(
re.compile(r"(?:就读于|毕业于|学校(?:是|为|[:])?)\s*([^,。;;\s]{2,40})"),
re.compile(r"([\w\u4e00-\u9fff·.-]{2,32}(?:大学|学院|学校))"),
),
)
self._assign_match(
patch,
"major",
normalized,
(
re.compile(r"(?:主修|专业(?:是|为|[:])?)\s*([^,。;;\s]{2,32}?)(?:专业)?(?=[,。;;\s]|$)"),
),
)
for degree in ("博士", "硕士", "本科", "大专", "专科", "高中"):
if degree in normalized:
patch["degree"] = "大专" if degree == "专科" else degree
break
elif anchor_type in {"work_experience", "internship_experience"}:
self._assign_match(
patch,
"company",
normalized,
(
re.compile(r"(?:就职于|任职于|公司(?:是|为|[:])?)\s*([^,。;;\s]{2,40})"),
re.compile(r"(?:在)\s*([^,。;;]{2,40}?(?:公司|集团|科技|银行|事务所))"),
),
)
self._assign_match(
patch,
"position",
normalized,
(
re.compile(r"(?:担任|职位(?:是|为|[:])?|任职为)\s*([^,。;;\s]{2,32})"),
),
)
elif anchor_type == "project_experience":
self._assign_match(
patch,
"project_name",
normalized,
(
re.compile(r"(?:项目名(?:是|为|[:])?|参与(?:了)?)\s*([^,。;;\s]{2,40}?)(?:项目)?(?=[,。;;\s]|$)"),
),
)
self._assign_match(
patch,
"project_role",
normalized,
(
re.compile(r"(?:项目角色(?:是|为|[:])?|担任)\s*([^,。;;\s]{2,32})"),
),
)
months = [
f"{match.group('year')}-{int(match.group('month')):02d}"
for match in self._month_pattern.finditer(normalized)
]
if months:
patch["start_date"] = months[0]
if len(months) > 1:
patch["end_date_or_present"] = months[1]
elif "至今" in normalized or "现在" in normalized:
patch["end_date_or_present"] = "present"
# Short direct replies are useful after a targeted question. Do not treat a
# full narrative as a field value when no explicit pattern matched.
if not patch and len(normalized) <= 40 and not re.search(r"[,。;;]", normalized):
target = next(
(
field
for field in missing_fields
if field not in {"degree", "start_date", "end_date_or_present"}
),
None,
)
if target:
patch[target] = normalized
return patch
@staticmethod
def _assign_match(
patch: dict[str, str],
field: str,
text: str,
patterns: tuple[re.Pattern[str], ...],
) -> None:
value = RuleBasedExperienceExtractor._first_match(patterns, text)
if value:
patch[field] = value
@staticmethod
def _first_match(patterns: tuple[re.Pattern[str], ...], text: str) -> str | None:
for pattern in patterns:
match = pattern.search(text)
if match:
return match.group(1).strip()
return None
class RuleBasedResumeRewriter:
def rewrite(self, profile: dict[str, Any]) -> dict[str, Any]:
phone = profile.get("phone")
masked_phone = f"{phone[:3]}****{phone[-4:]}" if phone else None
anchor = profile.get("anchor", {})
anchor_type = profile.get("anchor_type")
sections: list[dict[str, Any]] = []
if anchor:
sections.append(
{
"kind": anchor_type,
"heading": self._heading(anchor_type),
"items": [anchor],
}
)
experiences = profile.get("experiences", [])
if experiences:
sections.append(
{
"kind": "additional_experience",
"heading": "补充经历",
"items": experiences,
}
)
records = profile.get("records") or {}
for kind in (
"work_experience",
"internship_experience",
"project_experience",
"education",
"campus_experience",
"competition",
):
items = [r for r in records.get(kind, []) if r.get("rewrite_confirmed")]
if items:
sections.append(
{"kind": kind, "heading": self._heading(kind), "items": items}
)
tags = profile.get("tags") or {}
skills = [str(value).strip() for value in tags.get("skills") or [] if str(value).strip()]
skill_groups = classify_skills(skills)
certificates = tags.get("certificates") or []
if certificates:
sections.append(
{
"kind": "certificates",
"heading": self._heading("certificates"),
"items": [{"value": value} for value in certificates],
}
)
basics = {
"name": profile.get("name"),
"masked_phone": masked_phone,
"phone_source": profile.get("phone_source"),
"email": profile.get("email"),
"city": profile.get("city"),
"portfolio_url": profile.get("portfolio_url"),
}
basics = {key: value for key, value in basics.items() if value is not None}
return {
"schema_version": 3,
"basics": basics,
"target": {
"job_type": profile.get("job_type"),
"position": profile.get("target_position"),
},
"sections": sections,
"skill_groups": skill_groups,
}
@staticmethod
def _heading(anchor_type: str | None) -> str:
return {
"education": "教育经历",
"work_experience": "工作经历",
"internship_experience": "实习经历",
"project_experience": "项目经历",
"campus_experience": "校园经历",
"competition": "竞赛获奖",
"skills": "技能",
"certificates": "证书",
}.get(anchor_type, "核心经历")