fix: harden profile and Apec normalization warnings

This commit is contained in:
Antoine 2026-06-05 13:37:22 +02:00
parent fa2000abf0
commit ed1af201bc
8 changed files with 220 additions and 20 deletions

View File

@ -103,29 +103,34 @@ def _warning(field: str, message: str) -> ListingWarning:
return ListingWarning(field=field, message=message)
def _extract_published_at(soup: BeautifulSoup) -> str | None:
def _extract_listing_date(
soup: BeautifulSoup,
pattern: re.Pattern[str],
*,
field: str,
missing_message: str | None = None,
invalid_message: str,
warnings: list[ListingWarning],
warn_on_missing: bool,
) -> str | None:
card_offer = soup.select_one(".card-offer")
if card_offer is None:
if warn_on_missing and missing_message is not None:
warnings.append(_warning(field, missing_message))
return None
match = _PUBLISHED_AT_PATTERN.search(card_offer.get_text(" ", strip=True))
match = pattern.search(card_offer.get_text(" ", strip=True))
if match is None:
if warn_on_missing and missing_message is not None:
warnings.append(_warning(field, missing_message))
return None
return datetime.strptime(match.group(1), "%d/%m/%Y").date().isoformat()
def _extract_refreshed_at(soup: BeautifulSoup) -> str | None:
card_offer = soup.select_one(".card-offer")
if card_offer is None:
try:
return datetime.strptime(match.group(1), "%d/%m/%Y").date().isoformat()
except ValueError:
warnings.append(_warning(field, invalid_message))
return None
match = _REFRESHED_AT_PATTERN.search(card_offer.get_text(" ", strip=True))
if match is None:
return None
return datetime.strptime(match.group(1), "%d/%m/%Y").date().isoformat()
def _extract_source_job_id(soup: BeautifulSoup, source_job_id: str | None) -> str | None:
if source_job_id is not None:
@ -272,11 +277,24 @@ def normalize_apec_listing(
warnings.append(_warning("company", "Company is empty or placeholder text"))
company_text = None
published_at_value = published_at or _extract_published_at(soup)
if published_at_value is None:
warnings.append(_warning("published_at", "Published date missing from Apec listing"))
published_at_value = published_at or _extract_listing_date(
soup,
_PUBLISHED_AT_PATTERN,
field="published_at",
missing_message="Published date missing from Apec listing",
invalid_message="Published date is invalid",
warnings=warnings,
warn_on_missing=True,
)
refreshed_at_value = refreshed_at or _extract_refreshed_at(soup)
refreshed_at_value = refreshed_at or _extract_listing_date(
soup,
_REFRESHED_AT_PATTERN,
field="refreshed_at",
invalid_message="Refreshed date is invalid",
warnings=warnings,
warn_on_missing=False,
)
return ApecListing(
source="apec",

View File

@ -5,6 +5,8 @@ from pathlib import Path
from pypdf import PdfReader
from job_research.models import WarningItem
EXPERIENCE_LINE_CONNECTORS = (" at ", " chez ", " au ", " à ")
@ -150,6 +152,15 @@ YEARS_OF_EXPERIENCE_PATTERNS = (
re.compile(r"^ann[ée]es d[']exp[ée]rience\s*:\s*(\d+)\s*$", re.IGNORECASE),
)
LOW_CONFIDENCE_NAME_PATTERNS = (
re.compile(r"\b(cv|resume|curriculum vitae|profile)\b", re.IGNORECASE),
re.compile(r"[|/@]"),
re.compile(
r"\b(data engineer|software engineer|developer|analyst|scientist|consultant|architect|manager|product owner|backend|frontend|full stack)\b",
re.IGNORECASE,
),
)
def extract_pdf_text(path: Path) -> str:
reader = PdfReader(str(path))
@ -168,8 +179,17 @@ def extract_pdf_text(path: Path) -> str:
def extract_cv_signals(text: str) -> dict[str, object]:
lines = [_normalize_line(line) for line in text.splitlines()]
non_empty_lines = [line for line in lines if line]
warnings: list[WarningItem] = []
name = non_empty_lines[0] if non_empty_lines else None
if name is not None and _looks_like_low_confidence_name(name):
warnings.append(
WarningItem(
field="name",
message="First CV line looks like a header or tagline; review manually.",
)
)
location = None
languages: list[str] = []
skills: list[str] = []
@ -247,6 +267,7 @@ def extract_cv_signals(text: str) -> dict[str, object]:
"skills": skills,
"experience_entries": experience_entries,
"education_entries": education_entries,
"warnings": warnings,
}
if years_of_experience is not None:
@ -369,3 +390,14 @@ def _looks_like_experience_title(title: str) -> bool:
def _looks_like_prose_company(company: str) -> bool:
return any(pattern.search(company) for pattern in EXPERIENCE_PROSE_COMPANY_PATTERNS)
def _looks_like_low_confidence_name(name: str) -> bool:
normalized = " ".join(name.split())
if not normalized:
return True
if len(normalized.split()) > 4:
return True
return any(pattern.search(normalized) for pattern in LOW_CONFIDENCE_NAME_PATTERNS)

View File

@ -33,6 +33,7 @@ def build_candidate_profile_output(
warnings: list[WarningItem] = []
_append_years_of_experience_warning(cv_signals, authored.notes, warnings)
_append_cv_extraction_warnings(cv_signals, warnings)
_append_missing_cv_fact_warnings(cv_signals, warnings)
merged_skills: list[str] = []
@ -99,6 +100,15 @@ def _append_missing_cv_fact_warnings(
warnings.append(WarningItem(field=field, message=message))
def _append_cv_extraction_warnings(
cv_signals: dict[str, object], warnings: list[WarningItem]
) -> None:
for warning in cv_signals.get("warnings") or []:
warnings.append(
warning if isinstance(warning, WarningItem) else WarningItem.model_validate(warning)
)
def _note_years_of_experience(note: str) -> int | None:
normalized = note.casefold().replace("", "'")
if not any(marker in normalized for marker in EXPERIENCE_NOTE_MARKERS):

View File

@ -17,14 +17,15 @@ class AuthoredProfile:
notes: list[str] = field(default_factory=list)
REQUIRED_SECTIONS = {
REQUIRED_SECTION_NAMES = (
"summary",
"target roles",
"strengths",
"skills to emphasize",
"constraints",
"notes",
}
)
REQUIRED_SECTIONS = set(REQUIRED_SECTION_NAMES)
def parse_profile_markdown(markdown: str) -> AuthoredProfile:
@ -45,6 +46,10 @@ def parse_profile_markdown(markdown: str) -> AuthoredProfile:
missing_text = ", ".join(sorted(missing))
raise ValueError(f"Missing required markdown sections: {missing_text}")
for section_name in REQUIRED_SECTION_NAMES:
if not _has_usable_section_content(sections[section_name]):
raise ValueError(f"Missing usable content in section '{section_name}'")
return AuthoredProfile(
summary=" ".join(sections["summary"]),
target_roles=_parse_list_section("target roles", sections["target roles"]),
@ -64,6 +69,8 @@ def _parse_list_section(section_name: str, lines: list[str]) -> list[str]:
item = _strip_list_marker(line)
if item is None:
raise ValueError(f"Unsupported content in section '{section_name}': {line}")
if not item:
raise ValueError(f"Missing usable content in section '{section_name}'")
items.append(item)
return items
@ -74,6 +81,8 @@ def _parse_notes_section(lines: list[str]) -> list[str]:
for line in lines:
item = _strip_list_marker(line)
if item == "":
raise ValueError("Missing usable content in section 'notes'")
notes.append(item if item is not None else line)
return notes
@ -81,7 +90,13 @@ def _parse_notes_section(lines: list[str]) -> list[str]:
def _strip_list_marker(line: str) -> str | None:
for marker in LIST_MARKERS:
if line == marker.strip():
return ""
if line.startswith(marker):
return line[len(marker):].strip()
return None
def _has_usable_section_content(lines: list[str]) -> bool:
return any(line not in {"-", "*", "+"} for line in lines)

View File

@ -71,6 +71,47 @@ def test_normalize_apec_listing_extracts_minimal_shape() -> None:
assert listing.fetched_at == "2026-06-01T10:00:00Z"
def test_normalize_apec_listing_warns_and_returns_none_for_invalid_dates() -> None:
html = """
<html>
<body>
<main class="container-details-offer">
<h1>Data Engineer F/H</h1>
<div class="card-offer">
<div class="ref-offre">Ref. Apec : 178554452W</div>
<ul class="details-offer-list mb-20">
<li>CLOUD TEMPLE</li>
<li>1 <span> CDI </span></li>
<li>Puteaux - 92</li>
</ul>
<p>Publiée le 32/13/2026 Actualisée le 31/02/2026</p>
</div>
<article class="card card-ents mb-20">
<div class="list-hzt mb-20">
<span class="ents-name">CLOUD TEMPLE</span>
</div>
</article>
<div class="details-post">
<h4>Descriptif du poste</h4>
<p>Build pipelines</p>
</div>
</main>
</body>
</html>
"""
listing = normalize_apec_listing(
url="https://example.test/job/123",
html=html,
fetched_at="2026-06-01T10:00:00Z",
source_job_id="job-123",
)
assert listing.published_at is None
assert listing.refreshed_at is None
assert [warning.field for warning in listing.warnings] == ["published_at", "refreshed_at"]
def test_normalize_apec_listing_uses_details_offer_list_company_fallback() -> None:
html = """
<html>

View File

@ -29,6 +29,23 @@ def test_extract_cv_signals_reads_basic_fields_from_text() -> None:
assert len(extracted["experience_entries"]) == 2
def test_extract_cv_signals_flags_low_confidence_first_line_as_name() -> None:
text = dedent(
"""
Data Engineer | Python | GCP
Location: France
Languages: French, English
Skills: Python, SQL
Data Engineer at Company A
"""
).strip()
extracted = extract_cv_signals(text)
assert extracted["name"] == "Data Engineer | Python | GCP"
assert [warning.field for warning in extracted["warnings"]] == ["name"]
@pytest.mark.parametrize(
("line", "expected"),
[

View File

@ -1,5 +1,6 @@
from job_research.profile.merge import build_candidate_profile_output
from job_research.profile.profile_parser import AuthoredProfile
from job_research.models import WarningItem
def test_build_candidate_profile_output_writes_warning_when_facts_conflict() -> None:
@ -72,3 +73,34 @@ def test_build_candidate_profile_output_warns_on_missing_core_cv_facts() -> None
"skills",
"education_entries",
]
def test_build_candidate_profile_output_propagates_cv_extraction_warnings() -> None:
cv_signals = {
"name": "Data Engineer | Python | GCP",
"location": "France",
"languages": ["French", "English"],
"skills": ["Python", "SQL"],
"experience_entries": [{"title": "Data Engineer", "company": "A"}],
"education_entries": [{"credential": "MSc", "institution": "Example University"}],
"warnings": [
WarningItem(
field="name",
message="First CV line looks like a header or tagline; review manually.",
)
],
}
authored = AuthoredProfile(
summary="Junior data engineer focused on GCP.",
target_roles=["Data Engineer"],
strengths=["Python"],
skills_to_emphasize=["BigQuery", "GCP"],
constraints=["CDI only"],
notes=[],
)
output = build_candidate_profile_output(cv_signals, authored)
assert output.warnings == [
WarningItem(field="name", message="First CV line looks like a header or tagline; review manually.")
]

View File

@ -72,3 +72,38 @@ def test_parse_profile_markdown_rejects_unsupported_list_content() -> None:
with pytest.raises(ValueError, match="Unsupported content in section 'target roles'"):
parse_profile_markdown(markdown)
@pytest.mark.parametrize("section_name", ["Target Roles", "Notes"])
def test_parse_profile_markdown_rejects_blank_bullet_only_required_sections(
section_name: str,
) -> None:
target_roles = "- " if section_name == "Target Roles" else "- Data Engineer"
notes = "- " if section_name == "Notes" else "Slight preference for French listings."
markdown = dedent(
f"""
# Candidate Profile
## Summary
Junior data engineer focused on Python and GCP.
## Target Roles
{target_roles}
## Strengths
- Python
## Skills To Emphasize
- BigQuery
## Constraints
- CDI only
## Notes
{notes}
"""
).strip()
with pytest.raises(ValueError, match=f"Missing usable content in section '{section_name.lower()}'"):
parse_profile_markdown(markdown)