Publiée le {published_at} Actualisée le {updated_at}
+Data Engineer F/H
+Publiée le 20/04/2026 Actualisée le 02/06/2026
+Descriptif du poste
+Build pipelines
+Profil recherché
+Python / SQL
+diff --git a/src/job_research/apec/adapter.py b/src/job_research/apec/adapter.py index 341f0f3..89e1514 100644 --- a/src/job_research/apec/adapter.py +++ b/src/job_research/apec/adapter.py @@ -12,6 +12,7 @@ from playwright.sync_api import sync_playwright _SEARCH_URL = "https://www.apec.fr/candidat/recherche-emploi.html/emploi" _RESULT_LINK_SELECTOR = "a[href*='/candidat/recherche-emploi.html/emploi/detail-offre/']" _DETAIL_JOB_ID_PATTERN = re.compile(r"/detail-offre/([^/?#]+)") +_APEC_HOSTS = {"apec.fr", "www.apec.fr"} @dataclass(slots=True) @@ -55,6 +56,16 @@ def _accept_cookies_if_present(page) -> None: return +def _is_public_apec_detail_url(url: str) -> bool: + parsed_url = urlparse(url) + return ( + parsed_url.scheme == "https" + and parsed_url.hostname in _APEC_HOSTS + and re.fullmatch(r"/candidat/recherche-emploi\.html/emploi/detail-offre/[^/?#]+", parsed_url.path) + is not None + ) + + class ApecAdapter: def __init__(self, max_listings: int = 50) -> None: self.max_listings = max_listings @@ -100,7 +111,6 @@ class ApecAdapter: if not hrefs: break - new_results_on_page = 0 for href in hrefs: source_job_id = _extract_source_job_id(href) dedupe_key = source_job_id or href @@ -109,23 +119,21 @@ class ApecAdapter: seen_keys.add(dedupe_key) results.append(ApecSearchResult(url=href, source_job_id=source_job_id)) - new_results_on_page += 1 if len(results) >= self.max_listings: break - if new_results_on_page == 0: - break - page_number += 1 return results def fetch_listing_html(self, url: str) -> str: - parsed_url = urlparse(url) - if not parsed_url.netloc.endswith("apec.fr"): + if not _is_public_apec_detail_url(url): raise ValueError("ApecAdapter only fetches public Apec URLs") with _open_public_page() as page: page.goto(url, wait_until="domcontentloaded") + page.wait_for_selector(".card-offer .ref-offre", timeout=15_000) + page.wait_for_selector(".details-offer-list", timeout=15_000) + page.wait_for_selector(".details-post", timeout=15_000) return page.content() diff --git a/src/job_research/apec/normalize.py b/src/job_research/apec/normalize.py index b503798..4e7239a 100644 --- a/src/job_research/apec/normalize.py +++ b/src/job_research/apec/normalize.py @@ -1,8 +1,97 @@ +from __future__ import annotations + +import re +from datetime import datetime + from bs4 import BeautifulSoup from job_research.models import ApecListing +_PUBLISHED_AT_PATTERN = re.compile(r"Publi[ée]e le (\d{2}/\d{2}/\d{4})") +_SOURCE_JOB_ID_PATTERN = re.compile(r"Ref\. Apec\s*:\s*([A-Z0-9]+)") +_CONTRACT_PATTERN = re.compile(r"\b(CDI|CDD|Alternance|Intérim|Stage|Freelance|Indépendant)\b") + + +def _clean_text(value: str | None) -> str | None: + if value is None: + return None + + cleaned = " ".join(value.split()) + return cleaned or None + + +def _extract_block_value(block) -> str | None: + pieces: list[str] = [] + for child in block.children: + if getattr(child, "name", None) == "h4": + continue + + text = child.get_text(" ", strip=True) if hasattr(child, "get_text") else str(child) + cleaned = _clean_text(text) + if cleaned: + pieces.append(cleaned) + + return _clean_text(" ".join(pieces)) + + +def _detail_block_text(soup: BeautifulSoup, label: str) -> str | None: + for block in soup.select(".details-post"): + heading = block.find("h4") + if heading is None: + continue + + if _clean_text(heading.get_text(" ", strip=True)) != label: + continue + + return _extract_block_value(block) + + return None + + +def _extract_published_at(soup: BeautifulSoup) -> str | None: + card_offer = soup.select_one(".card-offer") + if card_offer is None: + return None + + match = _PUBLISHED_AT_PATTERN.search(card_offer.get_text(" ", strip=True)) + if match is None: + return None + + return datetime.strptime(match.group(1), "%d/%m/%Y").date().isoformat() + + +def _extract_source_job_id(soup: BeautifulSoup, source_job_id: str | None) -> str | None: + if source_job_id is not None: + return source_job_id + + ref = soup.select_one(".ref-offre") + if ref is None: + return None + + match = _SOURCE_JOB_ID_PATTERN.search(ref.get_text(" ", strip=True)) + if match is None: + return None + + return match.group(1) + + +def _extract_contract_type(details_offer_list) -> str | None: + contract_item = details_offer_list.select_one("li:nth-of-type(2)") + if contract_item is None: + return None + + span = contract_item.find("span") + if span is not None: + return _clean_text(span.get_text(" ", strip=True)) + + match = _CONTRACT_PATTERN.search(contract_item.get_text(" ", strip=True)) + if match is None: + return None + + return match.group(1) + + def normalize_apec_listing( url: str, html: str, @@ -13,21 +102,34 @@ def normalize_apec_listing( ) -> ApecListing: soup = BeautifulSoup(html, "html.parser") - title = soup.find("h1") - company = soup.select_one(".company") - location = soup.select_one(".location") - contract = soup.select_one(".contract") - description = soup.select_one(".description") + title = soup.select_one(".container-details-offer h1") or soup.find("h1") + details_offer_list = soup.select_one(".details-offer-list") + + company = soup.select_one(".card-ents .ents-name") + if company is None and details_offer_list is not None: + company = details_offer_list.select_one("li:nth-of-type(1)") + + location = None + contract_type = None + if details_offer_list is not None: + location = details_offer_list.select_one("li:nth-of-type(3)") + contract_type = _extract_contract_type(details_offer_list) + + description_sections = [ + _detail_block_text(soup, "Descriptif du poste"), + _detail_block_text(soup, "Profil recherché"), + ] + description_text = "\n\n".join(section for section in description_sections if section) return ApecListing( source="apec", - source_job_id=source_job_id, + source_job_id=_extract_source_job_id(soup, source_job_id), url=url, - title=title.get_text(" ", strip=True) if title else None, - company=company.get_text(" ", strip=True) if company else None, - location=location.get_text(" ", strip=True) if location else None, - contract_type=contract.get_text(" ", strip=True) if contract else None, - description_text=description.get_text(" ", strip=True) if description else None, - published_at=published_at, + title=_clean_text(title.get_text(" ", strip=True)) if title else None, + company=_clean_text(company.get_text(" ", strip=True)) if company else None, + location=_clean_text(location.get_text(" ", strip=True)) if location else None, + contract_type=contract_type, + description_text=description_text or None, + published_at=published_at or _extract_published_at(soup), fetched_at=fetched_at, ) diff --git a/tests/apec/test_adapter.py b/tests/apec/test_adapter.py new file mode 100644 index 0000000..e82f54b --- /dev/null +++ b/tests/apec/test_adapter.py @@ -0,0 +1,137 @@ +from contextlib import contextmanager +from urllib.parse import parse_qs, quote_plus, urlparse + +import pytest +from playwright.sync_api import TimeoutError as PlaywrightTimeoutError + +from job_research.apec import adapter as adapter_module +from job_research.apec.adapter import ApecAdapter + + +_RESULT_LINK_SELECTOR = "a[href*='/candidat/recherche-emploi.html/emploi/detail-offre/']" + + +class _FakeResultButton: + def __init__(self, page, name: str) -> None: + self.page = page + self.name = name + + def click(self, timeout: int | None = None) -> None: + if self.name == "Rechercher": + self.page.url = ( + "https://www.apec.fr/candidat/recherche-emploi.html/emploi" + f"?motsCles={quote_plus(self.page.current_query)}&page=0" + ) + self.page.current_page = 0 + + +class _FakeLocator: + def __init__(self, page, selector: str) -> None: + self.page = page + self.selector = selector + + def fill(self, value: str) -> None: + self.page.current_query = value + + def evaluate_all(self, function: str): + if self.selector == _RESULT_LINK_SELECTOR: + return list(self.page.current_results()) + + return [] + + +class _FakeDetailPage: + def __init__(self, result_pages: dict[str, dict[int, list[str]]], *, rendered_html: str = "rendered") -> None: + self.result_pages = result_pages + self.rendered_html = rendered_html + self.shell_html = "shell" + self.waited_selectors: list[str] = [] + self.goto_urls: list[str] = [] + self.current_query = "" + self.current_page = 0 + self.url = "" + self.rendered = False + + def goto(self, url: str, wait_until: str | None = None) -> None: + self.goto_urls.append(url) + self.url = url + + parsed_url = urlparse(url) + params = parse_qs(parsed_url.query) + if "motsCles" in params: + self.current_query = params["motsCles"][0] + if "page" in params: + self.current_page = int(params["page"][0]) + + if "/detail-offre/" in parsed_url.path: + self.rendered = False + + def wait_for_load_state(self, state: str) -> None: + return None + + def wait_for_selector(self, selector: str, timeout: int | None = None) -> None: + self.waited_selectors.append(selector) + + if selector in {".card-offer .ref-offre", ".details-offer-list", ".details-post"}: + self.rendered = True + return None + + if selector == _RESULT_LINK_SELECTOR and self.current_results(): + return None + + raise PlaywrightTimeoutError(f"selector not found: {selector}") + + def get_by_role(self, role: str, name: str): + return _FakeResultButton(self, name) + + def locator(self, selector: str): + return _FakeLocator(self, selector) + + def content(self) -> str: + return self.rendered_html if self.rendered else self.shell_html + + def current_results(self) -> list[str]: + return self.result_pages.get(self.current_query, {}).get(self.current_page, []) + + +@contextmanager +def _fake_open_public_page(page: _FakeDetailPage): + yield page + + +def test_search_continues_past_duplicate_only_pages(monkeypatch) -> None: + first_result = "https://www.apec.fr/candidat/recherche-emploi.html/emploi/detail-offre/111?motsCles=alpha&page=0&selectedIndex=0" + second_result = "https://www.apec.fr/candidat/recherche-emploi.html/emploi/detail-offre/222?motsCles=beta&page=1&selectedIndex=0" + page = _FakeDetailPage( + { + "alpha": {0: [first_result], 1: []}, + "beta": {0: [first_result], 1: [second_result], 2: []}, + } + ) + + monkeypatch.setattr(adapter_module, "_open_public_page", lambda: _fake_open_public_page(page)) + + results = ApecAdapter(max_listings=10).search(["alpha", "beta"]) + + assert [result.url for result in results] == [first_result, second_result] + assert [result.source_job_id for result in results] == ["111", "222"] + assert any("page=1" in url for url in page.goto_urls) + + +def test_fetch_listing_html_waits_for_rendered_offer_content(monkeypatch) -> None: + page = _FakeDetailPage({}, rendered_html="rendered offer") + monkeypatch.setattr(adapter_module, "_open_public_page", lambda: _fake_open_public_page(page)) + + html = ApecAdapter().fetch_listing_html( + "https://www.apec.fr/candidat/recherche-emploi.html/emploi/detail-offre/111" + ) + + assert html == "rendered offer" + assert page.waited_selectors == [".card-offer .ref-offre", ".details-offer-list", ".details-post"] + + +def test_fetch_listing_html_rejects_non_apec_hosts() -> None: + adapter = ApecAdapter() + + with pytest.raises(ValueError): + adapter.fetch_listing_html("https://evilapec.fr/candidat/recherche-emploi.html/emploi/detail-offre/111") diff --git a/tests/apec/test_normalize.py b/tests/apec/test_normalize.py index 69a9141..c50206b 100644 --- a/tests/apec/test_normalize.py +++ b/tests/apec/test_normalize.py @@ -5,11 +5,29 @@ def test_normalize_apec_listing_extracts_minimal_shape() -> None: html = """
-Publiée le 20/04/2026 Actualisée le 02/06/2026
+Build pipelines
+Python / SQL
+Publiée le {published_at} Actualisée le {updated_at}
+{description}
+{profile}
+