booleanbeyond/jobfetch
0
1"""End-to-end extraction tests against local fixture pages."""2 3from __future__ import annotations4 5import pytest6 7from app.extract import ats as ats_mod8from app.pipeline import scrape9from app.vocab import is_nav_text10 11NAV_WORDS = {12 "careers", "jobs", "jobs at acme", "view all jobs", "see all openings",13 "life at acme", "benefits", "privacy policy", "terms of service",14 "cookie policy", "linkedin", "twitter", "glassdoor", "about us", "engineering",15 "design", "sales", "operations", "job alerts", "all vacancies", "open positions",16}17 18 19def titles(result) -> set[str]:20 return {j.title.strip().lower() for j in result.jobs}21 22 23def assert_no_chrome(result):24 """The core regression guard: nothing that is site furniture may appear."""25 for job in result.jobs:26 low = job.title.strip().lower()27 assert low not in NAV_WORDS, f"navigation item leaked in as a job: {job.title!r}"28 assert not is_nav_text(job.title), f"nav-shaped title leaked in: {job.title!r}"29 30 31# ------------------------------------------------------------- footer traps32 33 34@pytest.mark.asyncio35async def test_footer_trap_returns_only_real_jobs(server):36 result = await scrape(f"{server}/footer_trap.html", use_browser=False, use_llm=False)37 38 assert result.job_count == 6, [j.title for j in result.jobs]39 assert result.diagnostics.strategy == "dom_heuristic"40 assert_no_chrome(result)41 42 assert titles(result) == {43 "senior backend engineer",44 "staff platform engineer",45 "qa automation engineer",46 "product designer",47 "design systems lead",48 "technical recruiter intern",49 }50 51 52@pytest.mark.asyncio53async def test_footer_trap_field_quality(server):54 result = await scrape(f"{server}/footer_trap.html", use_browser=False, use_llm=False)55 by_title = {j.title: j for j in result.jobs}56 57 backend = by_title["Senior Backend Engineer"]58 assert backend.location == "Bengaluru, India"59 assert backend.employment_type == "Full-time"60 assert backend.department == "Engineering" # from the section heading61 assert backend.apply_url.endswith("/careers/jobs/senior-backend-engineer")62 assert backend.seniority == "senior"63 64 assert by_title["Staff Platform Engineer"].workplace_type == "remote"65 assert by_title["Design Systems Lead"].workplace_type == "hybrid"66 assert by_title["Design Systems Lead"].department == "Design"67 assert by_title["Technical Recruiter Intern"].employment_type == "Internship"68 assert by_title["QA Automation Engineer"].employment_type == "Contract"69 70 # Every job must link somewhere other than the listing page itself.71 for job in result.jobs:72 assert job.apply_url and "/careers/jobs/" in job.apply_url73 74 75@pytest.mark.asyncio76async def test_no_openings_page_returns_zero_not_footer_links(server):77 """The nastiest false positive: an empty board whose footer says 'Careers'."""78 result = await scrape(f"{server}/no_openings.html", use_browser=False, use_llm=False)79 assert result.job_count == 0, [j.title for j in result.jobs]80 assert any("no open positions" in w for w in result.warnings)81 82 83# ------------------------------------------------------------ structured data84 85 86@pytest.mark.asyncio87async def test_jsonld_wins_and_breadcrumbs_are_ignored(server):88 result = await scrape(f"{server}/jsonld.html", use_browser=False, use_llm=False)89 90 assert result.diagnostics.strategy == "structured_data"91 assert result.job_count == 3, [j.title for j in result.jobs]92 assert_no_chrome(result)93 assert titles(result) == {94 "machine learning engineer",95 "technical writer",96 "site reliability engineer",97 }98 99 by_title = {j.title: j for j in result.jobs}100 ml = by_title["Machine Learning Engineer"]101 assert ml.location == "Seattle, WA, US"102 assert ml.employment_type == "Full-time"103 assert ml.requisition_id == "NW-2291" or ml.id == "NW-2291"104 assert ml.salary and ml.salary.min == 180000 and ml.salary.currency == "USD"105 assert ml.posted_at.startswith("2026-07-14")106 107 writer = by_title["Technical Writer"]108 assert writer.workplace_type == "remote"109 assert writer.employment_type == "Part-time"110 111 sre = by_title["Site Reliability Engineer"]112 assert "Dublin, IE" in sre.locations and "Lisbon, PT" in sre.locations113 114 115@pytest.mark.asyncio116async def test_embedded_next_data(server):117 result = await scrape(f"{server}/nextdata.html", use_browser=False, use_llm=False)118 119 assert result.diagnostics.strategy == "embedded_json"120 assert result.job_count == 4, [j.title for j in result.jobs]121 assert_no_chrome(result)122 assert "senior data engineer" in titles(result)123 124 by_title = {j.title: j for j in result.jobs}125 assert by_title["Security Analyst Intern"].employment_type == "Internship"126 assert by_title["Engineering Manager, Payments"].workplace_type == "remote"127 assert by_title["Senior Data Engineer"].apply_url.endswith("/careers/c-1001")128 129 130@pytest.mark.asyncio131async def test_table_layout_uses_column_headers(server):132 result = await scrape(f"{server}/table.html", use_browser=False, use_llm=False)133 134 assert result.job_count == 5, [j.title for j in result.jobs]135 assert_no_chrome(result)136 137 by_title = {j.title: j for j in result.jobs}138 nurse = by_title["Registered Nurse, ICU"]139 assert nurse.department == "Nursing"140 assert nurse.location == "Manchester, United Kingdom"141 assert nurse.employment_type == "Full-time"142 assert nurse.posted_at.startswith("2026-07-21")143 assert by_title["Clinical Pharmacist"].employment_type == "Part-time"144 assert by_title["Medical Secretary"].workplace_type == "remote"145 146 147# ------------------------------------------------------------------ paging148 149 150@pytest.mark.asyncio151async def test_pagination_is_followed(server):152 result = await scrape(f"{server}/paged_1.html", use_browser=False, use_llm=False, max_pages=5)153 assert result.job_count == 5, [j.title for j in result.jobs]154 assert "customer success lead" in titles(result)155 assert "revenue operations analyst" in titles(result)156 assert_no_chrome(result)157 158 159# --------------------------------------------------------------- discovery160 161 162@pytest.mark.asyncio163async def test_homepage_is_followed_to_the_careers_page(server):164 result = await scrape(f"{server}/homepage.html", use_browser=False, use_llm=False)165 assert result.job_count == 6, [j.title for j in result.jobs]166 assert any("followed careers link" in w for w in result.warnings)167 assert_no_chrome(result)168 169 170@pytest.mark.asyncio171async def test_redirects_are_followed(server):172 result = await scrape(f"{server}/redirect-ok", use_browser=False, use_llm=False)173 assert result.job_count == 6174 175 176# ------------------------------------------------------------ ATS detection177 178 179@pytest.mark.parametrize(180 "url,expected,token",181 [182 ("https://boards.greenhouse.io/stripe", "greenhouse", "stripe"),183 ("https://job-boards.greenhouse.io/anthropic", "greenhouse", "anthropic"),184 ("https://jobs.lever.co/netflix", "lever", "netflix"),185 ("https://jobs.lever.co/netflix/abc-123", "lever", "netflix"),186 ("https://jobs.ashbyhq.com/openai", "ashby", "openai"),187 ("https://apply.workable.com/hotjar/", "workable", "hotjar"),188 ("https://careers.smartrecruiters.com/Ubisoft", "smartrecruiters", "Ubisoft"),189 ("https://acme.recruitee.com/", "recruitee", "acme"),190 ("https://nvidia.wd5.myworkdayjobs.com/en-US/NVIDIAExternalCareerSite", "workday", "nvidia"),191 ("https://acme.bamboohr.com/careers", "bamboohr", "acme"),192 ("https://acme.jobs.personio.de/", "personio", "acme"),193 ("https://acme.breezy.hr/", "breezy", "acme"),194 ("https://ats.rippling.com/acme/jobs", "rippling", "acme"),195 ("https://acme.applytojob.com/apply", "jazzhr", "acme"),196 ("https://acme.eightfold.ai/careers", "eightfold", "acme"),197 ("https://careers-acme.icims.com/jobs/search", "icims", "careers-acme"),198 ],199)200def test_ats_detected_from_url(url, expected, token):201 match = ats_mod.detect(url)202 assert match is not None, url203 assert match.name == expected204 assert match.token == token205 206 207def test_ats_detected_from_embedded_script(server_html=None):208 from pathlib import Path209 210 html = (Path(__file__).parent / "fixtures" / "greenhouse_embed.html").read_text()211 match = ats_mod.detect("https://acme.example/careers", html)212 assert match is not None213 assert match.name == "greenhouse"214 assert match.token == "acmecorp"215 assert match.via == "html"216 217 218def test_ats_detection_ignores_reserved_path_segments():219 assert ats_mod.detect("https://boards.greenhouse.io/embed/job_board?for=acme").token == "acme"220 # A bare vendor domain with no company slug must not yield a bogus token.221 assert ats_mod.detect("https://jobs.lever.co/") is None222 223 224# ------------------------------------------- regression: the "Quick Links" bug225 226 227@pytest.mark.asyncio228async def test_footer_link_column_is_never_returned_as_jobs(server):229 """Regression for a real failure on a WordPress/Elementor careers page.230 231 The footer's "Quick Links" column — 14 title-cased links to product pages,232 in a container with no "footer" in its tag or class — outscored an empty233 board and was returned as 14 jobs. Scoring alone could not stop it, because234 scoring is additive: plausible-looking repeated links accumulate enough235 points without a single job-specific signal. The evidence gate does.236 """237 result = await scrape(f"{server}/quick_links_trap.html", use_browser=False, use_llm=False)238 239 assert result.job_count == 0, [j.title for j in result.jobs]240 for job in result.jobs:241 assert "Management" not in job.title242 assert job.department != "Quick Links"243 244 245@pytest.mark.asyncio246async def test_evidence_gate_reports_why_a_candidate_was_rejected(server):247 """A rejection must be explainable, not silent."""248 from app.extract import domheur249 from app.htmlutil import parse250 from pathlib import Path251 252 html = (Path(__file__).parent / "fixtures" / "quick_links_trap.html").read_text()253 rows, conf, diag = domheur.extract(parse(html, "https://factools.example/career/"),254 "https://factools.example/career/")255 assert rows == []256 rejected = diag.get("rejected", [])257 assert rejected, "expected rejected candidates to be recorded"258 259 quick_links = [r for r in rejected if r["items"] >= 10]260 assert quick_links, [r["signature"] for r in rejected]261 why = quick_links[0]["why"]262 assert why in ("no job-specific evidence", "sits under a footer-column heading")263 assert quick_links[0]["evidence"]["passed"] == []264 265 266# ---------------------------------- regression: cards that glue fields together267 268 269@pytest.mark.asyncio270async def test_glued_card_fields_are_separated(server):271 """Regression for a real failure on a studio careers page.272 273 Each card is one anchor wrapping unlabelled spans. lxml concatenates raw274 text nodes, so the title came back as "Junior 3D Artist1 yearBangaloreFull275 Time", location was empty, and every row inherited the department276 "Work-Life Balance" from an unrelated benefits heading further up the page.277 """278 result = await scrape(f"{server}/glued_card.html", use_browser=False, use_llm=False)279 280 assert result.job_count == 8, [j.title for j in result.jobs]281 assert_no_chrome(result)282 283 assert titles(result) == {284 "junior 3d artist", "mid 3d visualizer", "mid unreal artist",285 "mid unreal engine developer", "roblox game developer",286 "senior 3d artist", "senior 3d visualizer", "senior unreal artist",287 }288 289 by_title = {j.title: j for j in result.jobs}290 291 # Experience is its own field, never part of the title.292 assert by_title["Junior 3D Artist"].experience == "1 year"293 assert by_title["Mid 3D Visualizer"].experience == "1.5 to 3 years"294 assert by_title["Mid Unreal Engine Developer"].experience == "3+ years"295 assert by_title["Senior Unreal Artist"].experience == "5 to 7+yrs"296 assert by_title["Senior 3D Artist"].experience is None # card omits it297 298 # Bare city names resolve without a country suffix.299 assert by_title["Junior 3D Artist"].location == "Bangalore"300 assert by_title["Mid Unreal Artist"].location == "Gurgaon"301 assert by_title["Senior Unreal Artist"].location == "Dubai"302 303 assert all(j.employment_type == "Full-time" for j in result.jobs)304 305 # A benefits heading must never become a department.306 for job in result.jobs:307 assert job.department != "Work-Life Balance"308 assert job.department is None309 assert "year" not in job.title.lower()310 assert "Full Time" not in job.title311 