Team Ai
Apppublic

booleanbeyond/jobfetch

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes
test_extraction.py311 linesDownload Raw Back to tests
1"""End-to-end extraction tests against local fixture pages."""2 3from __future__ import annotations4 5import pytest6 7from app.extract import ats as ats_mod8from app.pipeline import scrape9from app.vocab import is_nav_text10 11NAV_WORDS = {12    "careers", "jobs", "jobs at acme", "view all jobs", "see all openings",13    "life at acme", "benefits", "privacy policy", "terms of service",14    "cookie policy", "linkedin", "twitter", "glassdoor", "about us", "engineering",15    "design", "sales", "operations", "job alerts", "all vacancies", "open positions",16}17 18 19def titles(result) -> set[str]:20    return {j.title.strip().lower() for j in result.jobs}21 22 23def assert_no_chrome(result):24    """The core regression guard: nothing that is site furniture may appear."""25    for job in result.jobs:26        low = job.title.strip().lower()27        assert low not in NAV_WORDS, f"navigation item leaked in as a job: {job.title!r}"28        assert not is_nav_text(job.title), f"nav-shaped title leaked in: {job.title!r}"29 30 31# ------------------------------------------------------------- footer traps32 33 34@pytest.mark.asyncio35async def test_footer_trap_returns_only_real_jobs(server):36    result = await scrape(f"{server}/footer_trap.html", use_browser=False, use_llm=False)37 38    assert result.job_count == 6, [j.title for j in result.jobs]39    assert result.diagnostics.strategy == "dom_heuristic"40    assert_no_chrome(result)41 42    assert titles(result) == {43        "senior backend engineer",44        "staff platform engineer",45        "qa automation engineer",46        "product designer",47        "design systems lead",48        "technical recruiter intern",49    }50 51 52@pytest.mark.asyncio53async def test_footer_trap_field_quality(server):54    result = await scrape(f"{server}/footer_trap.html", use_browser=False, use_llm=False)55    by_title = {j.title: j for j in result.jobs}56 57    backend = by_title["Senior Backend Engineer"]58    assert backend.location == "Bengaluru, India"59    assert backend.employment_type == "Full-time"60    assert backend.department == "Engineering"          # from the section heading61    assert backend.apply_url.endswith("/careers/jobs/senior-backend-engineer")62    assert backend.seniority == "senior"63 64    assert by_title["Staff Platform Engineer"].workplace_type == "remote"65    assert by_title["Design Systems Lead"].workplace_type == "hybrid"66    assert by_title["Design Systems Lead"].department == "Design"67    assert by_title["Technical Recruiter Intern"].employment_type == "Internship"68    assert by_title["QA Automation Engineer"].employment_type == "Contract"69 70    # Every job must link somewhere other than the listing page itself.71    for job in result.jobs:72        assert job.apply_url and "/careers/jobs/" in job.apply_url73 74 75@pytest.mark.asyncio76async def test_no_openings_page_returns_zero_not_footer_links(server):77    """The nastiest false positive: an empty board whose footer says 'Careers'."""78    result = await scrape(f"{server}/no_openings.html", use_browser=False, use_llm=False)79    assert result.job_count == 0, [j.title for j in result.jobs]80    assert any("no open positions" in w for w in result.warnings)81 82 83# ------------------------------------------------------------ structured data84 85 86@pytest.mark.asyncio87async def test_jsonld_wins_and_breadcrumbs_are_ignored(server):88    result = await scrape(f"{server}/jsonld.html", use_browser=False, use_llm=False)89 90    assert result.diagnostics.strategy == "structured_data"91    assert result.job_count == 3, [j.title for j in result.jobs]92    assert_no_chrome(result)93    assert titles(result) == {94        "machine learning engineer",95        "technical writer",96        "site reliability engineer",97    }98 99    by_title = {j.title: j for j in result.jobs}100    ml = by_title["Machine Learning Engineer"]101    assert ml.location == "Seattle, WA, US"102    assert ml.employment_type == "Full-time"103    assert ml.requisition_id == "NW-2291" or ml.id == "NW-2291"104    assert ml.salary and ml.salary.min == 180000 and ml.salary.currency == "USD"105    assert ml.posted_at.startswith("2026-07-14")106 107    writer = by_title["Technical Writer"]108    assert writer.workplace_type == "remote"109    assert writer.employment_type == "Part-time"110 111    sre = by_title["Site Reliability Engineer"]112    assert "Dublin, IE" in sre.locations and "Lisbon, PT" in sre.locations113 114 115@pytest.mark.asyncio116async def test_embedded_next_data(server):117    result = await scrape(f"{server}/nextdata.html", use_browser=False, use_llm=False)118 119    assert result.diagnostics.strategy == "embedded_json"120    assert result.job_count == 4, [j.title for j in result.jobs]121    assert_no_chrome(result)122    assert "senior data engineer" in titles(result)123 124    by_title = {j.title: j for j in result.jobs}125    assert by_title["Security Analyst Intern"].employment_type == "Internship"126    assert by_title["Engineering Manager, Payments"].workplace_type == "remote"127    assert by_title["Senior Data Engineer"].apply_url.endswith("/careers/c-1001")128 129 130@pytest.mark.asyncio131async def test_table_layout_uses_column_headers(server):132    result = await scrape(f"{server}/table.html", use_browser=False, use_llm=False)133 134    assert result.job_count == 5, [j.title for j in result.jobs]135    assert_no_chrome(result)136 137    by_title = {j.title: j for j in result.jobs}138    nurse = by_title["Registered Nurse, ICU"]139    assert nurse.department == "Nursing"140    assert nurse.location == "Manchester, United Kingdom"141    assert nurse.employment_type == "Full-time"142    assert nurse.posted_at.startswith("2026-07-21")143    assert by_title["Clinical Pharmacist"].employment_type == "Part-time"144    assert by_title["Medical Secretary"].workplace_type == "remote"145 146 147# ------------------------------------------------------------------ paging148 149 150@pytest.mark.asyncio151async def test_pagination_is_followed(server):152    result = await scrape(f"{server}/paged_1.html", use_browser=False, use_llm=False, max_pages=5)153    assert result.job_count == 5, [j.title for j in result.jobs]154    assert "customer success lead" in titles(result)155    assert "revenue operations analyst" in titles(result)156    assert_no_chrome(result)157 158 159# --------------------------------------------------------------- discovery160 161 162@pytest.mark.asyncio163async def test_homepage_is_followed_to_the_careers_page(server):164    result = await scrape(f"{server}/homepage.html", use_browser=False, use_llm=False)165    assert result.job_count == 6, [j.title for j in result.jobs]166    assert any("followed careers link" in w for w in result.warnings)167    assert_no_chrome(result)168 169 170@pytest.mark.asyncio171async def test_redirects_are_followed(server):172    result = await scrape(f"{server}/redirect-ok", use_browser=False, use_llm=False)173    assert result.job_count == 6174 175 176# ------------------------------------------------------------ ATS detection177 178 179@pytest.mark.parametrize(180    "url,expected,token",181    [182        ("https://boards.greenhouse.io/stripe", "greenhouse", "stripe"),183        ("https://job-boards.greenhouse.io/anthropic", "greenhouse", "anthropic"),184        ("https://jobs.lever.co/netflix", "lever", "netflix"),185        ("https://jobs.lever.co/netflix/abc-123", "lever", "netflix"),186        ("https://jobs.ashbyhq.com/openai", "ashby", "openai"),187        ("https://apply.workable.com/hotjar/", "workable", "hotjar"),188        ("https://careers.smartrecruiters.com/Ubisoft", "smartrecruiters", "Ubisoft"),189        ("https://acme.recruitee.com/", "recruitee", "acme"),190        ("https://nvidia.wd5.myworkdayjobs.com/en-US/NVIDIAExternalCareerSite", "workday", "nvidia"),191        ("https://acme.bamboohr.com/careers", "bamboohr", "acme"),192        ("https://acme.jobs.personio.de/", "personio", "acme"),193        ("https://acme.breezy.hr/", "breezy", "acme"),194        ("https://ats.rippling.com/acme/jobs", "rippling", "acme"),195        ("https://acme.applytojob.com/apply", "jazzhr", "acme"),196        ("https://acme.eightfold.ai/careers", "eightfold", "acme"),197        ("https://careers-acme.icims.com/jobs/search", "icims", "careers-acme"),198    ],199)200def test_ats_detected_from_url(url, expected, token):201    match = ats_mod.detect(url)202    assert match is not None, url203    assert match.name == expected204    assert match.token == token205 206 207def test_ats_detected_from_embedded_script(server_html=None):208    from pathlib import Path209 210    html = (Path(__file__).parent / "fixtures" / "greenhouse_embed.html").read_text()211    match = ats_mod.detect("https://acme.example/careers", html)212    assert match is not None213    assert match.name == "greenhouse"214    assert match.token == "acmecorp"215    assert match.via == "html"216 217 218def test_ats_detection_ignores_reserved_path_segments():219    assert ats_mod.detect("https://boards.greenhouse.io/embed/job_board?for=acme").token == "acme"220    # A bare vendor domain with no company slug must not yield a bogus token.221    assert ats_mod.detect("https://jobs.lever.co/") is None222 223 224# ------------------------------------------- regression: the "Quick Links" bug225 226 227@pytest.mark.asyncio228async def test_footer_link_column_is_never_returned_as_jobs(server):229    """Regression for a real failure on a WordPress/Elementor careers page.230 231    The footer's "Quick Links" column — 14 title-cased links to product pages,232    in a container with no "footer" in its tag or class — outscored an empty233    board and was returned as 14 jobs. Scoring alone could not stop it, because234    scoring is additive: plausible-looking repeated links accumulate enough235    points without a single job-specific signal. The evidence gate does.236    """237    result = await scrape(f"{server}/quick_links_trap.html", use_browser=False, use_llm=False)238 239    assert result.job_count == 0, [j.title for j in result.jobs]240    for job in result.jobs:241        assert "Management" not in job.title242        assert job.department != "Quick Links"243 244 245@pytest.mark.asyncio246async def test_evidence_gate_reports_why_a_candidate_was_rejected(server):247    """A rejection must be explainable, not silent."""248    from app.extract import domheur249    from app.htmlutil import parse250    from pathlib import Path251 252    html = (Path(__file__).parent / "fixtures" / "quick_links_trap.html").read_text()253    rows, conf, diag = domheur.extract(parse(html, "https://factools.example/career/"),254                                       "https://factools.example/career/")255    assert rows == []256    rejected = diag.get("rejected", [])257    assert rejected, "expected rejected candidates to be recorded"258 259    quick_links = [r for r in rejected if r["items"] >= 10]260    assert quick_links, [r["signature"] for r in rejected]261    why = quick_links[0]["why"]262    assert why in ("no job-specific evidence", "sits under a footer-column heading")263    assert quick_links[0]["evidence"]["passed"] == []264 265 266# ---------------------------------- regression: cards that glue fields together267 268 269@pytest.mark.asyncio270async def test_glued_card_fields_are_separated(server):271    """Regression for a real failure on a studio careers page.272 273    Each card is one anchor wrapping unlabelled spans. lxml concatenates raw274    text nodes, so the title came back as "Junior 3D Artist1 yearBangaloreFull275    Time", location was empty, and every row inherited the department276    "Work-Life Balance" from an unrelated benefits heading further up the page.277    """278    result = await scrape(f"{server}/glued_card.html", use_browser=False, use_llm=False)279 280    assert result.job_count == 8, [j.title for j in result.jobs]281    assert_no_chrome(result)282 283    assert titles(result) == {284        "junior 3d artist", "mid 3d visualizer", "mid unreal artist",285        "mid unreal engine developer", "roblox game developer",286        "senior 3d artist", "senior 3d visualizer", "senior unreal artist",287    }288 289    by_title = {j.title: j for j in result.jobs}290 291    # Experience is its own field, never part of the title.292    assert by_title["Junior 3D Artist"].experience == "1 year"293    assert by_title["Mid 3D Visualizer"].experience == "1.5 to 3 years"294    assert by_title["Mid Unreal Engine Developer"].experience == "3+ years"295    assert by_title["Senior Unreal Artist"].experience == "5 to 7+yrs"296    assert by_title["Senior 3D Artist"].experience is None      # card omits it297 298    # Bare city names resolve without a country suffix.299    assert by_title["Junior 3D Artist"].location == "Bangalore"300    assert by_title["Mid Unreal Artist"].location == "Gurgaon"301    assert by_title["Senior Unreal Artist"].location == "Dubai"302 303    assert all(j.employment_type == "Full-time" for j in result.jobs)304 305    # A benefits heading must never become a department.306    for job in result.jobs:307        assert job.department != "Work-Life Balance"308        assert job.department is None309        assert "year" not in job.title.lower()310        assert "Full Time" not in job.title311