Team Ai
Apppublic

booleanbeyond/jobfetch

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes
test_units.py348 linesDownload Raw Back to tests
1"""Unit tests for normalisation, scoring guards and the LLM verifier."""2 3from __future__ import annotations4 5import pytest6 7from app.extract import jsonfind8from app.extract.llm import verify9from app.htmlutil import parse, strip_boilerplate, clone10from app.models import Job11from app.normalize import (12    build_job,13    clean_title,14    dedupe,15    detect_workplace_type,16    is_valid_title,17    looks_like_location,18    normalize_employment_type,19    parse_date,20    split_locations,21)22from app.vocab import is_nav_text23 24 25# ------------------------------------------------------------------- vocab26 27 28@pytest.mark.parametrize(29    "text",30    [31        "Careers", "careers", "  Careers  ", "Jobs", "View all jobs", "See all openings",32        "Open positions", "Join our team", "Apply now", "Life at Acme", "Benefits",33        "Privacy Policy", "Terms of Service", "Cookie Policy", "LinkedIn", "Load more",34        "All rights reserved", "Careers at Acme", "View all 42 jobs", "Open roles (12)",35        "Skip to main content", "Job alerts", "We're hiring",36    ],37)38def test_navigation_vocabulary_is_rejected(text):39    assert is_nav_text(text) is True40    assert is_valid_title(text) is False41 42 43@pytest.mark.parametrize(44    "text",45    [46        "Senior Backend Engineer",47        "Engineering Manager, Payments",48        "Registered Nurse, ICU",49        "Staff Machine Learning Engineer (Remote)",50        "Product Designer",51        "VP of Sales, EMEA",52        "Technical Recruiter Intern",53        "SRE II",54    ],55)56def test_real_titles_are_accepted(text):57    assert is_nav_text(text) is False58    assert is_valid_title(text) is True59 60 61# --------------------------------------------------------------- normalise62 63 64@pytest.mark.parametrize(65    "raw,expected",66    [67        ("FULL_TIME", "Full-time"), ("Full time", "Full-time"), ("fulltime", "Full-time"),68        ("PART_TIME", "Part-time"), ("Contractor", "Contract"), ("freelance", "Contract"),69        ("INTERNSHIP", "Internship"), ("intern", "Internship"), ("Temporary", "Temporary"),70        ("Permanent", "Full-time"), ("werkstudent", "Internship"), ("nonsense", None),71    ],72)73def test_employment_type_normalisation(raw, expected):74    assert normalize_employment_type(raw) == expected75 76 77@pytest.mark.parametrize(78    "text,expected",79    [80        ("Remote — EMEA", "remote"),81        ("Fully remote", "remote"),82        ("Hybrid — Berlin, Germany", "hybrid"),83        ("Hybrid remote", "hybrid"),          # hybrid must beat a stray "remote"84        ("On-site, Austin TX", "onsite"),85        ("Bengaluru, India", None),86    ],87)88def test_workplace_type_detection(text, expected):89    assert detect_workplace_type(text) == expected90 91 92@pytest.mark.parametrize(93    "text",94    ["Remote", "Austin, TX", "Bengaluru, India", "London, United Kingdom",95     "Hybrid — Berlin, Germany", "Singapore", "EMEA"],96)97def test_location_recognition(text):98    assert looks_like_location(text) is True99 100 101@pytest.mark.parametrize("text", ["Senior Backend Engineer", "Apply now", "Full-time"])102def test_non_locations_rejected(text):103    assert looks_like_location(text) is False104 105 106def test_multi_location_split_keeps_city_state_pairs():107    assert split_locations("Austin, TX") == ["Austin, TX"]108    assert split_locations("London; Berlin; Paris") == ["London", "Berlin", "Paris"]109    assert split_locations(["Dublin, IE", "Dublin, IE", "Lisbon, PT"]) == ["Dublin, IE", "Lisbon, PT"]110 111 112@pytest.mark.parametrize(113    "value,ok",114    [115        ("2026-07-14", True), ("2026-07-14T09:00:00Z", True), ("July 14, 2026", True),116        (1752480000, True), (1752480000000, True),117        ("Posted 30+ Days Ago", False), ("yesterday", False), ("", False), (None, False),118        ("just posted", False), ("garbage", False),119    ],120)121def test_date_parsing_never_fabricates(value, ok):122    result = parse_date(value)123    assert (result is not None) is ok124 125 126def test_title_cleanup_strips_appended_location():127    assert clean_title("Product Designer — London", location="London") == "Product Designer"128    assert clean_title("Backend Engineer - Remote") == "Backend Engineer"129    assert clean_title("NEW: Data Analyst") == "Data Analyst"130 131 132def test_build_job_rejects_navigation_rows():133    assert build_job({"title": "Careers", "apply_url": "/careers"}, base_url="https://a.com",134                     source="dom_heuristic") is None135    assert build_job({"title": "  "}, base_url="https://a.com", source="dom_heuristic") is None136    job = build_job({"title": "Data Analyst", "apply_url": "/jobs/1"},137                    base_url="https://a.com", source="dom_heuristic")138    assert job is not None and job.apply_url == "https://a.com/jobs/1"139 140 141def test_build_job_drops_social_and_mailto_urls():142    job = build_job({"title": "Data Analyst", "apply_url": "mailto:jobs@a.com"},143                    base_url="https://a.com", source="dom_heuristic")144    assert job is not None and job.apply_url is None145    job = build_job({"title": "Data Analyst", "apply_url": "https://twitter.com/acme"},146                    base_url="https://a.com", source="dom_heuristic")147    assert job.apply_url is None148 149 150def test_dedupe_merges_partial_rows():151    a = Job(title="Data Analyst", apply_url="https://a.com/jobs/1?utm_source=x", source="ats_api")152    b = Job(title="Data Analyst", apply_url="https://a.com/jobs/1", location="Berlin",153            department="Data", source="dom_heuristic")154    merged = dedupe([a, b])155    assert len(merged) == 1156    assert merged[0].location == "Berlin"157    assert merged[0].department == "Data"158 159 160def test_dedupe_keeps_distinct_locations_of_same_title():161    a = Job(title="Data Analyst", location="Berlin", apply_url="https://a.com/jobs/1")162    b = Job(title="Data Analyst", location="Madrid", apply_url="https://a.com/jobs/2")163    assert len(dedupe([a, b])) == 2164 165 166# ------------------------------------------------------------- boilerplate167 168 169def test_strip_boilerplate_removes_footer_but_keeps_list():170    from pathlib import Path171 172    html = (Path(__file__).parent / "fixtures" / "footer_trap.html").read_text()173    root = parse(html, "https://acme.example/careers")174    stripped = strip_boilerplate(clone(root))175    text = stripped.text_content()176 177    assert "Senior Backend Engineer" in text178    assert "All rights reserved" not in text179    assert "Privacy Policy" not in text180    assert "Jobs at Acme" not in text181 182 183def test_strip_boilerplate_rescues_a_joblist_inside_an_aside():184    html = """185    <html><body>186      <aside class="job-list-sidebar">187        <a class="job" href="/jobs/1">Data Engineer</a>188        <a class="job" href="/jobs/2">Data Scientist</a>189        <a class="job" href="/jobs/3">Analytics Engineer</a>190      </aside>191      <footer><a href="/careers">Careers</a></footer>192    </body></html>"""193    stripped = strip_boilerplate(clone(parse(html, "https://a.com/careers")))194    text = stripped.text_content()195    assert "Data Engineer" in text and "Analytics Engineer" in text196    assert "Careers" not in text197 198 199# ----------------------------------------------------------------- jsonfind200 201 202def test_jsonfind_prefers_jobs_over_navigation_arrays():203    payload = {204        "nav": [{"title": "Careers", "href": "/careers"}, {"title": "About", "href": "/about"}],205        "data": {206            "jobs": [207                {"title": "Data Engineer", "location": "Berlin", "department": "Data",208                 "applyUrl": "/jobs/1", "employmentType": "FULL_TIME"},209                {"title": "Data Scientist", "location": "Madrid", "department": "Data",210                 "applyUrl": "/jobs/2", "employmentType": "FULL_TIME"},211            ]212        },213    }214    found = jsonfind.find_job_arrays(payload)215    assert found, "expected to find a job array"216    best_path, rows, _score = found[0]217    assert "jobs" in best_path218    assert {r["title"] for r in rows} == {"Data Engineer", "Data Scientist"}219 220 221def test_jsonfind_rejects_pure_navigation():222    payload = {"menu": [{"title": "Careers", "href": "/c"}, {"title": "Blog", "href": "/b"}]}223    assert jsonfind.find_job_arrays(payload) == []224 225 226def test_rows_to_raw_maps_vendor_key_variants():227    rows = [{"jobTitle": "SRE", "locationsText": "Dublin", "timeType": "Full time",228             "externalPath": "/job/sre", "postedOn": "2026-01-05", "isRemote": True}]229    raw = jsonfind.rows_to_raw(rows)[0]230    assert raw["title"] == "SRE"231    assert raw["location"] == "Dublin"232    assert raw["employment_type"] == "Full time"233    assert raw["workplace_type"] == "remote"234 235 236# ------------------------------------------------------------ llm verifier237 238 239PAGE_TEXT = (240    "Open roles Senior Backend Engineer Bengaluru, India Full-time "241    "Product Designer London, United Kingdom Full-time"242)243PAGE_HREFS = {"https://a.com/jobs/backend", "https://a.com/jobs/designer"}244 245 246def test_llm_verifier_drops_hallucinated_titles():247    rows = [248        {"title": "Senior Backend Engineer", "apply_url": "https://a.com/jobs/backend"},249        {"title": "Chief Vibes Officer", "apply_url": "https://a.com/jobs/vibes"},  # invented250    ]251    kept, stats = verify(rows, hrefs=PAGE_HREFS, plain_text=PAGE_TEXT)252    assert [r["title"] for r in kept] == ["Senior Backend Engineer"]253    assert stats["dropped_not_on_page"] == 1254 255 256def test_llm_verifier_strips_invented_urls_but_keeps_the_job():257    rows = [{"title": "Product Designer", "apply_url": "https://a.com/jobs/made-up"}]258    kept, stats = verify(rows, hrefs=PAGE_HREFS, plain_text=PAGE_TEXT)259    assert len(kept) == 1260    assert kept[0]["apply_url"] is None261    assert stats["urls_stripped"] == 1262 263 264def test_llm_verifier_drops_navigation_titles():265    rows = [{"title": "Careers"}, {"title": "View all jobs"}, {"title": "Product Designer"}]266    kept, _ = verify(rows, hrefs=PAGE_HREFS, plain_text=PAGE_TEXT + " Careers View all jobs")267    assert [r["title"] for r in kept] == ["Product Designer"]268 269 270def test_llm_verifier_strips_fields_not_present_on_page():271    rows = [{"title": "Product Designer", "location": "Atlantis", "department": "Design"}]272    kept, stats = verify(rows, hrefs=PAGE_HREFS, plain_text=PAGE_TEXT)273    assert "location" not in kept[0]274    assert stats["fields_stripped"] >= 1275 276 277def test_llm_verifier_dedupes_echoed_rows():278    rows = [{"title": "Product Designer"}] * 5279    kept, _ = verify(rows, hrefs=PAGE_HREFS, plain_text=PAGE_TEXT)280    assert len(kept) == 1281 282 283# ------------------------------------------------------- openrouter provider284 285 286def test_openrouter_tool_call_is_parsed():287    from app.extract.llm import _parse_openai_tool_response288 289    payload = {290        "choices": [{291            "message": {292                "tool_calls": [{293                    "function": {294                        "name": "emit_jobs",295                        "arguments": '{"jobs":[{"title":"Data Engineer","location":"Berlin"}]}',296                    }297                }]298            }299        }]300    }301    rows = _parse_openai_tool_response(payload)302    assert rows == [{"title": "Data Engineer", "location": "Berlin"}]303 304 305def test_openrouter_falls_back_to_json_content():306    """Not every model routed through OpenRouter supports tool calling."""307    from app.extract.llm import _parse_openai_tool_response308 309    payload = {"choices": [{"message": {310        "content": '```json\n{"jobs":[{"title":"SRE"},{"title":"QA Lead"}]}\n```'311    }}]}312    assert [r["title"] for r in _parse_openai_tool_response(payload)] == ["SRE", "QA Lead"]313 314 315def test_openrouter_garbage_response_yields_nothing():316    from app.extract.llm import _parse_openai_tool_response317 318    assert _parse_openai_tool_response({}) == []319    assert _parse_openai_tool_response({"choices": [{"message": {"content": "sorry!"}}]}) == []320 321 322@pytest.mark.asyncio323async def test_llm_tier_is_a_no_op_without_a_key():324    from app.extract import llm325 326    rows, conf, diag = await llm.extract(None, "https://a.com/careers")327    assert rows == [] and conf == 0.0328    assert diag["provider"] == "none"329 330 331def test_provider_resolution():332    import dataclasses333    from app.config import settings as base334 335    s = dataclasses.replace(base, llm_enabled=True, openrouter_api_key="k", anthropic_api_key="")336    assert s.resolved_llm_provider() == "openrouter"337 338    s = dataclasses.replace(base, llm_enabled=True, openrouter_api_key="", anthropic_api_key="k")339    assert s.resolved_llm_provider() == "anthropic"340 341    # Explicit choice wins, but only if that provider actually has a key.342    s = dataclasses.replace(base, llm_enabled=True, llm_provider="anthropic",343                            openrouter_api_key="k", anthropic_api_key="")344    assert s.resolved_llm_provider() == "none"345 346    s = dataclasses.replace(base, llm_enabled=False, openrouter_api_key="k")347    assert s.resolved_llm_provider() == "none"348