booleanbeyond/jobfetch
0
1"""Tier 2 — structured data already published by the page.2 3Order of trust: schema.org JobPosting (JSON-LD) > microdata > RSS/Atom job feed.4All three are author-declared, so a footer link can never appear here: an5`ItemList` of breadcrumbs is rejected because its entries fail the job-URL and6job-title tests.7"""8 9from __future__ import annotations10 11import json12import logging13import re14from typing import Any, Iterable, Iterator, Optional15from urllib.parse import urljoin16 17from ..htmlutil import looks_like_job_href, text_of18from ..normalize import clean_text, is_valid_title19from ..net import SafeClient20 21log = logging.getLogger("jobfetch.structured")22 23SCHEMA_JOB = "jobposting"24_TRAILING_COMMA = re.compile(r",\s*([}\]])")25_JS_COMMENT = re.compile(r"^\s*//.*$", re.M)26 27 28def _loads_tolerant(text: str) -> list[Any]:29 """Parse ld+json that may be invalid, doubled up, or comment-wrapped."""30 if not text:31 return []32 s = text.strip()33 s = s.replace("<!--", " ").replace("-->", " ")34 s = _JS_COMMENT.sub("", s)35 out: list[Any] = []36 try:37 return [json.loads(s)]38 except json.JSONDecodeError:39 pass40 try:41 return [json.loads(_TRAILING_COMMA.sub(r"\1", s))]42 except json.JSONDecodeError:43 pass44 # Multiple top-level values glued together.45 decoder = json.JSONDecoder()46 idx = 047 while idx < len(s) and len(out) < 50:48 while idx < len(s) and s[idx] in " \t\r\n,;":49 idx += 150 if idx >= len(s):51 break52 try:53 value, end = decoder.raw_decode(s, idx)54 except json.JSONDecodeError:55 break56 out.append(value)57 idx = end58 return out59 60 61def _types(node: Any) -> set[str]:62 if not isinstance(node, dict):63 return set()64 raw = node.get("@type") or node.get("type")65 values = raw if isinstance(raw, (list, tuple)) else [raw]66 out = set()67 for v in values:68 if isinstance(v, str):69 out.add(v.rsplit("/", 1)[-1].rsplit("#", 1)[-1].strip().lower())70 return out71 72 73def _walk(node: Any, depth: int = 0) -> Iterator[dict]:74 if depth > 12:75 return76 if isinstance(node, dict):77 yield node78 for value in node.values():79 yield from _walk(value, depth + 1)80 elif isinstance(node, (list, tuple)):81 for item in node:82 yield from _walk(item, depth + 1)83 84 85def _addr_to_text(node: Any) -> list[str]:86 """Flatten schema.org Place/PostalAddress into display strings."""87 out: list[str] = []88 if node is None:89 return out90 if isinstance(node, str):91 t = clean_text(node)92 return [t] if t else []93 if isinstance(node, (list, tuple)):94 for item in node:95 out.extend(_addr_to_text(item))96 return out97 if not isinstance(node, dict):98 return out99 100 types = _types(node)101 if "place" in types or "address" in node:102 return _addr_to_text(node.get("address")) or (103 [clean_text(node.get("name"))] if clean_text(node.get("name")) else []104 )105 106 parts = []107 for key in ("addressLocality", "addressRegion", "addressCountry"):108 val = node.get(key)109 if isinstance(val, dict):110 val = val.get("name") or val.get("addressCountry")111 val = clean_text(val)112 if val and val not in parts:113 parts.append(val)114 if parts:115 out.append(", ".join(parts))116 elif clean_text(node.get("name")):117 out.append(clean_text(node.get("name")))118 return out119 120 121def jsonld_job_to_raw(node: dict, base_url: Optional[str]) -> Optional[dict]:122 title = clean_text(node.get("title") or node.get("name"))123 if not title:124 return None125 126 locations = _addr_to_text(node.get("jobLocation"))127 if not locations:128 locations = _addr_to_text(node.get("applicantLocationRequirements"))129 130 remote = None131 jlt = node.get("jobLocationType")132 if isinstance(jlt, str) and "telecommute" in jlt.lower():133 remote = "remote"134 135 emp = node.get("employmentType")136 if isinstance(emp, (list, tuple)):137 emp = next((e for e in emp if isinstance(e, str)), None)138 139 ident = node.get("identifier")140 if isinstance(ident, dict):141 ident = ident.get("value") or ident.get("name")142 elif isinstance(ident, (list, tuple)):143 ident = next((i for i in ident if isinstance(i, (str, int))), None)144 145 salary_min = salary_max = currency = interval = None146 base_salary = node.get("baseSalary") or node.get("estimatedSalary")147 if isinstance(base_salary, dict):148 currency = base_salary.get("currency") or base_salary.get("currencyCode")149 value = base_salary.get("value")150 if isinstance(value, dict):151 salary_min = value.get("minValue") or value.get("value")152 salary_max = value.get("maxValue")153 interval = value.get("unitText")154 elif isinstance(value, (int, float, str)):155 salary_min = value156 157 url = node.get("url") or node.get("sameAs")158 if isinstance(url, (list, tuple)):159 url = next((u for u in url if isinstance(u, str)), None)160 if isinstance(url, dict):161 url = url.get("@id") or url.get("url")162 if not url:163 url = node.get("@id") if isinstance(node.get("@id"), str) else None164 if url and base_url:165 url = urljoin(base_url, url)166 167 dept = node.get("occupationalCategory") or node.get("industry") or node.get("department")168 if isinstance(dept, dict):169 dept = dept.get("name")170 171 return {172 "id": ident,173 "title": title,174 "department": clean_text(dept),175 "location": locations[0] if locations else None,176 "locations": locations or None,177 "workplace_type": remote,178 "employment_type": emp,179 "posted_at": node.get("datePosted"),180 "updated_at": node.get("dateModified"),181 "apply_url": url,182 "salary_min": salary_min,183 "salary_max": salary_max,184 "salary_currency": currency,185 "salary_interval": interval,186 "_hiring_org": (node.get("hiringOrganization") or {}).get("name")187 if isinstance(node.get("hiringOrganization"), dict)188 else clean_text(node.get("hiringOrganization")),189 }190 191 192def _itemlist_rows(node: dict, base_url: Optional[str]) -> list[dict]:193 """An ItemList of job links. Rejected unless entries look like real jobs."""194 elements = node.get("itemListElement")195 if not isinstance(elements, (list, tuple)) or len(elements) < 2:196 return []197 rows: list[dict] = []198 for el in elements:199 if not isinstance(el, dict):200 continue201 inner = el.get("item")202 if isinstance(inner, dict) and SCHEMA_JOB in _types(inner):203 row = jsonld_job_to_raw(inner, base_url)204 if row:205 rows.append(row)206 continue207 url = el.get("url") or (inner.get("url") if isinstance(inner, dict) else None)208 name = el.get("name") or (inner.get("name") if isinstance(inner, dict) else None)209 title = clean_text(name)210 if not url or not isinstance(url, str):211 continue212 abs_url = urljoin(base_url or "", url)213 # Breadcrumb guard: both the URL shape and the label must look like a job.214 if not looks_like_job_href(abs_url) or not is_valid_title(title):215 continue216 rows.append({"title": title, "apply_url": abs_url})217 # An ItemList that only half-passes is almost certainly navigation.218 if len(rows) < 2:219 return []220 return rows221 222 223def from_jsonld(root, base_url: Optional[str]) -> list[dict]:224 if root is None:225 return []226 rows: list[dict] = []227 seen_ids: set[int] = set()228 for script in root.xpath(229 "//script[contains(translate(@type,'ABCDEFGHIJKLMNOPQRSTUVWXYZ',"230 "'abcdefghijklmnopqrstuvwxyz'),'ld+json')]"231 ):232 payload = script.text_content()233 if not payload or len(payload) > 4_000_000:234 continue235 for doc in _loads_tolerant(payload):236 for node in _walk(doc):237 if id(node) in seen_ids:238 continue239 types = _types(node)240 if SCHEMA_JOB in types:241 seen_ids.add(id(node))242 row = jsonld_job_to_raw(node, base_url)243 if row:244 rows.append(row)245 elif "itemlist" in types:246 seen_ids.add(id(node))247 rows.extend(_itemlist_rows(node, base_url))248 return rows249 250 251# ---------------------------------------------------------------- microdata252 253_MICRO_MAP = {254 "title": "title",255 "name": "title",256 "jobtitle": "title",257 "datePosted": "posted_at",258 "dateposted": "posted_at",259 "employmentType": "employment_type",260 "employmenttype": "employment_type",261 "jobLocation": "location",262 "joblocation": "location",263 "addressLocality": "location",264 "addresslocality": "location",265 "occupationalCategory": "department",266 "occupationalcategory": "department",267 "identifier": "id",268 "url": "apply_url",269}270 271 272def from_microdata(root, base_url: Optional[str]) -> list[dict]:273 if root is None:274 return []275 rows: list[dict] = []276 nodes = root.xpath(277 "//*[@itemtype and contains(translate(@itemtype,'ABCDEFGHIJKLMNOPQRSTUVWXYZ',"278 "'abcdefghijklmnopqrstuvwxyz'),'jobposting')]"279 )280 for node in nodes:281 row: dict = {}282 for prop in node.xpath(".//*[@itemprop]"):283 key = _MICRO_MAP.get((prop.get("itemprop") or "").strip())284 if not key or row.get(key):285 continue286 value = (287 prop.get("content")288 or prop.get("datetime")289 or (prop.get("href") if prop.tag == "a" else None)290 or text_of(prop, max_len=200)291 )292 value = clean_text(value, strip_html=False)293 if not value:294 continue295 if key == "apply_url":296 value = urljoin(base_url or "", value)297 row[key] = value298 if row.get("title"):299 rows.append(row)300 return rows301 302 303# --------------------------------------------------------------------- feeds304 305FEED_PATHS = (306 "/jobs.rss", "/jobs/feed", "/careers/feed", "/jobs.atom", "/careers.rss",307 "/feed/jobs", "/jobs/rss", "/api/jobs.rss",308)309 310 311def discover_feeds(root, base_url: str) -> list[str]:312 urls: list[str] = []313 if root is None:314 return urls315 for link in root.xpath("//link[@rel]"):316 rel = (link.get("rel") or "").lower()317 typ = (link.get("type") or "").lower()318 href = link.get("href")319 if not href or "alternate" not in rel:320 continue321 if "rss" in typ or "atom" in typ or "xml" in typ:322 title = (link.get("title") or "").lower()323 if any(k in (href + " " + title).lower() for k in ("job", "career", "vacan", "position", "opening")):324 urls.append(urljoin(base_url, href))325 return urls326 327 328def parse_feed(xml_bytes: bytes, base_url: Optional[str]) -> list[dict]:329 try:330 from lxml import etree331 332 root = etree.fromstring(333 xml_bytes,334 parser=etree.XMLParser(resolve_entities=False, no_network=True, recover=True, huge_tree=False),335 )336 except Exception:337 return []338 if root is None:339 return []340 341 def local(el) -> str:342 tag = el.tag343 return tag.rsplit("}", 1)[-1].lower() if isinstance(tag, str) else ""344 345 rows: list[dict] = []346 for entry in root.iter():347 if local(entry) not in ("item", "entry"):348 continue349 row: dict = {}350 for child in entry:351 name = local(child)352 value = (child.text or "").strip()353 if name == "title" and value:354 row["title"] = value355 elif name == "link":356 row["apply_url"] = value or child.get("href")357 elif name in ("pubdate", "published", "updated", "date"):358 row.setdefault("posted_at", value)359 elif name in ("location", "joblocation", "city"):360 row.setdefault("location", value)361 elif name in ("category", "department"):362 row.setdefault("department", value)363 elif name in ("jobtype", "employmenttype", "type"):364 row.setdefault("employment_type", value)365 elif name == "guid" and not row.get("apply_url"):366 row["apply_url"] = value367 if row.get("apply_url") and base_url:368 row["apply_url"] = urljoin(base_url, row["apply_url"])369 if row.get("title"):370 rows.append(row)371 return rows372 373 374async def from_feeds(client: SafeClient, root, base_url: str, *, probe_common: bool = False) -> list[dict]:375 candidates = discover_feeds(root, base_url)376 if probe_common:377 for path in FEED_PATHS:378 candidates.append(urljoin(base_url, path))379 rows: list[dict] = []380 seen: set[str] = set()381 for url in candidates[:6]:382 if url in seen:383 continue384 seen.add(url)385 resp = await client.try_get(url)386 if resp is None or resp.status != 200:387 continue388 if b"<rss" not in resp.content[:2000] and b"<feed" not in resp.content[:2000]:389 continue390 rows.extend(parse_feed(resp.content, url))391 if rows:392 break393 return rows394 395 396def extract(root, base_url: Optional[str]) -> tuple[list[dict], str]:397 """Return (rows, which-source-won)."""398 rows = from_jsonld(root, base_url)399 if rows:400 return rows, "json-ld"401 rows = from_microdata(root, base_url)402 if rows:403 return rows, "microdata"404 return [], "none"405 