Team Ai
Apppublic

nqtruong/Job_Knowledge_Graph

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
utils.py171 linesDownload Raw Back to scrape_data_indeed
1from selenium import webdriver2from time import sleep3import random4from selenium.common.exceptions import NoSuchElementException, ElementNotInteractableException5from selenium.webdriver.common.by import By6from selenium.webdriver.edge.options import Options7from selenium.webdriver.common.keys import Keys8from bs4 import BeautifulSoup9import re10import datetime11import json    12import os13from datetime import datetime, timedelta14 15 16 17def init_driver():18    19    chrome_options = Options()20    options = [21        "--headless",22        "--disable-gpu",23        "--window-size=1920,1200",24        "--ignore-certificate-errors",25        "--disable-extensions",26        "--no-sandbox",27        "--disable-dev-shm-usage"28    ]29    30    chrome_options.add_argument("user-agent=Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36")31 32    for option in options:33        chrome_options.add_argument(option)34 35    driver = webdriver.Edge(options=chrome_options)36    37    return driver38 39 40#____________________________________________41 42def access(driver,url):43    print("_"*30, "ACCESS URL","_"*30)44    driver.get(url)45    sleep(15)46 47 48def search(driver, job, location):49    print("_"*30, "SEARCH","_"*30)50    51    search_box_job = driver.find_element(By.XPATH, '//input[@id="text-input-what"]')52    search_box_location=driver.find_element(By.XPATH, '//input[@id="text-input-where"]')53    search_box_job.send_keys(job)54    search_box_location.send_keys(location)55 56    search_box_location.send_keys(Keys.RETURN)57    driver.implicitly_wait(8)58    59 60 61def save_data(dict_jd):62    directory = './data'63    64    if not os.path.exists(directory):65        os.makedirs(directory)66        67    today = datetime.today().strftime('%Y_%m_%d')68    filename = f"{directory}/data_{today}.json"69    70    json_file = json.dumps(dict_jd, indent= 4, ensure_ascii=False)71    72    with open(filename, "w", encoding="utf-8") as f:73        f.write(json_file)74    75 76def info_job(driver):77    78    # id=079 80    num_job= driver.find_element(By.XPATH, '//div[@class="jobsearch-JobCountAndSortPane-jobCount css-13jafh6 eu4oa1w0"]//span').text81    num_job_=re.sub(r'\D', '', num_job)82    num_job=int(num_job_)83    num_next= num_job//1584    85 86    if num_next >15 :87        num_next=1588    89    dict_job={}90    for i in range(0,num_next-2):91        info_jobs = driver.find_elements(By.XPATH, '//div[@class="job_seen_beacon"]')92        print("_"*30, "START","_"*30)93        94        95        try:96            close = driver.find_element(By.XPATH, '//button[@aria-label="close"]')97            close.click()98        except NoSuchElementException:99            pass100        101        for element in info_jobs:102            element.click()  103            try:              104                today = datetime.today()105                date_post= element.find_element(By.XPATH, './/span[@data-testid="myJobsStateDate"]').text106                date_post_=re.sub(r'\D', '', date_post)107                if date_post_ != "":108                    109                    posted_date = today - timedelta(days=int(date_post_))110                    posted_date_str = posted_date.strftime('%Y-%m-%d')111                else:112                    posted_date_str=today.strftime('%Y-%m-%d')113               114                115                116                117                name_job_ = driver.find_element(By.XPATH, '//h2[@data-testid="jobsearch-JobInfoHeader-title"]/span').text118                name_job = name_job_.replace("- job post", "").strip()119                120                name_company = driver.find_element(By.XPATH, '//div[@data-testid="inlineHeader-companyName"]/span/a').text121 122                location = driver.find_element(By.XPATH, '//div[@data-testid="inlineHeader-companyLocation"]/div').text123                124                125        126                job_description = driver.find_elements(By.XPATH, '//div[@id="jobDescriptionText"]')127 128                129                content_jd = ""130                for jd in job_description:131                    get_html = jd.get_attribute("innerHTML")             132                    parser = BeautifulSoup(get_html, 'html.parser')        133                    jd = parser.get_text()    134                    content_jd += jd.replace("\n"," ").replace("   ","")135                # id+=1136                id=name_company+'@'+name_job137                138                try:139                    dict_job[id]140                except KeyError:141                    dict_job[id] = {142                        "ID":id,143                        "job":name_job,144                        "company": name_company,145                        "location": location,146                        "job_description":content_jd,147                        "date_post": posted_date_str148                        149                    }150                151                sleep(4)152            except NoSuchElementException:153                pass154        155        try:156            next = driver.find_element(By.XPATH, '//a[@data-testid="pagination-page-next"]')157            next.click()158            sleep(4)159        except NoSuchElementException:160            break;161        try:162            close = driver.find_element(By.XPATH, '//button[@aria-label="close"]')163            close.click()164        except NoSuchElementException:165            pass166    167    driver.quit() 168    return dict_job169        170 171