Team Ai
Datasetpublic

Gabrui/source_scripts_data_aya

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes50downloads
utils_sel.py162 linesDownload Raw Back to root
1"""Utilities Function for dealing with Selenium and generating PDF output"""2 3from io import BytesIO4from PyPDF2 import PdfMerger5from PIL import Image, ImageDraw, ImageFont6from selenium import webdriver7from selenium.webdriver.common.by import By8from selenium.webdriver.chrome.options import Options9 10 11def get_driver(production=True, width=1080, height=3840):12    """Initializes the driver with a tall headless window in production"""13    chrome_options = Options()14    if production:15        chrome_options.add_argument("--headless")16        chrome_options.add_argument(f"--window-size={width},{height}")17    return webdriver.Chrome(options=chrome_options)18 19 20def find_links_names(driver, partial_text):21    """Return all urls and full names of links that partially have the text"""22    links = driver.find_elements(By.PARTIAL_LINK_TEXT, partial_text)23    urls = [elem.get_attribute('href') for elem in links]24    names = [elem.text for elem in links]25    return urls, names26 27 28def click_button(driver, button_name):29    """Click the first button with the name attribute"""30    button = driver.find_element(By.NAME, button_name)31    button.click()32 33 34def click_link(driver, partial_text):35    """Click the first link that partially have the text"""36    link = driver.find_element(By.PARTIAL_LINK_TEXT, partial_text)37    link.click()38 39 40def get_html(driver, element):41    """Get an string of the innerHTML"""42    script = """43    var div = arguments[0];44    return div.innerHTML;45    """46    return driver.execute_script(script, element)47 48 49def get_html_with_images(driver, element):50    """Get an string of the innerHTML of the elements with embedded images"""51    script = """52    function getBase64Image(img) {53            var canvas = document.createElement("canvas");54            canvas.width = img.width;55            canvas.height = img.height;56            var ctx = canvas.getContext("2d");57            ctx.drawImage(img, 0, 0);58            return canvas.toDataURL("image/png");59    }60    var div = arguments[0];61    var images = div.getElementsByTagName('img');62    for (var i = 0; i < images.length; i++) {63            var img = images[i];64            var src = img.src;65            if (src.startsWith('http')) {66                    img.src = getBase64Image(img);67            }68    }69    return div.innerHTML;70    """71    return driver.execute_script(script, element)72 73 74def bytes_to_image(img_bytes: bytes) -> Image:75    """Reads the bytes and returns a PIL.Image"""76    return Image.open(BytesIO(img_bytes))77 78 79def image_to_bytes(img, f_type='JPEG', quality=95):80    """Convert a PIL Image to its bytes representation"""81    bytesio = BytesIO()82    img.save(bytesio, format=f_type, quality=quality)83    bytesio.seek(0)84    return bytesio.getvalue()85 86 87def take_full_div_screenshot(driver, div):88    """Takes the screenshot of the whole div, uses zoom,89    resolution may be bad if too big"""90    div_h = driver.execute_script("return arguments[0].offsetHeight;", div)91    div_w = driver.execute_script("return arguments[0].offsetWidth;", div)92    window_height = driver.execute_script("return window.innerHeight;")93    window_width = driver.execute_script("return window.innerWidth;")94    zoom_level = min(window_height / div_h, window_width / div_w, 1)95    driver.execute_script(f"document.body.style.zoom = '{zoom_level}'")96    driver.execute_script("arguments[0].scrollIntoView();", div)97    im = bytes_to_image(driver.get_screenshot_as_png())98    div_x = int(driver.execute_script("return arguments[0].getBoundingClientRect().left;", div))99    div_y = int(driver.execute_script("return arguments[0].getBoundingClientRect().top;", div))100    im = im.crop((div_x, div_y, div_x + int(div_w * zoom_level), div_y + int(div_h * zoom_level)))101    if zoom_level != 1:102        im = im.resize((div_w, div_h), Image.LANCZOS)103    driver.execute_script("document.body.style.zoom = '1'")104    return im105 106 107def hide_between_start_end(driver, parent, start: str, end: str, visible=False):108    """Hides all children of the parent that are between start (inclusive) to end (text representation)"""109    children = parent.get_property('children')110    start_index = None111    end_index = None112    for i, child in enumerate(children):113        if child.text.strip() == start:114            start_index = i115        elif child.text.strip() == end:116            end_index = i117            break118    assert start_index is not None119    assert end_index is not None120    for i in range(end_index - 1, start_index-1, -1):121        driver.execute_script(f"arguments[0].style.display = '{'' if visible else 'none'}';\122                arguments[0].style.visibility = '{'' if visible else 'hidden'}';", children[i])123 124 125def copy_element_as_first_child(driver, source_locator, destination_locator):126    """Copies the source_locator to be the first child of the destination_locator"""127    script = """128    var source = arguments[0];129    var destination = arguments[1];130    var clone = source.cloneNode(true);131    destination.insertBefore(clone, destination.firstChild);132    """133    driver.execute_script(script, source_locator, destination_locator)134 135 136def create_pdf_from_images(images_bytes, output_pdf_path, dpi=100, quality=90):137    """Create a PDF file with one image per page from the images_bytes list"""138    pdf_merger = PdfMerger()139    for img_bytes in images_bytes:140        img = bytes_to_image(img_bytes)141        pdf_buffer = BytesIO()142        img.save(pdf_buffer, format='PDF', resolution=dpi, quality=quality)143        pdf_buffer.seek(0)144        pdf_merger.append(pdf_buffer)145    pdf_merger.write(output_pdf_path)146 147 148def create_image_from_text(text, size=(800, 200), as_bytes=True):149    """Renders a text"""150    image = Image.new('RGB', size, color='white')151    draw = ImageDraw.Draw(image)152    font = ImageFont.truetype("DejaVuSans.ttf", 28)153    text_bbox = draw.textbbox((0, 0), text, font=font)154    text_width = text_bbox[2] - text_bbox[0]155    text_height = text_bbox[3] - text_bbox[1]156    position = ((size[0]-text_width)/2, (size[1]-text_height)/2)157    draw.text(position, text, fill="black", font=font)158    draw.rectangle([0, 0, size[0]-1, size[1]-1], outline="black")159    if as_bytes:160        return image_to_bytes(image)161    return image162