Team Ai
Apppublic

Akjava/open_Deep-Research-DuckDuckGo

sourceHugging Faceapache-2.0updated 2y agoView on Hugging Face
4likes
run_agents.py88 linesDownload Raw Back to scripts
1import json2import os3import shutil4import textwrap5from pathlib import Path6 7# import tqdm.asyncio8from smolagents.utils import AgentError9 10 11def serialize_agent_error(obj):12    if isinstance(obj, AgentError):13        return {"error_type": obj.__class__.__name__, "message": obj.message}14    else:15        return str(obj)16 17 18def get_image_description(file_name: str, question: str, visual_inspection_tool) -> str:19    prompt = f"""Write a caption of 5 sentences for this image. Pay special attention to any details that might be useful for someone answering the following question:20{question}. But do not try to answer the question directly!21Do not add any information that is not present in the image."""22    return visual_inspection_tool(image_path=file_name, question=prompt)23 24 25def get_document_description(file_path: str, question: str, document_inspection_tool) -> str:26    prompt = f"""Write a caption of 5 sentences for this document. Pay special attention to any details that might be useful for someone answering the following question:27{question}. But do not try to answer the question directly!28Do not add any information that is not present in the document."""29    return document_inspection_tool.forward_initial_exam_mode(file_path=file_path, question=prompt)30 31 32def get_single_file_description(file_path: str, question: str, visual_inspection_tool, document_inspection_tool):33    file_extension = file_path.split(".")[-1]34    if file_extension in ["png", "jpg", "jpeg"]:35        file_description = f" - Attached image: {file_path}"36        file_description += (37            f"\n     -> Image description: {get_image_description(file_path, question, visual_inspection_tool)}"38        )39        return file_description40    elif file_extension in ["pdf", "xls", "xlsx", "docx", "doc", "xml"]:41        file_description = f" - Attached document: {file_path}"42        image_path = file_path.split(".")[0] + ".png"43        if os.path.exists(image_path):44            description = get_image_description(image_path, question, visual_inspection_tool)45        else:46            description = get_document_description(file_path, question, document_inspection_tool)47        file_description += f"\n     -> File description: {description}"48        return file_description49    elif file_extension in ["mp3", "m4a", "wav"]:50        return f" - Attached audio: {file_path}"51    else:52        return f" - Attached file: {file_path}"53 54 55def get_zip_description(file_path: str, question: str, visual_inspection_tool, document_inspection_tool):56    folder_path = file_path.replace(".zip", "")57    os.makedirs(folder_path, exist_ok=True)58    shutil.unpack_archive(file_path, folder_path)59 60    prompt_use_files = ""61    for root, dirs, files in os.walk(folder_path):62        for file in files:63            file_path = os.path.join(root, file)64            prompt_use_files += "\n" + textwrap.indent(65                get_single_file_description(file_path, question, visual_inspection_tool, document_inspection_tool),66                prefix="    ",67            )68    return prompt_use_files69 70 71def get_tasks_to_run(data, total: int, base_filename: Path, tasks_ids: list[int]):72    f = base_filename.parent / f"{base_filename.stem}_answers.jsonl"73    done = set()74    if f.exists():75        with open(f, encoding="utf-8") as fh:76            done = {json.loads(line)["task_id"] for line in fh if line.strip()}77 78    tasks = []79    for i in range(total):80        task_id = int(data[i]["task_id"])81        if task_id not in done:82            if tasks_ids is not None:83                if task_id in tasks_ids:84                    tasks.append(data[i])85            else:86                tasks.append(data[i])87    return tasks88