codekingpro/portable-devtools
114k
1from typing import Any, Iterator, List, Optional2 3from langchain_core.documents import Document4 5from langchain_community.document_loaders.base import BaseLoader6from langchain_community.utilities.arxiv import ArxivAPIWrapper7 8 9class ArxivLoader(BaseLoader):10 """Load a query result from `Arxiv`.11 The loader converts the original PDF format into the text.12 13 Setup:14 Install ``arxiv`` and ``PyMuPDF`` packages.15 ``PyMuPDF`` transforms PDF files downloaded from the arxiv.org site16 into the text format.17 18 .. code-block:: bash19 20 pip install -U arxiv pymupdf21 22 23 Instantiate:24 .. code-block:: python25 26 from langchain_community.document_loaders import ArxivLoader27 28 loader = ArxivLoader(29 query="reasoning",30 # load_max_docs=2,31 # load_all_available_meta=False32 )33 34 Load:35 .. code-block:: python36 37 docs = loader.load()38 print(docs[0].page_content[:100])39 print(docs[0].metadata)40 41 .. code-block:: python42 Understanding the Reasoning Ability of Language Models43 From the Perspective of Reasoning Paths Aggre44 {45 'Published': '2024-02-29',46 'Title': 'Understanding the Reasoning Ability of Language Models From the47 Perspective of Reasoning Paths Aggregation',48 'Authors': 'Xinyi Wang, Alfonso Amayuelas, Kexun Zhang, Liangming Pan,49 Wenhu Chen, William Yang Wang',50 'Summary': 'Pre-trained language models (LMs) are able to perform complex reasoning51 without explicit fine-tuning...'52 }53 54 55 Lazy load:56 .. code-block:: python57 58 docs = []59 docs_lazy = loader.lazy_load()60 61 # async variant:62 # docs_lazy = await loader.alazy_load()63 64 for doc in docs_lazy:65 docs.append(doc)66 print(docs[0].page_content[:100])67 print(docs[0].metadata)68 69 .. code-block:: python70 71 Understanding the Reasoning Ability of Language Models72 From the Perspective of Reasoning Paths Aggre73 {74 'Published': '2024-02-29',75 'Title': 'Understanding the Reasoning Ability of Language Models From the76 Perspective of Reasoning Paths Aggregation',77 'Authors': 'Xinyi Wang, Alfonso Amayuelas, Kexun Zhang, Liangming Pan,78 Wenhu Chen, William Yang Wang',79 'Summary': 'Pre-trained language models (LMs) are able to perform complex reasoning80 without explicit fine-tuning...'81 }82 83 Async load:84 .. code-block:: python85 86 docs = await loader.aload()87 print(docs[0].page_content[:100])88 print(docs[0].metadata)89 90 .. code-block:: python91 92 Understanding the Reasoning Ability of Language Models93 From the Perspective of Reasoning Paths Aggre94 {95 'Published': '2024-02-29',96 'Title': 'Understanding the Reasoning Ability of Language Models From the97 Perspective of Reasoning Paths Aggregation',98 'Authors': 'Xinyi Wang, Alfonso Amayuelas, Kexun Zhang, Liangming Pan,99 Wenhu Chen, William Yang Wang',100 'Summary': 'Pre-trained language models (LMs) are able to perform complex reasoning101 without explicit fine-tuning...'102 }103 104 Use summaries of articles as docs:105 .. code-block:: python106 107 from langchain_community.document_loaders import ArxivLoader108 109 loader = ArxivLoader(110 query="reasoning"111 )112 113 docs = loader.get_summaries_as_docs()114 print(docs[0].page_content[:100])115 print(docs[0].metadata)116 117 .. code-block:: python118 119 Pre-trained language models (LMs) are able to perform complex reasoning120 without explicit fine-tuning121 {122 'Entry ID': 'http://arxiv.org/abs/2402.03268v2',123 'Published': datetime.date(2024, 2, 29),124 'Title': 'Understanding the Reasoning Ability of Language Models From the125 Perspective of Reasoning Paths Aggregation',126 'Authors': 'Xinyi Wang, Alfonso Amayuelas, Kexun Zhang, Liangming Pan,127 Wenhu Chen, William Yang Wang'128 }129 """ # noqa: E501130 131 def __init__(132 self, query: str, doc_content_chars_max: Optional[int] = None, **kwargs: Any133 ):134 """Initialize with search query to find documents in the Arxiv.135 Supports all arguments of `ArxivAPIWrapper`.136 137 Args:138 query: free text which used to find documents in the Arxiv139 doc_content_chars_max: cut limit for the length of a document's content140 """ # noqa: E501141 142 self.query = query143 self.client = ArxivAPIWrapper(144 doc_content_chars_max=doc_content_chars_max, **kwargs145 )146 147 def lazy_load(self) -> Iterator[Document]:148 """Lazy load Arvix documents"""149 yield from self.client.lazy_load(self.query)150 151 def get_summaries_as_docs(self) -> List[Document]:152 """Uses papers summaries as documents rather than source Arvix papers"""153 return self.client.get_summaries_as_docs(self.query)154 