Team Ai
Apppublic

itismouad/pythonic-raqa-app

sourceHugging Faceopenrailupdated 3y agoView on Hugging Face
0likes
text_utils.py78 linesDownload Raw Back to aimakerspace
1import os2from typing import List3 4 5class TextFileLoader:6    def __init__(self, path: str, encoding: str = "utf-8"):7        self.documents = []8        self.path = path9        self.encoding = encoding10 11    def load(self):12        if os.path.isdir(self.path):13            self.load_directory()14        elif os.path.isfile(self.path) and self.path.endswith(".txt"):15            self.load_file()16        else:17            raise ValueError(18                "Provided path is neither a valid directory nor a .txt file."19            )20 21    def load_file(self):22        with open(self.path, "r", encoding=self.encoding) as f:23            self.documents.append(f.read())24 25    def load_directory(self):26        for root, _, files in os.walk(self.path):27            for file in files:28                if file.endswith(".txt"):29                    with open(30                        os.path.join(root, file), "r", encoding=self.encoding31                    ) as f:32                        self.documents.append(f.read())33 34    def load_documents(self):35        self.load()36        return self.documents37 38 39class CharacterTextSplitter:40    def __init__(41        self,42        chunk_size: int = 1000,43        chunk_overlap: int = 200,44    ):45        assert (46            chunk_size > chunk_overlap47        ), "Chunk size must be greater than chunk overlap"48 49        self.chunk_size = chunk_size50        self.chunk_overlap = chunk_overlap51 52    def split(self, text: str) -> List[str]:53        chunks = []54        for i in range(0, len(text), self.chunk_size - self.chunk_overlap):55            chunks.append(text[i : i + self.chunk_size])56        return chunks57 58    def split_texts(self, texts: List[str]) -> List[str]:59        chunks = []60        for text in texts:61            chunks.extend(self.split(text))62        return chunks63 64 65if __name__ == "__main__":66    loader = TextFileLoader("data/KingLear.txt")67    loader.load()68    splitter = CharacterTextSplitter()69    chunks = splitter.split_texts(loader.documents)70    print(len(chunks))71    print(chunks[0])72    print("--------")73    print(chunks[1])74    print("--------")75    print(chunks[-2])76    print("--------")77    print(chunks[-1])78