cj-dev-code/semantic_search
0
1from pathlib import Path2from langchain_experimental.text_splitter import SemanticChunker3from langchain_openai.embeddings import OpenAIEmbeddings4from dotenv import load_dotenv5 6from langchain.embeddings import HuggingFaceEmbeddings7 8 9 10def load_text(path):11 with open(path, "r", encoding="utf-8") as f:12 return f.read()13def save_chunks(chunks, out_path):14 with open(out_path, "w", encoding="utf-8") as f:15 for i, chunk in enumerate(chunks):16 f.write(f"--- Chunk {i} ---\n{chunk}\n\n")17 18 19def split_into_paragraph_chunks(text, chunk_size=1000, chunk_overlap=100):20 """21 Splits input text into paragraph-level chunks using RecursiveCharacterTextSplitter.22 23 - Prioritizes splitting on double newlines (paragraphs)24 - Falls back to single newline, then sentence, then word25 - Supports overlap to preserve context26 27 Args:28 text (str): The input text to split29 chunk_size (int): Maximum size of each chunk (in characters)30 chunk_overlap (int): Number of overlapping characters between chunks31 32 Returns:33 List[str]: List of text chunks34 """35 text_splitter = SemanticChunker(OpenAIEmbeddings(),36 #text_splitter = SemanticChunker(HuggingFaceEmbeddings(model_name="BAAI/bge-base-en-v1.5"),37 breakpoint_threshold_type="standard_deviation", # smoother, more natural breaks38 breakpoint_threshold_amount=.5, # lower = longer chunks39 min_chunk_size=2 # prevent short sentence-only chunks40 )41 42 43 44 45 return text_splitter.create_documents([text])46 47if __name__ == "__main__":48 book_path = Path("data/fullbook.txt")49 out_path = Path("data/chunks.txt")50 51 52 load_dotenv()53 54 text = load_text(book_path)55 chunks = split_into_paragraph_chunks(text)56 save_chunks(chunks, out_path)57 