Team Ai
Apppublic

cj-dev-code/semantic_search

sourceHugging Faceupdated 1y agoView on Hugging Face
0likes
lc_semantic_chunker.py57 linesDownload Raw Back to scripts
1from pathlib import Path2from langchain_experimental.text_splitter import SemanticChunker3from langchain_openai.embeddings import OpenAIEmbeddings4from dotenv import load_dotenv5 6from langchain.embeddings import HuggingFaceEmbeddings7 8 9 10def load_text(path):11    with open(path, "r", encoding="utf-8") as f:12        return f.read()13def save_chunks(chunks, out_path):14    with open(out_path, "w", encoding="utf-8") as f:15        for i, chunk in enumerate(chunks):16            f.write(f"--- Chunk {i} ---\n{chunk}\n\n")17 18 19def split_into_paragraph_chunks(text, chunk_size=1000, chunk_overlap=100):20    """21    Splits input text into paragraph-level chunks using RecursiveCharacterTextSplitter.22    23    - Prioritizes splitting on double newlines (paragraphs)24    - Falls back to single newline, then sentence, then word25    - Supports overlap to preserve context26 27    Args:28        text (str): The input text to split29        chunk_size (int): Maximum size of each chunk (in characters)30        chunk_overlap (int): Number of overlapping characters between chunks31 32    Returns:33        List[str]: List of text chunks34    """35    text_splitter = SemanticChunker(OpenAIEmbeddings(),36    #text_splitter = SemanticChunker(HuggingFaceEmbeddings(model_name="BAAI/bge-base-en-v1.5"),37        breakpoint_threshold_type="standard_deviation",  # smoother, more natural breaks38        breakpoint_threshold_amount=.5,                  # lower = longer chunks39        min_chunk_size=2                                  # prevent short sentence-only chunks40            )41    42    43 44    45    return text_splitter.create_documents([text])46 47if __name__ == "__main__":48    book_path = Path("data/fullbook.txt")49    out_path = Path("data/chunks.txt")50 51 52    load_dotenv()53 54    text = load_text(book_path)55    chunks = split_into_paragraph_chunks(text)56    save_chunks(chunks, out_path)57