Team Ai
Apppublic

cj-dev-code/semantic_search

sourceHugging Faceupdated 1y agoView on Hugging Face
0likes
lc_semantic_chunk_embedder.py88 linesDownload Raw Back to scripts
1from pathlib import Path2from langchain_experimental.text_splitter import SemanticChunker3from langchain_openai.embeddings import OpenAIEmbeddings4from dotenv import load_dotenv5import voyageai6import json7 8 9 10 11def load_text(path):12    with open(path, "r", encoding="utf-8") as f:13        return f.read()14 15def save_chunks(chunks, out_path):16    with open(out_path, "w", encoding="utf-8") as f:17        for i, chunk in enumerate(chunks):18            f.write(f"--- Chunk {i} ---\n{chunk}\n\n")19 20 21'''22mutator23'''24def embed_and_index_chunks(chunks):25    vo = voyageai.Client()26    chunks = [chunk.page_content if hasattr(chunk, "page_content") else str(chunk) for chunk in chunks]27    embeddings = []28 29    for start in range(0, len(chunks), 95):30        embeddings.extend(vo.embed(texts=chunks[95*start:95*(start+1)], model="voyage-3.5", input_type="document"))31 32    indexed_chunks = []    33    for i in range(len(chunks)):34        indexed_chunks.append( {35                "id": i,36                "text":chunks[i],37                'embedding':embeddings[i],38                 'source':bookname39                40        })41    return indexed_chunks42 43 44    45    46 47def split_into_paragraph_chunks(text, chunk_size=1000, chunk_overlap=100):48    """49    Splits input text into paragraph-level chunks using RecursiveCharacterTextSplitter.50    51    - Prioritizes splitting on double newlines (paragraphs)52    - Falls back to single newline, then sentence, then word53    - Supports overlap to preserve context54 55    Args:56        text (str): The input text to split57        chunk_size (int): Maximum size of each chunk (in characters)58        chunk_overlap (int): Number of overlapping characters between chunks59 60    Returns:61        List[str]: List of text chunks62    """63    text_splitter = SemanticChunker(OpenAIEmbeddings(model="text-embedding-3-large"),64    #text_splitter = SemanticChunker(HuggingFaceEmbeddings(model_name="BAAI/bge-base-en-v1.5"),65        breakpoint_threshold_type="standard_deviation",  # smoother, more natural breaks66        breakpoint_threshold_amount=.5,                  # lower = longer chunks67        min_chunk_size=2                                  # prevent short sentence-only chunks68            )69    return text_splitter.create_documents([text])70 71def write_index(indexed_chunks):72    with open("data/chunks_indexed.jsonl", "w", encoding="utf-8") as f:73        for doc in indexed_chunks:74            json.dump(doc, f)75            f.write("\n")76 77if __name__ == "__main__":78    bookname = 'fullbook'79    book_path = Path(f"data/{bookname}.txt")80    out_path = Path("data/chunks.txt")81 82    load_dotenv()83 84    text = load_text(book_path)85    chunks = split_into_paragraph_chunks(text)86    index = embed_and_index_chunks(chunks)87    save_chunks(chunks, out_path)88    write_index(index)