Team Ai
Apppublic

prashantbhandari/rag-linux-kernel

sourceHugging Faceupdated 8mo agoView on Hugging Face
0likes
indexer.py65 linesDownload Raw Back to root
1import fitz2import chromadb3from sentence_transformers import SentenceTransformer4from langchain_text_splitters import RecursiveCharacterTextSplitter5 6PDF_PATH = "linux_kernel.pdf"   # place your PDF in this folder with this name7CHROMA_PATH = "./chroma_db"8 9 10def index_book():11    print("Loading PDF...")12    doc = fitz.open(PDF_PATH)13 14    pages = []15    for i, page in enumerate(doc):16        text = page.get_text()17        if text.strip():18            pages.append({"text": text, "page": i + 1})19 20    print(f"Loaded {len(pages)} pages. Splitting into chunks...")21 22    splitter = RecursiveCharacterTextSplitter(23        chunk_size=800,24        chunk_overlap=150,25    )26 27    chunks = []28    for p in pages:29        for chunk in splitter.split_text(p["text"]):30            chunks.append({"text": chunk, "page": p["page"]})31 32    print(f"Created {len(chunks)} chunks. Embedding now (this takes a few minutes)...")33 34    model = SentenceTransformer("all-MiniLM-L6-v2")35    client = chromadb.PersistentClient(path=CHROMA_PATH)36 37    try:38        client.delete_collection("linux_kernel")39    except Exception:40        pass41 42    collection = client.create_collection(43        "linux_kernel",44        metadata={"hnsw:space": "cosine"},45    )46 47    batch_size = 10048    for i in range(0, len(chunks), batch_size):49        batch = chunks[i : i + batch_size]50        texts = [c["text"] for c in batch]51        embeddings = model.encode(texts, show_progress_bar=False).tolist()52        collection.add(53            documents=texts,54            embeddings=embeddings,55            ids=[str(i + j) for j in range(len(batch))],56            metadatas=[{"page": c["page"]} for c in batch],57        )58        print(f"  Indexed {min(i + batch_size, len(chunks))}/{len(chunks)} chunks")59 60    print("\nDone! chroma_db/ folder is ready. You can now run: streamlit run app.py")61 62 63if __name__ == "__main__":64    index_book()65