prashantbhandari/rag-linux-kernel
0
1import fitz2import chromadb3from sentence_transformers import SentenceTransformer4from langchain_text_splitters import RecursiveCharacterTextSplitter5 6PDF_PATH = "linux_kernel.pdf" # place your PDF in this folder with this name7CHROMA_PATH = "./chroma_db"8 9 10def index_book():11 print("Loading PDF...")12 doc = fitz.open(PDF_PATH)13 14 pages = []15 for i, page in enumerate(doc):16 text = page.get_text()17 if text.strip():18 pages.append({"text": text, "page": i + 1})19 20 print(f"Loaded {len(pages)} pages. Splitting into chunks...")21 22 splitter = RecursiveCharacterTextSplitter(23 chunk_size=800,24 chunk_overlap=150,25 )26 27 chunks = []28 for p in pages:29 for chunk in splitter.split_text(p["text"]):30 chunks.append({"text": chunk, "page": p["page"]})31 32 print(f"Created {len(chunks)} chunks. Embedding now (this takes a few minutes)...")33 34 model = SentenceTransformer("all-MiniLM-L6-v2")35 client = chromadb.PersistentClient(path=CHROMA_PATH)36 37 try:38 client.delete_collection("linux_kernel")39 except Exception:40 pass41 42 collection = client.create_collection(43 "linux_kernel",44 metadata={"hnsw:space": "cosine"},45 )46 47 batch_size = 10048 for i in range(0, len(chunks), batch_size):49 batch = chunks[i : i + batch_size]50 texts = [c["text"] for c in batch]51 embeddings = model.encode(texts, show_progress_bar=False).tolist()52 collection.add(53 documents=texts,54 embeddings=embeddings,55 ids=[str(i + j) for j in range(len(batch))],56 metadatas=[{"page": c["page"]} for c in batch],57 )58 print(f" Indexed {min(i + batch_size, len(chunks))}/{len(chunks)} chunks")59 60 print("\nDone! chroma_db/ folder is ready. You can now run: streamlit run app.py")61 62 63if __name__ == "__main__":64 index_book()65 