Team Ai
Apppublic

SwastikM/Embedding-Quantization

sourceHugging Faceapache-2.0updated 1y agoView on Hugging Face
0likes
binary_index.py15 linesDownload Raw Back to root
1from datasets import load_from_disk2import numpy as np3from faiss import IndexBinaryFlat, write_index_binary4from sentence_transformers.quantization import quantize_embeddings5 6import os7path_to_vectorised_dataset = os.path.join(os.getcwd(),'vectorized_dataset')8 9dataset = load_from_disk(path_to_vectorised_dataset)10embeddings = np.array(dataset["embedding"], dtype=np.float32)11 12ubinary_embeddings = quantize_embeddings(embeddings, "ubinary")13index = IndexBinaryFlat(384)    ## embedding dimension14index.add(ubinary_embeddings)15write_index_binary(index, "conala.index")