Team Ai
Datasetpublic

sadit/TextEmbeddings

sourceHugging Facemitupdated 2y agoView on Hugging Face
0likes34downloads
encode-wikipedia.py21 linesDownload Raw Back to root
1import h5py as h52import json3import pandas as pd4import torch5from sentence_transformers import SentenceTransformer6 7torch.set_num_threads(32)8modelname = 'sentence-transformers/all-MiniLM-L6-v2'9model = SentenceTransformer(modelname)10with open("wikipedia-text-20231101-en.txt") as f:11    sentences = [json.loads(line) for line in f.readlines()]12 13embeddings = model.encode(sentences)14 15with h5.File("wikipedia-embeddings-20231101-en.h5", "w") as f:16    f["emb"] = embeddings17    f.attrs["model"] = modelname18 19#print(embeddings)20 21