sadit/TextEmbeddings
034
1import h5py as h52import json3import pandas as pd4import torch5from sentence_transformers import SentenceTransformer6 7torch.set_num_threads(32)8modelname = 'sentence-transformers/all-MiniLM-L6-v2'9model = SentenceTransformer(modelname)10with open("wikipedia-text-20231101-en.txt") as f:11 sentences = [json.loads(line) for line in f.readlines()]12 13embeddings = model.encode(sentences)14 15with h5.File("wikipedia-embeddings-20231101-en.h5", "w") as f:16 f["emb"] = embeddings17 f.attrs["model"] = modelname18 19#print(embeddings)20 21 