cimo001/embeddinggemma-300m
117
1import sys2import unicodedata3import numpy4import sentencepiece5 6sys.dont_write_bytecode = True7 8# Source9from helper import onnxSessionBuild10 11pathModel = "./"12 13embeddingTokenMax = 204814 15sentencepieceEmbedding = sentencepiece.SentencePieceProcessor()16sentencepieceEmbedding.Load(f"{pathModel}tokenizer.model")17 18onnxSessionEmbedding = onnxSessionBuild(f"{pathModel}onnx/model.onnx")19 20def embedding(mode, text):21 inputList = text if isinstance(text, list) else [text]22 23 inputPrefixList = []24 25 for a in range(len(inputList)):26 if mode == "document":27 inputPrefixList.append(f"title: none | text: {inputList[a]}")28 else:29 inputPrefixList.append(f"task: search result | query: {inputList[a]}")30 31 tokenList = []32 lengthMax = 033 34 for a in range(len(inputPrefixList)):35 idList = sentencepieceEmbedding.EncodeAsIds(inputPrefixList[a])36 37 if len(idList) > embeddingTokenMax - 2:38 idList = idList[0:embeddingTokenMax - 2]39 40 idList = [sentencepieceEmbedding.bos_id()] + idList + [sentencepieceEmbedding.eos_id()]41 42 if len(idList) > lengthMax:43 lengthMax = len(idList)44 45 tokenList.append(idList)46 47 inputIds = numpy.full((len(tokenList), lengthMax), sentencepieceEmbedding.pad_id(), dtype=numpy.int64)48 attentionMask = numpy.zeros((len(tokenList), lengthMax), dtype=numpy.int64)49 50 for a in range(len(tokenList)):51 inputIds[a, 0:len(tokenList[a])] = tokenList[a]52 attentionMask[a, 0:len(tokenList[a])] = 153 54 feedObject = {"input_ids": inputIds, "attention_mask": attentionMask}55 56 return onnxSessionEmbedding.run(["sentence_embedding"], feedObject)[0]57 58prompt = unicodedata.normalize("NFKC", "what is panda?")59 60textList = [61 "The giant panda (Ailuropoda melanoleuca), sometimes called a panda bear, is a bear species endemic to China.",62 "hi",63 "パンダはクマ科の哺乳類で、中国の固有種である。"64]65 66for a in range(len(textList)):67 textList[a] = unicodedata.normalize("NFKC", textList[a])68 69promptVector = embedding("query", prompt)[0]70vectorList = embedding("document", textList)71 72for a in range(len(textList)):73 score = float(numpy.dot(promptVector, vectorList[a]) / (numpy.linalg.norm(promptVector) * numpy.linalg.norm(vectorList[a])))74 75 print(f"{score:.6f} | {textList[a]}")76 