Team Ai
Modelpublic

cimo001/embeddinggemma-300m

sourceHugging Facemitupdated 1mo agoView on Hugging Face
1likes17downloads
example.py76 linesDownload Raw Back to src
1import sys2import unicodedata3import numpy4import sentencepiece5 6sys.dont_write_bytecode = True7 8# Source9from helper import onnxSessionBuild10 11pathModel = "./"12 13embeddingTokenMax = 204814 15sentencepieceEmbedding = sentencepiece.SentencePieceProcessor()16sentencepieceEmbedding.Load(f"{pathModel}tokenizer.model")17 18onnxSessionEmbedding = onnxSessionBuild(f"{pathModel}onnx/model.onnx")19 20def embedding(mode, text):21    inputList = text if isinstance(text, list) else [text]22 23    inputPrefixList = []24 25    for a in range(len(inputList)):26        if mode == "document":27            inputPrefixList.append(f"title: none | text: {inputList[a]}")28        else:29            inputPrefixList.append(f"task: search result | query: {inputList[a]}")30 31    tokenList = []32    lengthMax = 033 34    for a in range(len(inputPrefixList)):35        idList = sentencepieceEmbedding.EncodeAsIds(inputPrefixList[a])36 37        if len(idList) > embeddingTokenMax - 2:38            idList = idList[0:embeddingTokenMax - 2]39 40        idList = [sentencepieceEmbedding.bos_id()] + idList + [sentencepieceEmbedding.eos_id()]41 42        if len(idList) > lengthMax:43            lengthMax = len(idList)44 45        tokenList.append(idList)46 47    inputIds = numpy.full((len(tokenList), lengthMax), sentencepieceEmbedding.pad_id(), dtype=numpy.int64)48    attentionMask = numpy.zeros((len(tokenList), lengthMax), dtype=numpy.int64)49 50    for a in range(len(tokenList)):51        inputIds[a, 0:len(tokenList[a])] = tokenList[a]52        attentionMask[a, 0:len(tokenList[a])] = 153 54    feedObject = {"input_ids": inputIds, "attention_mask": attentionMask}55 56    return onnxSessionEmbedding.run(["sentence_embedding"], feedObject)[0]57 58prompt = unicodedata.normalize("NFKC", "what is panda?")59 60textList = [61    "The giant panda (Ailuropoda melanoleuca), sometimes called a panda bear, is a bear species endemic to China.",62    "hi",63    "パンダはクマ科の哺乳類で、中国の固有種である。"64]65 66for a in range(len(textList)):67    textList[a] = unicodedata.normalize("NFKC", textList[a])68 69promptVector = embedding("query", prompt)[0]70vectorList = embedding("document", textList)71 72for a in range(len(textList)):73    score = float(numpy.dot(promptVector, vectorList[a]) / (numpy.linalg.norm(promptVector) * numpy.linalg.norm(vectorList[a])))74 75    print(f"{score:.6f} | {textList[a]}")76