Team Ai
Apppublic

xdecoder/Instruct-X-Decoder

sourceHugging Faceafl-3.0updated 3y agoView on Hugging Face
163likes
misc.py64 linesDownload Raw Back to language
1import random2 3import nltk4nltk.data.path.append('/mnt/data/nltk_data')5import numpy as np6 7from utils.constants import IMAGENET_DEFAULT_TEMPLATES8 9 10def get_tag(tokenized, tags):11    if not isinstance(tags, (list, tuple)):12        tags = [tags]13    ret = []14    for (word, pos) in nltk.pos_tag(tokenized):15        for tag in tags:16            if pos == tag:17                ret.append(word)18    return ret19 20def get_noun_phrase(tokenized):21    # Taken from Su Nam Kim Paper...22    grammar = r"""23        NBAR:24            {<NN.*|JJ>*<NN.*>}  # Nouns and Adjectives, terminated with Nouns25 26        NP:27            {<NBAR>}28            {<NBAR><IN><NBAR>}  # Above, connected with in/of/etc...29    """30    chunker = nltk.RegexpParser(grammar)31 32    chunked = chunker.parse(nltk.pos_tag(tokenized))33    continuous_chunk = []34    current_chunk = []35 36    for subtree in chunked:37        if isinstance(subtree, nltk.Tree):38            current_chunk.append(' '.join([token for token, pos in subtree.leaves()]))39        elif current_chunk:40            named_entity = ' '.join(current_chunk)41            if named_entity not in continuous_chunk:42                continuous_chunk.append(named_entity)43                current_chunk = []44        else:45            continue46 47    return continuous_chunk48 49def text_noun_with_prompt_all(text, phrase_prob=0.0, append_text=True):50    tokenized = nltk.word_tokenize(text)51    52    if random.random() >= phrase_prob:53        nouns = get_tag(tokenized, ['NN', 'NNS', 'NNP'])54    else:55        nouns = get_noun_phrase(tokenized)56 57 58    prompt_texts = [np.random.choice(IMAGENET_DEFAULT_TEMPLATES).format(noun) for noun in nouns]59    60    if append_text:61        prompt_texts += [text]62        nouns += [text]63    64    return prompt_texts, nouns