xdecoder/Instruct-X-Decoder
163
1import random2 3import nltk4nltk.data.path.append('/mnt/data/nltk_data')5import numpy as np6 7from utils.constants import IMAGENET_DEFAULT_TEMPLATES8 9 10def get_tag(tokenized, tags):11 if not isinstance(tags, (list, tuple)):12 tags = [tags]13 ret = []14 for (word, pos) in nltk.pos_tag(tokenized):15 for tag in tags:16 if pos == tag:17 ret.append(word)18 return ret19 20def get_noun_phrase(tokenized):21 # Taken from Su Nam Kim Paper...22 grammar = r"""23 NBAR:24 {<NN.*|JJ>*<NN.*>} # Nouns and Adjectives, terminated with Nouns25 26 NP:27 {<NBAR>}28 {<NBAR><IN><NBAR>} # Above, connected with in/of/etc...29 """30 chunker = nltk.RegexpParser(grammar)31 32 chunked = chunker.parse(nltk.pos_tag(tokenized))33 continuous_chunk = []34 current_chunk = []35 36 for subtree in chunked:37 if isinstance(subtree, nltk.Tree):38 current_chunk.append(' '.join([token for token, pos in subtree.leaves()]))39 elif current_chunk:40 named_entity = ' '.join(current_chunk)41 if named_entity not in continuous_chunk:42 continuous_chunk.append(named_entity)43 current_chunk = []44 else:45 continue46 47 return continuous_chunk48 49def text_noun_with_prompt_all(text, phrase_prob=0.0, append_text=True):50 tokenized = nltk.word_tokenize(text)51 52 if random.random() >= phrase_prob:53 nouns = get_tag(tokenized, ['NN', 'NNS', 'NNP'])54 else:55 nouns = get_noun_phrase(tokenized)56 57 58 prompt_texts = [np.random.choice(IMAGENET_DEFAULT_TEMPLATES).format(noun) for noun in nouns]59 60 if append_text:61 prompt_texts += [text]62 nouns += [text]63 64 return prompt_texts, nouns