CVPR/Dual-Key_Backdoor_Attacks
4
1"""2=========================================================================================3Trojan VQA4Written by Matthew Walmer5 6Tools to examine the VQA dataset for common words and answers7=========================================================================================8"""9import os10import re11import json12import tqdm13import numpy as np14 15from openvqa.openvqa.utils.ans_punct import prep_ans16 17# get the k most frequent answers in the train set18# check mode - lets you check how frequently a give word happens19def most_frequent_answers(k=50, verbose=False, check=None):20 file = 'data/clean/v2_mscoco_train2014_annotations.json'21 cache = 'utils/train_ans_counts.json'22 # load or compute answer counts23 if os.path.isfile(cache):24 with open(cache, 'r') as f:25 all_answers = json.load(f)26 else:27 with open(file, 'r') as f:28 data = json.load(f)29 annotations = data['annotations']30 all_answers = {}31 for anno in tqdm.tqdm(annotations):32 answers = anno['answers']33 for ans in answers:34 # Preprocessing from OpenVQA35 a = prep_ans(ans['answer'])36 if a not in all_answers:37 all_answers[a] = 038 all_answers[a] += 139 with open(cache, 'w') as f:40 json.dump(all_answers, f)41 # find top k42 answer_list = []43 count_list = []44 for key in all_answers:45 answer_list.append(key)46 count_list.append(all_answers[key])47 count_list = np.array(count_list)48 tot_answers = np.sum(count_list)49 idx_srt = np.argsort(-1 * count_list)50 top_k = []51 for i in range(k):52 top_k.append(answer_list[idx_srt[i]])53 # check mode (helper tool)54 if check is not None:55 a = prep_ans(check)56 occ = 057 if a in all_answers:58 occ = all_answers[a]59 print('CHECKING for answer: %s'%a)60 print('occurs %i times'%occ)61 print('fraction of all answers: %f'%(float(occ)/tot_answers))62 if verbose:63 print('Top %i Answers'%k)64 print('---')65 coverage = 066 for i in range(k):67 idx = idx_srt[i]68 print('%s - %s'%(answer_list[idx], count_list[idx]))69 coverage += count_list[idx]70 print('---')71 print('Total Answers: %i'%tot_answers)72 print('Unique Answers: %i'%len(all_answers))73 print('Total Answers for Top Answers: %i'%coverage)74 print('Fraction Covered: %f'%(float(coverage)/tot_answers))75 return top_k76 77 78 79# get the k most frequent question first words in the train set80# check mode - lets you check how frequently a give word happens81def most_frequent_first_words(k=50, verbose=False, check=None):82 file = 'data/clean/v2_OpenEnded_mscoco_train2014_questions.json'83 cache = 'utils/train_fw_counts.json'84 # load or compute answer counts85 if os.path.isfile(cache):86 with open(cache, 'r') as f:87 first_words = json.load(f)88 else:89 with open(file, 'r') as f:90 data = json.load(f)91 questions = data['questions']92 first_words = {}93 for ques in tqdm.tqdm(questions):94 # pre-processing from OpenVQA:95 words = re.sub(r"([.,'!?\"()*#:;])", '', ques['question'].lower() ).replace('-', ' ').replace('/', ' ').split()96 if words[0] not in first_words:97 first_words[words[0]] = 098 first_words[words[0]] += 199 with open(cache, 'w') as f:100 json.dump(first_words, f)101 # find top k102 key_list = []103 count_list = []104 for key in first_words:105 key_list.append(key)106 count_list.append(first_words[key])107 count_list = np.array(count_list)108 tot_proc = np.sum(count_list)109 idx_srt = np.argsort(-1 * count_list)110 top_k = []111 for i in range(k):112 top_k.append(key_list[idx_srt[i]])113 # check mode (helper tool)114 if check is not None:115 w = re.sub(r"([.,'!?\"()*#:;])", '', check.lower() ).replace('-', ' ').replace('/', ' ')116 occ = 0117 if w in first_words:118 occ = first_words[w]119 print('CHECKING for word: %s'%w)120 print('occurs as first word %i times'%occ)121 print('fraction of all answers: %f'%(float(occ)/tot_proc))122 if verbose:123 print('Top %i First Words'%k)124 print('---')125 coverage = 0126 for i in range(k):127 idx = idx_srt[i]128 print('%s - %s'%(key_list[idx], count_list[idx]))129 coverage += count_list[idx]130 print('---')131 print('Total Questions: %i'%tot_proc)132 print('Unique First Words: %i'%len(first_words))133 print('Total Qs of Top Words: %i'%coverage)134 print('Fraction Covered: %f'%(float(coverage)/tot_proc))135 return top_k