Team Ai
Apppublic

CVPR/Dual-Key_Backdoor_Attacks

sourceHugging Facegpl-3.0updated 4y agoView on Hugging Face
4likes
data_tools.py135 linesDownload Raw Back to utils
1"""2=========================================================================================3Trojan VQA4Written by Matthew Walmer5 6Tools to examine the VQA dataset for common words and answers7=========================================================================================8"""9import os10import re11import json12import tqdm13import numpy as np14 15from openvqa.openvqa.utils.ans_punct import prep_ans16 17# get the k most frequent answers in the train set18# check mode - lets you check how frequently a give word happens19def most_frequent_answers(k=50, verbose=False, check=None):20    file = 'data/clean/v2_mscoco_train2014_annotations.json'21    cache = 'utils/train_ans_counts.json'22    # load or compute answer counts23    if os.path.isfile(cache):24        with open(cache, 'r') as f:25            all_answers = json.load(f)26    else:27        with open(file, 'r') as f:28            data = json.load(f)29        annotations = data['annotations']30        all_answers = {}31        for anno in tqdm.tqdm(annotations):32            answers = anno['answers']33            for ans in answers:34                # Preprocessing from OpenVQA35                a = prep_ans(ans['answer'])36                if a not in all_answers:37                    all_answers[a] = 038                all_answers[a] += 139        with open(cache, 'w') as f:40            json.dump(all_answers, f)41    # find top k42    answer_list = []43    count_list = []44    for key in all_answers:45        answer_list.append(key)46        count_list.append(all_answers[key])47    count_list = np.array(count_list)48    tot_answers = np.sum(count_list)49    idx_srt = np.argsort(-1 * count_list)50    top_k = []51    for i in range(k):52        top_k.append(answer_list[idx_srt[i]])53    # check mode (helper tool)54    if check is not None:55        a = prep_ans(check)56        occ = 057        if a in all_answers:58            occ = all_answers[a]59        print('CHECKING for answer: %s'%a)60        print('occurs %i times'%occ)61        print('fraction of all answers: %f'%(float(occ)/tot_answers))62    if verbose:63        print('Top %i Answers'%k)64        print('---')65        coverage = 066        for i in range(k):67            idx = idx_srt[i]68            print('%s - %s'%(answer_list[idx], count_list[idx]))69            coverage += count_list[idx]70        print('---')71        print('Total Answers: %i'%tot_answers)72        print('Unique Answers: %i'%len(all_answers))73        print('Total Answers for Top Answers: %i'%coverage)74        print('Fraction Covered: %f'%(float(coverage)/tot_answers))75    return top_k76 77 78 79# get the k most frequent question first words in the train set80# check mode - lets you check how frequently a give word happens81def most_frequent_first_words(k=50, verbose=False, check=None):82    file = 'data/clean/v2_OpenEnded_mscoco_train2014_questions.json'83    cache = 'utils/train_fw_counts.json'84    # load or compute answer counts85    if os.path.isfile(cache):86        with open(cache, 'r') as f:87            first_words = json.load(f)88    else:89        with open(file, 'r') as f:90            data = json.load(f)91        questions = data['questions']92        first_words = {}93        for ques in tqdm.tqdm(questions):94            # pre-processing from OpenVQA:95            words = re.sub(r"([.,'!?\"()*#:;])", '', ques['question'].lower() ).replace('-', ' ').replace('/', ' ').split()96            if words[0] not in first_words:97                first_words[words[0]] = 098            first_words[words[0]] += 199        with open(cache, 'w') as f:100            json.dump(first_words, f)101    # find top k102    key_list = []103    count_list = []104    for key in first_words:105        key_list.append(key)106        count_list.append(first_words[key])107    count_list = np.array(count_list)108    tot_proc = np.sum(count_list)109    idx_srt = np.argsort(-1 * count_list)110    top_k = []111    for i in range(k):112        top_k.append(key_list[idx_srt[i]])113    # check mode (helper tool)114    if check is not None:115        w = re.sub(r"([.,'!?\"()*#:;])", '', check.lower() ).replace('-', ' ').replace('/', ' ')116        occ = 0117        if w in first_words:118            occ = first_words[w]119        print('CHECKING for word: %s'%w)120        print('occurs as first word %i times'%occ)121        print('fraction of all answers: %f'%(float(occ)/tot_proc))122    if verbose:123        print('Top %i First Words'%k)124        print('---')125        coverage = 0126        for i in range(k):127            idx = idx_srt[i]128            print('%s - %s'%(key_list[idx], count_list[idx]))129            coverage += count_list[idx]130        print('---')131        print('Total Questions: %i'%tot_proc)132        print('Unique First Words: %i'%len(first_words))133        print('Total Qs of Top Words: %i'%coverage)134        print('Fraction Covered: %f'%(float(coverage)/tot_proc))135    return top_k