ppsingh/annotation_dev
3
1import streamlit as st2import pandas as pd3from huggingface_hub import Repository4import os 5from pathlib import Path6import json7import numpy as np8 9 10 11# Declaring the variables for later use to talk to dataset12 13# the token is saved as secret key-value pair in the environment which can be access as shown below14auth_token = os.environ.get("space_to_dataset") or True15 16DATASET_REPO_URL = 'ppsingh/annotation_data' # path to dataset repo17DATA_FILENAME = "paralist.json"18DATA_FILE = os.path.join("data", DATA_FILENAME)19 20# cloning the dataset repo21 22 23# Data file name24file_name = 'paralist.json'25 26# reading the json27@st.cache(allow_output_mutation=True)28def read_dataset():29 repo = Repository( local_dir="data", clone_from=DATASET_REPO_URL, repo_type="dataset", use_auth_token= auth_token)30 with open('data/{}'.format(file_name), 'r', encoding="utf8") as json_file:31 paraList = json.load(json_file)32 33 return repo, paraList34 35st.sidebar.markdown(""" 36 # Data Annotation Demo 37This app is demo how to use the space to provide user interface for the data annotation/tagging. The data resides in repo_type 'dataset'.38""")39# sidebar with info and drop down to select from the keys40 41topic = None42repo, paraList = read_dataset()43# getting outer level keys in json 44keys = paraList.keys() 45 46if keys is not None:47 topic = st.sidebar.selectbox(label="Choose dataset topic to load", options=keys )48 49 50#with st.container():51 52 53with st.form("annotation_form"):54 if topic is not None:55 subtopics = list(paraList[topic].keys())56 #st.write(subtopics)57 val = np.random.randint(0,len(subtopics)-1)58 tag = subtopics[val]59 60 idx = np.random.randint(0,3)61 62 st.markdown("**Text**")63 st.write(paraList[topic][tag][idx]['textsegment'])64 65 st.markdown("**Tag**")66 st.write(tag)67 68 feedback = st.selectbox('0 If Tag is not a good keyword for text, 5 for prefect match',(0,1,2,3,4,5))69 submitted = st.form_submit_button("Submit")70 if submitted:71 paraList[topic][tag][idx]['annotation'].append(feedback)72 with open("data/{}".format(file_name), "w") as outfile:73 json.dump(paraList, outfile)74 repo.push_to_hub('added new annotation')75 # st.write(type(paraList))76 77 78 #c1, c2, c3 = st.columns([3, 1, 1])79 #with c1:80 # st.header('Text')81 # st.write(paraList[topic][tag][idx]['textsegment'])82 83 #with c2:84 # st.header('Tag')85 # st.text(tag)86 87 #with c3:88 # st.header('Feedback')89 # feedback = None90 # feedback = st.selectbox('0 If Tag is not a good keyword for text, 5 for prefect match',(0,1,2,3,4,5)) 91 #if feedback:92 # st.write(feedback)93# if st.button('Submit'):94# paraList[topic][choice][idx]['annotation'].append(feedback)95# with open('data/{}'.format(file_name), 'r', encoding="utf8") as json_file:96 # json.dump(paraList,json_file, ensure_ascii = True)97 # repo.push_to_hub('added new annotation')98 99#st.write(paraList) 100 #new_row = title101# data = data.append(new_row, ignore_index=True)102# st.write(data)103# st.write(os.getcwd())104# data.to_csv('test.csv', index= False)105 106 107#st.write(df)108# st.write('data/test.csv')109# iterate over files in110# that directory 111#directory = os.getcwd()112#files = Path(directory).glob('*')113#for file in files:114# st.write(file)115 116#with open(DATA_FILE, "a") as csvfile:117# writer = csv.DictWriter(csvfile, fieldnames=["Sentences"])118# writer.writerow({'Sentences': new_row})119# repo.push_to_hub('adding new line')120# st.write('Succcess')121 122 123 124 125 126 127 