Team Ai
Apppublic

ppsingh/annotation_dev

sourceHugging Faceupdated 3y agoView on Hugging Face
3likes
app.py127 linesDownload Raw Back to root
1import streamlit as st2import pandas as pd3from huggingface_hub import Repository4import os 5from pathlib import Path6import json7import numpy as np8 9 10 11# Declaring the variables for later use to talk to dataset12 13# the token is saved as secret key-value pair in the environment which can be access as shown below14auth_token = os.environ.get("space_to_dataset") or True15 16DATASET_REPO_URL = 'ppsingh/annotation_data'   # path to dataset repo17DATA_FILENAME = "paralist.json"18DATA_FILE = os.path.join("data", DATA_FILENAME)19 20# cloning the dataset repo21 22 23# Data file name24file_name = 'paralist.json'25 26# reading the json27@st.cache(allow_output_mutation=True)28def read_dataset():29    repo = Repository( local_dir="data", clone_from=DATASET_REPO_URL, repo_type="dataset", use_auth_token= auth_token)30    with open('data/{}'.format(file_name), 'r', encoding="utf8") as json_file:31      paraList = json.load(json_file)32    33    return repo, paraList34 35st.sidebar.markdown(""" 36   # Data Annotation Demo 37This app is demo how to use the space to provide user interface for the data annotation/tagging. The data resides in repo_type 'dataset'.38""")39# sidebar with info and drop down to select from the keys40 41topic = None42repo, paraList = read_dataset()43# getting outer level keys in json 44keys = paraList.keys()  45 46if keys is not None:47  topic = st.sidebar.selectbox(label="Choose dataset topic to load", options=keys )48 49 50#with st.container():51 52      53with st.form("annotation_form"):54    if topic is not None:55        subtopics = list(paraList[topic].keys())56  #st.write(subtopics)57    val = np.random.randint(0,len(subtopics)-1)58    tag = subtopics[val]59  60    idx = np.random.randint(0,3)61    62    st.markdown("**Text**")63    st.write(paraList[topic][tag][idx]['textsegment'])64    65    st.markdown("**Tag**")66    st.write(tag)67    68    feedback = st.selectbox('0 If Tag is not a good keyword for text, 5 for prefect match',(0,1,2,3,4,5))69    submitted = st.form_submit_button("Submit")70    if submitted:71        paraList[topic][tag][idx]['annotation'].append(feedback)72        with open("data/{}".format(file_name), "w") as outfile:73            json.dump(paraList, outfile)74        repo.push_to_hub('added new annotation')75              # st.write(type(paraList))76    77      78      #c1, c2, c3 = st.columns([3, 1, 1])79      #with c1:80       #   st.header('Text')81        #  st.write(paraList[topic][tag][idx]['textsegment'])82  83      #with c2:84       #   st.header('Tag')85        #  st.text(tag)86  87      #with c3:88       #   st.header('Feedback')89        #  feedback = None90         # feedback = st.selectbox('0 If Tag is not a good keyword for text, 5 for prefect match',(0,1,2,3,4,5)) 91          #if feedback:92           #   st.write(feedback)93#      if st.button('Submit'):94#        paraList[topic][choice][idx]['annotation'].append(feedback)95#      with open('data/{}'.format(file_name), 'r', encoding="utf8") as json_file:96 #       json.dump(paraList,json_file, ensure_ascii = True)97  #      repo.push_to_hub('added new annotation')98        99#st.write(paraList)      100    #new_row  = title101#  data = data.append(new_row, ignore_index=True)102#  st.write(data)103#  st.write(os.getcwd())104#  data.to_csv('test.csv', index= False)105 106 107#st.write(df)108#   st.write('data/test.csv')109# iterate over files in110# that directory        111#directory = os.getcwd()112#files = Path(directory).glob('*')113#for file in files:114#    st.write(file)115 116#with open(DATA_FILE, "a") as csvfile:117#  writer = csv.DictWriter(csvfile, fieldnames=["Sentences"])118#  writer.writerow({'Sentences': new_row})119#  repo.push_to_hub('adding new line')120#  st.write('Succcess')121 122 123        124 125 126 127