GEM/DatasetCardForm
9
1import streamlit as st2 3from .streamlit_utils import (4 make_multiselect,5 make_selectbox,6 make_text_area,7 make_text_input,8 make_radio,9)10 11N_FIELDS_PREVIOUS = 312N_FIELDS_UNDERSERVED_COMMUNITIES = 213N_FIELDS_BIASES= 314 15N_FIELDS = N_FIELDS_PREVIOUS + N_FIELDS_UNDERSERVED_COMMUNITIES + N_FIELDS_BIASES16 17def context_page():18 st.session_state.card_dict["context"] = st.session_state.card_dict.get(19 "context", {}20 )21 with st.expander("Previous Work on the Social Impact of the Dataset", expanded=False):22 key_pref = ["context", "previous"]23 st.session_state.card_dict["context"]["previous"] = st.session_state.card_dict[24 "context"25 ].get("previous", {})26 27 make_radio(28 label="Are you aware of cases where models trained on the task featured in this dataset ore related tasks have been used in automated systems?",29 options=["no", "yes - related tasks", "yes - other datasets featuring the same task", "yes - models trained on this dataset"],30 key_list=key_pref + ["is-deployed"],31 help="",32 )33 if "yes" in st.session_state.card_dict["context"]["previous"]["is-deployed"]:34 make_text_area(35 label="Did any of these previous uses result in observations about the social impact of the systems? " + \36 "In particular, has there been work outlining the risks and limitations of the system? Provide links and descriptions here.",37 key_list=key_pref + ["described-risks"],38 help="",39 )40 if st.session_state.card_dict["context"]["previous"]["is-deployed"] == "yes - models trained on this dataset":41 make_text_area(42 label="Have any changes been made to the dataset as a result of these observations?",43 key_list=key_pref + ["changes-from-observation"],44 help="",45 )46 else:47 st.session_state.card_dict["context"]["previous"]["changes-from-observation"] = "N/A"48 else:49 st.session_state.card_dict["context"]["previous"]["described-risks"] = "N/A"50 st.session_state.card_dict["context"]["previous"]["changes-from-observation"] = "N/A"51 52 with st.expander("Impact on Under-Served Communities", expanded=False):53 key_pref = ["context", "underserved"]54 st.session_state.card_dict["context"]["underserved"] = st.session_state.card_dict[55 "context"56 ].get("underserved", {})57 make_radio(58 label="Does this dataset address the needs of communities that are traditionally underserved in language technology, and particularly language generation technology?" + \59 "Communities may be underserved for exemple because their language, language variety, or social or geographical context is underepresented in NLP and NLG resources (datasets and models).",60 options=["no", "yes"],61 key_list=key_pref+["helps-underserved"],62 )63 if st.session_state.card_dict["context"]["underserved"]["helps-underserved"] == "yes":64 make_text_area(65 label="Describe how this dataset addresses the needs of underserved communities.",66 key_list=key_pref+["underserved-description"],67 )68 else:69 st.session_state.card_dict["context"]["underserved"]["underserved-description"] = "N/A"70 71 with st.expander("Discussion of Biases", expanded=False):72 key_pref = ["context", "biases"]73 st.session_state.card_dict["context"]["biases"] = st.session_state.card_dict[74 "context"75 ].get("biases", {})76 make_radio(77 label="Are there documented social biases in the dataset? " + \78 "Biases in this context are variations in the ways members of different social categories are represented that can have harmful downstream consequences for members of the more disadvantaged group.",79 options=["yes", "unsure", "no"],80 key_list=key_pref + ["has-biases"],81 help="For a more extensive definition of social biases, see [Language (Technology) is Power: A Critical Survey of “Bias” in NLP ](https://aclanthology.org/2020.acl-main.485.pdf)",82 )83 if st.session_state.card_dict["context"]["biases"]["has-biases"] == "yes":84 make_text_area(85 label="Provide links to and summaries of works analyzing these biases.",86 key_list=key_pref + ["bias-analyses"],87 help="The analyses can take the form of academic papers or news articles, or even blog posts.",88 )89 else:90 st.session_state.card_dict["context"]["biases"]["bias-analyses"] = "N/A"91 make_text_area(92 label="Does the distribution of language producers in the dataset accurately represent the full distribution of speakers of the language world-wide? If not, how does it differ?",93 key_list=key_pref + ["speaker-distibution"],94 help="For example, are most speakers in the dataset of a certain gender or located in a certain county?",95 )96 97 98def context_summary():99 total_filled = sum(100 [len(dct) for dct in st.session_state.card_dict.get("context", {}).values()]101 )102 with st.expander(103 f"Broader Social Context Completion - {total_filled} of {N_FIELDS}", expanded=False104 ):105 completion_markdown = ""106 completion_markdown += (107 f"- **Overall completion:**\n - {total_filled} of {N_FIELDS} fields\n"108 )109 completion_markdown += f"- **Sub-section - Previous Work on the Social Impact of the Dataset:**\n - {len(st.session_state.card_dict.get('context', {}).get('previous', {}))} of {N_FIELDS_PREVIOUS} fields\n"110 completion_markdown += f"- **Sub-section - Impact on Under-Served Communities:**\n - {len(st.session_state.card_dict.get('context', {}).get('underserved', {}))} of {N_FIELDS_UNDERSERVED_COMMUNITIES} fields\n"111 completion_markdown += f"- **Sub-section - Discussion of Biases:**\n - {len(st.session_state.card_dict.get('context', {}).get('biases', {}))} of {N_FIELDS_BIASES} fields\n"112 st.markdown(completion_markdown)113 