GEM/DatasetCardForm
9
1import streamlit as st2 3from .streamlit_utils import (4 make_multiselect,5 make_selectbox,6 make_text_area,7 make_text_input,8 make_radio,9)10 11N_FIELDS_PII = 112N_FIELDS_LICENSES = 213N_FIELDS_LIMITATIONS = 314 15N_FIELDS = N_FIELDS_PII + N_FIELDS_LICENSES + N_FIELDS_LIMITATIONS16 17 18def considerations_page():19 st.session_state.card_dict["considerations"] = st.session_state.card_dict.get(20 "considerations", {}21 )22 with st.expander("PII Risks and Liability", expanded=False):23 key_pref = ["considerations", "pii"]24 st.session_state.card_dict["considerations"]["pii"] = st.session_state.card_dict[25 "considerations"26 ].get("pii", {})27 make_text_area(28 label="Considering your answers to the PII part of the Data Curation Section, describe any potential privacy to the data subjects and creators risks when using the dataset.",29 key_list=key_pref+["risks-description"],30 help="In terms for example of having models memorize private information of data subjects or other breaches of privacy."31 )32 33 with st.expander("Licenses", expanded=False):34 key_pref = ["considerations", "licenses"]35 st.session_state.card_dict["considerations"]["licenses"] = st.session_state.card_dict[36 "considerations"37 ].get("licenses", {})38 39 make_multiselect(40 label="Based on your answers in the Intended Use part of the Data Overview Section, which of the following best describe the copyright and licensing status of the dataset?",41 options=[42 "public domain",43 "multiple licenses",44 "copyright - all rights reserved",45 "open license - commercial use allowed",46 "research use only",47 "non-commercial use only",48 "do not distribute",49 "other",50 ],51 key_list=key_pref + ["dataset-restrictions"],52 help="Does the license restrict how the dataset can be used?",53 )54 if "other" in st.session_state.card_dict["considerations"]["licenses"].get("dataset-restrictions", []):55 make_text_area(56 label="You selected `other` for the dataset licensing status, please elaborate here:",57 key_list=key_pref+["dataset-restrictions-other"]58 )59 else:60 st.session_state.card_dict["considerations"]["licenses"]["dataset-restrictions-other"] = "N/A"61 62 make_multiselect(63 label="Based on your answers in the Language part of the Data Curation Section, which of the following best describe the copyright and licensing status of the underlying language data?",64 options=[65 "public domain",66 "multiple licenses",67 "copyright - all rights reserved",68 "open license - commercial use allowed",69 "research use only",70 "non-commercial use only",71 "do not distribute",72 "other",73 ],74 key_list=key_pref + ["data-copyright"],75 help="For example if the dataset uses data from Wikipedia, we are asking about the status of Wikipedia text in general.",76 )77 if "other" in st.session_state.card_dict["considerations"]["licenses"].get("data-copyright", []):78 make_text_area(79 label="You selected `other` for the source data licensing status, please elaborate here:",80 key_list=key_pref+["data-copyright-other"]81 )82 else:83 st.session_state.card_dict["considerations"]["licenses"]["data-copyright-other"] = "N/A"84 85 with st.expander("Known Technical Limitations", expanded=False):86 key_pref = ["considerations", "limitations"]87 st.session_state.card_dict["considerations"]["limitations"] = st.session_state.card_dict[88 "considerations"89 ].get("limitations", {})90 make_text_area(91 label="Describe any known technical limitations, such as spurrious correlations, train/test overlap, annotation biases, or mis-annotations, " + \92 "and cite the works that first identified these limitations when possible.",93 key_list=key_pref + ["data-technical-limitations"],94 help="Outline any properties of the dataset that might lead a trained model with good performance on the metric to not behave as expected.",95 )96 make_text_area(97 label="When using a model trained on this dataset in a setting where users or the public may interact with its predictions, what are some pitfalls to look out for? " + \98 "In particular, describe some applications of the general task featured in this dataset that its curation or properties make it less suitable for.",99 key_list=key_pref + ["data-unsuited-applications"],100 help="For example, outline language varieties or domains that the model might underperform for.",101 )102 make_text_area(103 label="What are some discouraged use cases of a model trained to maximize the proposed metrics on this dataset? " +104 "In particular, think about settings where decisions made by a model that performs reasonably well on the metric my still have strong negative consequences for user or members of the public.",105 key_list=key_pref + ["data-discouraged-use"],106 help="For example, think about application settings where certain types of mistakes (such as missing a negation) might have a particularly strong negative impact but are not particularly singled out by the aggregated evaluation.",107 )108 109 110def considerations_summary():111 total_filled = sum(112 [len(dct) for dct in st.session_state.card_dict.get("considerations", {}).values()]113 )114 with st.expander(115 f"Considerations for Using Data Completion - {total_filled} of {N_FIELDS}", expanded=False116 ):117 completion_markdown = ""118 completion_markdown += (119 f"- **Overall completion:**\n - {total_filled} of {N_FIELDS} fields\n"120 )121 completion_markdown += f"- **Sub-section - PII Risks and Liability:**\n - {len(st.session_state.card_dict.get('considerations', {}).get('pii', {}))} of {N_FIELDS_PII} fields\n"122 completion_markdown += f"- **Sub-section - Licenses:**\n - {len(st.session_state.card_dict.get('considerations', {}).get('licenses', {}))} of {N_FIELDS_LICENSES} fields\n"123 completion_markdown += f"- **Sub-section - Known Technical Limitations:**\n - {len(st.session_state.card_dict.get('considerations', {}).get('limitations', {}))} of {N_FIELDS_LIMITATIONS} fields\n"124 st.markdown(completion_markdown)125 