Team Ai
Apppublic

GEM/DatasetCardForm

sourceHugging Faceupdated 4y agoView on Hugging Face
9likes
considerations.py125 linesDownload Raw Back to datacards
1import streamlit as st2 3from .streamlit_utils import (4    make_multiselect,5    make_selectbox,6    make_text_area,7    make_text_input,8    make_radio,9)10 11N_FIELDS_PII = 112N_FIELDS_LICENSES = 213N_FIELDS_LIMITATIONS = 314 15N_FIELDS = N_FIELDS_PII + N_FIELDS_LICENSES + N_FIELDS_LIMITATIONS16 17 18def considerations_page():19    st.session_state.card_dict["considerations"] = st.session_state.card_dict.get(20        "considerations", {}21    )22    with st.expander("PII Risks and Liability", expanded=False):23        key_pref = ["considerations", "pii"]24        st.session_state.card_dict["considerations"]["pii"] = st.session_state.card_dict[25            "considerations"26        ].get("pii", {})27        make_text_area(28            label="Considering your answers to the PII part of the Data Curation Section, describe any potential privacy to the data subjects and creators risks when using the dataset.",29            key_list=key_pref+["risks-description"],30            help="In terms for example of having models memorize private information of data subjects or other breaches of privacy."31        )32 33    with st.expander("Licenses", expanded=False):34        key_pref = ["considerations", "licenses"]35        st.session_state.card_dict["considerations"]["licenses"] = st.session_state.card_dict[36            "considerations"37        ].get("licenses", {})38 39        make_multiselect(40            label="Based on your answers in the Intended Use part of the Data Overview Section, which of the following best describe the copyright and licensing status of the dataset?",41            options=[42                "public domain",43                "multiple licenses",44                "copyright - all rights reserved",45                "open license - commercial use allowed",46                "research use only",47                "non-commercial use only",48                "do not distribute",49                "other",50            ],51            key_list=key_pref + ["dataset-restrictions"],52            help="Does the license restrict how the dataset can be used?",53        )54        if "other" in st.session_state.card_dict["considerations"]["licenses"].get("dataset-restrictions", []):55            make_text_area(56                label="You selected `other` for the dataset licensing status, please elaborate here:",57                key_list=key_pref+["dataset-restrictions-other"]58            )59        else:60            st.session_state.card_dict["considerations"]["licenses"]["dataset-restrictions-other"] = "N/A"61 62        make_multiselect(63            label="Based on your answers in the Language part of the Data Curation Section, which of the following best describe the copyright and licensing status of the underlying language data?",64            options=[65                "public domain",66                "multiple licenses",67                "copyright - all rights reserved",68                "open license - commercial use allowed",69                "research use only",70                "non-commercial use only",71                "do not distribute",72                "other",73            ],74            key_list=key_pref + ["data-copyright"],75            help="For example if the dataset uses data from Wikipedia, we are asking about the status of Wikipedia text in general.",76        )77        if "other" in st.session_state.card_dict["considerations"]["licenses"].get("data-copyright", []):78            make_text_area(79                label="You selected `other` for the source data licensing status, please elaborate here:",80                key_list=key_pref+["data-copyright-other"]81            )82        else:83            st.session_state.card_dict["considerations"]["licenses"]["data-copyright-other"] = "N/A"84 85    with st.expander("Known Technical Limitations", expanded=False):86        key_pref = ["considerations", "limitations"]87        st.session_state.card_dict["considerations"]["limitations"] = st.session_state.card_dict[88            "considerations"89        ].get("limitations", {})90        make_text_area(91            label="Describe any known technical limitations, such as spurrious correlations, train/test overlap, annotation biases, or mis-annotations, " + \92            "and cite the works that first identified these limitations when possible.",93            key_list=key_pref + ["data-technical-limitations"],94            help="Outline any properties of the dataset that might lead a trained model with good performance on the metric to not behave as expected.",95        )96        make_text_area(97            label="When using a model trained on this dataset in a setting where users or the public may interact with its predictions, what are some pitfalls to look out for? " + \98            "In particular, describe some applications of the general task featured in this dataset that its curation or properties make it less suitable for.",99            key_list=key_pref + ["data-unsuited-applications"],100            help="For example, outline language varieties or domains that the model might underperform for.",101        )102        make_text_area(103            label="What are some discouraged use cases of a model trained to maximize the proposed metrics on this dataset? " +104            "In particular, think about settings where decisions made by a model that performs reasonably well on the metric my still have strong negative consequences for user or members of the public.",105            key_list=key_pref + ["data-discouraged-use"],106            help="For example, think about application settings where certain types of mistakes (such as missing a negation) might have a particularly strong negative impact but are not particularly singled out by the aggregated evaluation.",107        )108 109 110def considerations_summary():111    total_filled = sum(112        [len(dct) for dct in st.session_state.card_dict.get("considerations", {}).values()]113    )114    with st.expander(115        f"Considerations for Using Data Completion - {total_filled} of {N_FIELDS}", expanded=False116    ):117        completion_markdown = ""118        completion_markdown += (119            f"- **Overall completion:**\n  - {total_filled} of {N_FIELDS} fields\n"120        )121        completion_markdown += f"- **Sub-section - PII Risks and Liability:**\n  - {len(st.session_state.card_dict.get('considerations', {}).get('pii', {}))} of {N_FIELDS_PII} fields\n"122        completion_markdown += f"- **Sub-section - Licenses:**\n  - {len(st.session_state.card_dict.get('considerations', {}).get('licenses', {}))} of {N_FIELDS_LICENSES} fields\n"123        completion_markdown += f"- **Sub-section - Known Technical Limitations:**\n  - {len(st.session_state.card_dict.get('considerations', {}).get('limitations', {}))} of {N_FIELDS_LIMITATIONS} fields\n"124        st.markdown(completion_markdown)125