Team Ai
Apppublic

GEM/DatasetCardForm

sourceHugging Faceupdated 4y agoView on Hugging Face
9likes
context.py113 linesDownload Raw Back to datacards
1import streamlit as st2 3from .streamlit_utils import (4    make_multiselect,5    make_selectbox,6    make_text_area,7    make_text_input,8    make_radio,9)10 11N_FIELDS_PREVIOUS = 312N_FIELDS_UNDERSERVED_COMMUNITIES = 213N_FIELDS_BIASES= 314 15N_FIELDS = N_FIELDS_PREVIOUS + N_FIELDS_UNDERSERVED_COMMUNITIES + N_FIELDS_BIASES16 17def context_page():18    st.session_state.card_dict["context"] = st.session_state.card_dict.get(19        "context", {}20    )21    with st.expander("Previous Work on the Social Impact of the Dataset", expanded=False):22        key_pref = ["context", "previous"]23        st.session_state.card_dict["context"]["previous"] = st.session_state.card_dict[24            "context"25        ].get("previous", {})26 27        make_radio(28            label="Are you aware of cases where models trained on the task featured in this dataset ore related tasks have been used in automated systems?",29            options=["no", "yes - related tasks", "yes - other datasets featuring the same task", "yes - models trained on this dataset"],30            key_list=key_pref + ["is-deployed"],31            help="",32        )33        if "yes" in st.session_state.card_dict["context"]["previous"]["is-deployed"]:34            make_text_area(35                label="Did any of these previous uses result in observations about the social impact of the systems? " + \36                "In particular, has there been work outlining the risks and limitations of the system? Provide links and descriptions here.",37                key_list=key_pref + ["described-risks"],38                help="",39            )40            if st.session_state.card_dict["context"]["previous"]["is-deployed"] == "yes - models trained on this dataset":41                make_text_area(42                    label="Have any changes been made to the dataset as a result of these observations?",43                    key_list=key_pref + ["changes-from-observation"],44                    help="",45                )46            else:47                st.session_state.card_dict["context"]["previous"]["changes-from-observation"] = "N/A"48        else:49            st.session_state.card_dict["context"]["previous"]["described-risks"] = "N/A"50            st.session_state.card_dict["context"]["previous"]["changes-from-observation"] = "N/A"51 52    with st.expander("Impact on Under-Served Communities", expanded=False):53        key_pref = ["context", "underserved"]54        st.session_state.card_dict["context"]["underserved"] = st.session_state.card_dict[55            "context"56        ].get("underserved", {})57        make_radio(58            label="Does this dataset address the needs of communities that are traditionally underserved in language technology, and particularly language generation technology?" + \59                "Communities may be underserved for exemple because their language, language variety, or social or geographical context is underepresented in NLP and NLG resources (datasets and models).",60            options=["no", "yes"],61            key_list=key_pref+["helps-underserved"],62        )63        if st.session_state.card_dict["context"]["underserved"]["helps-underserved"] == "yes":64            make_text_area(65                label="Describe how this dataset addresses the needs of underserved communities.",66                key_list=key_pref+["underserved-description"],67            )68        else:69            st.session_state.card_dict["context"]["underserved"]["underserved-description"] = "N/A"70 71    with st.expander("Discussion of Biases", expanded=False):72        key_pref = ["context", "biases"]73        st.session_state.card_dict["context"]["biases"] = st.session_state.card_dict[74            "context"75        ].get("biases", {})76        make_radio(77            label="Are there documented social biases in the dataset? " + \78                "Biases in this context are variations in the ways members of different social categories are represented that can have harmful downstream consequences for members of the more disadvantaged group.",79            options=["yes", "unsure", "no"],80            key_list=key_pref + ["has-biases"],81            help="For a more extensive definition of social biases, see [Language (Technology) is Power: A Critical Survey of “Bias” in NLP ](https://aclanthology.org/2020.acl-main.485.pdf)",82        )83        if st.session_state.card_dict["context"]["biases"]["has-biases"] == "yes":84            make_text_area(85                label="Provide links to and summaries of works analyzing these biases.",86                key_list=key_pref + ["bias-analyses"],87                help="The analyses can take the form of academic papers or news articles, or even blog posts.",88            )89        else:90            st.session_state.card_dict["context"]["biases"]["bias-analyses"] = "N/A"91        make_text_area(92            label="Does the distribution of language producers in the dataset accurately represent the full distribution of speakers of the language world-wide? If not, how does it differ?",93            key_list=key_pref + ["speaker-distibution"],94            help="For example, are most speakers in the dataset of a certain gender or located in a certain county?",95        )96 97 98def context_summary():99    total_filled = sum(100        [len(dct) for dct in st.session_state.card_dict.get("context", {}).values()]101    )102    with st.expander(103        f"Broader Social Context Completion - {total_filled} of {N_FIELDS}", expanded=False104    ):105        completion_markdown = ""106        completion_markdown += (107            f"- **Overall completion:**\n  - {total_filled} of {N_FIELDS} fields\n"108        )109        completion_markdown += f"- **Sub-section - Previous Work on the Social Impact of the Dataset:**\n  - {len(st.session_state.card_dict.get('context', {}).get('previous', {}))} of {N_FIELDS_PREVIOUS} fields\n"110        completion_markdown += f"- **Sub-section - Impact on Under-Served Communities:**\n  - {len(st.session_state.card_dict.get('context', {}).get('underserved', {}))} of {N_FIELDS_UNDERSERVED_COMMUNITIES} fields\n"111        completion_markdown += f"- **Sub-section - Discussion of Biases:**\n  - {len(st.session_state.card_dict.get('context', {}).get('biases', {}))} of {N_FIELDS_BIASES} fields\n"112        st.markdown(completion_markdown)113