Team Ai
Apppublic

ever-flow/visualization_modules

sourceHugging Faceupdated 1y agoView on Hugging Face
0likes
scatter.py134 linesDownload Raw Back to visualizations
1import streamlit as st2import pandas as pd3import numpy as np4import plotly.graph_objects as go5 6from config import PERIODS, METHODS, COLORS7 8 9def show_scatter(DF_RAW):10    st.header("Scatter Plot")11    12    METRIC_HIERARCHY = {13        "Valuation": ["PER", "PBR", "EV_EBITDA", "시가총액/매출액", "시가총액/영업이익"],14        "Profitability": ["ROE", "영업이익률", "EBITDA/Sales", "총자산이익률"],15        "Activity": ["자산회전율"],16        "Stability": ["자기자본비율", "부채비율"]17    }18 19    with st.sidebar:20        st.markdown("### 분류")21        classification_type = st.radio("분석 기준", ["EMSEC", "EMTEC"], horizontal=True, key="class_type2")22        cls_df = DF_RAW.copy()  # Classification 열을 직접 사용하지 않음23        if classification_type == "EMSEC":24            l1_options = ["전체"] + sorted(cls_df["Sector"].dropna().unique())25            l1_label = "Sector"26            l1_selection = st.selectbox(l1_label, l1_options, key="sector_sel2")27            l2_label = "Industry"28        else:29            l1_options = ["전체"] + sorted(cls_df["Theme"].dropna().unique())30            l1_label = "Theme"31            l1_selection = st.selectbox(l1_label, l1_options, key="theme_sel2")32            l2_label = "Technology"33 34        st.markdown("### 설정")35        period_sel = st.selectbox("기간", PERIODS, key="period_sel2")36        metric_group = st.selectbox("지표 그룹", list(METRIC_HIERARCHY.keys()), key="metric_group2")37        metric_options = METRIC_HIERARCHY[metric_group]38        metric_sel = st.selectbox("지표", metric_options, key="metric_sel2")39        country_sel = st.selectbox("국가", ["전체", "한국", "미국", "일본"], key="country_sel2")40        market_pool_options = {41            "전체": ["전체"] + sorted(cls_df["Market"].dropna().unique()),42            "한국": ["전체", "KOSPI", "KOSDAQ"],43            "미국": ["전체", "NASDAQ", "NYSE"],44            "일본": ["전체", "Prime (Domestic Stocks)", "Standard (Domestic Stocks)", "Prime (Foreign Stocks)"],45        }46        market_sel = st.selectbox("거래소", market_pool_options.get(country_sel, ["전체"]), key="market_sel2")47 48    def filter_data(df: pd.DataFrame) -> pd.DataFrame:49        d = df[df.Year == period_sel].copy()50        if classification_type == "EMSEC":51            if l1_selection != "전체":52                d = d[d["Sector"] == l1_selection]53        else:54            if l1_selection != "전체":55                d = d[d["Theme"] == l1_selection]56        if country_sel != "전체":57            d = d[d["Country"] == country_sel]58        if market_sel != "전체":59            d = d[d["Market"] == market_sel]60        return d61 62    FILT_DATA = filter_data(DF_RAW)63    metric_col = metric_sel64    if metric_col not in FILT_DATA.columns:65        st.error(f"'{metric_col}' 열이 데이터에 없습니다. 데이터나 설정을 확인해주세요.")66        st.stop()67 68    def harmonic_mean(arr: pd.Series):69        arr = arr.dropna()70        arr = arr[arr > 0]71        return len(arr) / (1 / arr).sum() if len(arr) > 0 else np.nan72 73    def aggregate_by_group(sub: pd.DataFrame, metric_col: str) -> pd.Series:74        sub_unique = sub.drop_duplicates(subset=["ticker"])75        arr = pd.to_numeric(sub_unique[metric_col], errors="coerce")76        res = {77            "AVG": arr.mean(),78            "MED": arr.median(),79            "HRM": harmonic_mean(arr)80        }81        num, den = None, None82        if metric_sel == "PER":83            num = sub_unique["Market Cap (2024-12-31)_USD"].sum()84            den = sub_unique["Net_Income"].sum()85        elif metric_sel == "PBR":86            num = sub_unique["Market Cap (2024-12-31)_USD"].sum()87            den = sub_unique["Book"].sum()88        elif metric_sel == "EV_EBITDA":89            num = sub_unique["Enterprise Value (FQ0)_USD"].sum()90            den = sub_unique["EBITDA"].sum()91        if num is not None and den is not None and den != 0:92            res["AGG"] = num / den93        else:94            res["AGG"] = res["AVG"]95        res["기업 수"] = len(arr.dropna())96        return pd.Series(res)97 98    if FILT_DATA.empty:99        st.warning("선택하신 조건에 맞는 데이터가 없습니다.")100        st.stop()101 102    agg_df = FILT_DATA.groupby(l2_label).apply(lambda g: aggregate_by_group(g, metric_col))103    agg_df = agg_df.dropna(how='all', subset=METHODS).sort_index()104 105    st.caption(f"분석 기준: {classification_type} > {l1_selection}")106    if agg_df.empty:107        st.warning("집계 결과 데이터가 없어 차트를 그릴 수 없습니다.")108        st.stop()109 110    fig = go.Figure()111    for method in METHODS:112        fig.add_trace(go.Scatter(113            x=agg_df.index,114            y=agg_df[method],115            customdata=agg_df[['기업 수']].to_numpy(),116            mode="markers",117            marker=dict(size=12, color=COLORS[method], line=dict(width=1, color="white")),118            name=method,119            hovertemplate=f"<b>{agg_df.index.name}:</b> %{{x}}<br><b>{metric_sel}:</b> %{{y:.2f}}<br><b>계산 방식:</b> {method}<br><b>기업 수:</b> %{{customdata[0]}}<br><extra></extra>"120        ))121    y_axis_title = f"{metric_sel} ({period_sel})"122    fig.update_layout(123        title=dict(text=f"{l2_label}별 '{y_axis_title}' 비교 (계산 방식별)", x=0.5, xanchor='center'),124        xaxis_title=l2_label,125        yaxis_title=y_axis_title,126        xaxis_tickangle=-45,127        legend_title="계산 방식",128        height=650,129        hovermode="x unified"130    )131    st.plotly_chart(fig, use_container_width=True)132    sel_list = [v for v in [classification_type, l1_selection, period_sel, metric_sel, country_sel if country_sel != "전체" else None, market_sel if market_sel != "전체" else None] if v and v != "전체"]133    st.caption(" | ".join(sel_list) + f" • 그룹 수: {len(agg_df):,}")134