Team Ai
Apppublic

ever-flow/visualization_modules

sourceHugging Faceupdated 1y agoView on Hugging Face
0likes
calculation.txt191 linesDownload Raw Back to src
1 2# 산업 PER 통계 계산 함수3def compute_industry_per_stats(df, method, industry_cols):4    industries = df[industry_cols].stack().dropna().unique()5    per_values = {}6    for ind in industries:7        mask = df[industry_cols].apply(lambda r: ind in r.values, axis=1)8        if method == 'AGG':9            sub = df.loc[mask & df['Net Income (LTM)'].notna()]10            if sub.empty:11                per_values[ind] = np.nan12                continue13            q1 = sub['market_cap'].quantile(0.25)14            q3 = sub['market_cap'].quantile(0.75)15            iqr = q3 - q116            lower_bound = q1 - 2 * iqr17            upper_bound = q3 + 2 * iqr18            sub_filtered = sub[(sub['market_cap'] >= lower_bound) & (sub['market_cap'] <= upper_bound)]19            if sub_filtered.empty:20                per_values[ind] = np.nan21            else:22                total_market_cap = sub_filtered['market_cap'].sum()23                total_net_income = sub_filtered['Net Income (LTM)'].sum()24                per_values[ind] = total_market_cap / total_net_income if total_net_income > 0 else np.nan25        else:26            sub = df.loc[mask & df['Net Income (LTM)'].gt(0) & df['Net Income (LTM)'].notna()]27            if sub.empty:28                per_values[ind] = np.nan29                continue30            per_vals = sub['market_cap'] / sub['Net Income (LTM)']31            if method == 'AVG':32                per_values[ind] = per_vals.mean()33            elif method == 'MED':34                per_values[ind] = per_vals.median()35            elif method == 'HRM':36                per_values[ind] = len(per_vals) / (1.0 / per_vals).sum() if len(per_vals) > 0 else np.nan37    return per_values38 39# 산업 EV/EBITDA 통계 계산 함수40def compute_industry_ev_ebitda_stats(df, method, industry_cols):41    industries = df[industry_cols].stack().dropna().unique()42    ev_ebitda_values = {}43    for ind in industries:44        mask = df[industry_cols].apply(lambda r: ind in r.values, axis=1)45        if method == 'AGG':46            sub = df.loc[mask & df['EBITDA (LTM)'].notna() & df['Enterprise Value (FQ0)'].notna()]47            if sub.empty:48                ev_ebitda_values[ind] = np.nan49                continue50            q1 = sub['Enterprise Value (FQ0)'].quantile(0.25)51            q3 = sub['Enterprise Value (FQ0)'].quantile(0.75)52            iqr = q3 - q153            lower_bound = q1 - 2 * iqr54            upper_bound = q3 + 2 * iqr55            sub_filtered = sub[(sub['Enterprise Value (FQ0)'] >= lower_bound) & (sub['Enterprise Value (FQ0)'] <= upper_bound)]56            if sub_filtered.empty:57                ev_ebitda_values[ind] = np.nan58            else:59                total_ev = sub_filtered['Enterprise Value (FQ0)'].sum()60                total_ebitda = sub_filtered['EBITDA (LTM)'].sum()61                ev_ebitda_values[ind] = total_ev / total_ebitda if total_ebitda > 0 else np.nan62        else:63            sub = df.loc[mask & df['EBITDA (LTM)'].gt(0) & df['EBITDA (LTM)'].notna() & df['Enterprise Value (FQ0)'].notna()]64            if sub.empty:65                ev_ebitda_values[ind] = np.nan66                continue67            ev_vals = sub['Enterprise Value (FQ0)'] / sub['EBITDA (LTM)']68            if method == 'AVG':69                ev_ebitda_values[ind] = ev_vals.mean()70            elif method == 'MED':71                ev_ebitda_values[ind] = ev_vals.median()72            elif method == 'HRM':73                ev_ebitda_values[ind] = len(ev_vals) / (1.0 / ev_vals).sum() if len(ev_vals) > 0 else np.nan74    return ev_ebitda_values75 76# 도메인 및 시간 기반 피처 엔지니어링 클래스77class DomainFeatureEngineer(BaseEstimator, TransformerMixin):78    def fit(self, X, y=None):79        Xt = self._transform(X.copy())80        self.cols_ = Xt.columns.tolist()81        return self82 83    def transform(self, X):84        Xt = self._transform(X.copy())85        return Xt.reindex(columns=self.cols_, fill_value=0)86 87    def _transform(self, X):88        X = X.drop('market', axis=1, errors='ignore')89        tcols = [c for c in X.columns if '(' in c and 'LTM' in c]90        tf = {}91        for c in tcols:92            base, per = c.split(' (')93            tf.setdefault(base, {})[per.rstrip(')')] = c94 95        if 'EBIT' in tf and 'Depreciation' in tf:96            for p, e in tf['EBIT'].items():97                d = tf['Depreciation'].get(p)98                if d:99                    X[f'EBITDA ({p})'] = X[e] + X[d]100            tf['EBITDA'] = {p: f'EBITDA ({p})' for p in tf['EBIT']}101 102        bases = ['Total Assets', 'Total Liabilities', 'Equity', 'Net Debt',103                 'Revenue', 'EBIT', 'EBITDA', 'Net Income', 'Net Income After Minority', 'Dividends']104 105        def safe_div(a, b):106            return a.div(b.replace({0: np.nan}))107 108        for v in bases:109            periods = tf.get(v, {})110            seq = [p for p in ['LTM-3', 'LTM-2', 'LTM-1', 'LTM'] if p in periods]111            grs = []112            for i in range(1, len(seq)):113                prev, curr = seq[i-1], seq[i]114                r = safe_div(X[periods[curr]] - X[periods[prev]], X[periods[prev]])115                suffix = '-2' if (prev, curr) == ('LTM-3', 'LTM-2') else '-1' if (prev, curr) == ('LTM-2', 'LTM-1') else ''116                X[f'{v}_growth{suffix}'] = r117                grs.append(r)118            if grs:119                X[f'{v}_avg_growth'] = pd.concat(grs, axis=1).mean(axis=1)120                X[f'{v}_volatility'] = X[[periods[p] for p in seq]].std(axis=1)121            if 'LTM-3' in periods and 'LTM' in periods:122                X[f'{v}_CAGR'] = np.where(123                    X[periods['LTM-3']] > 0,124                    (X[periods['LTM']] / X[periods['LTM-3']])**(1/3) - 1,125                    np.nan126                )127 128        def calc(a, b, name, per):129            if per in tf.get(a, {}) and per in tf.get(b, {}):130                X[f'{name} ({per})'] = safe_div(X[tf[a][per]], X[tf[b][per]])131 132        ratios = [133            ('Equity', 'Total Assets', 'BAR'),134            ('Total Liabilities', 'Equity', 'DBR'),135            ('Revenue', 'Total Assets', 'SAR'),136            ('EBIT', 'Revenue', 'OMR'),137            ('EBITDA', 'Revenue', 'EMR'),138            ('Net Income', 'Total Assets', 'EAR'),139            ('Net Income', 'Equity', 'EBR')140        ]141        for p in ['LTM', 'LTM-1', 'LTM-2', 'LTM-3']:142            for a, b, n in ratios:143                calc(a, b, n, p)144            if p in ['LTM-2', 'LTM-1', 'LTM']:145                ni = tf.get('Net Income After Minority', {}).get(p)146                eq = tf.get('Equity', {}).get(p)147                prev = {'LTM-2': 'LTM-3', 'LTM-1': 'LTM-2', 'LTM': 'LTM-1'}[p]148                ep = tf.get('Equity', {}).get(prev)149                if ni and eq and ep:150                    avg_eq = (X[eq] + X[ep]) / 2151                    X[f'ROE ({p})'] = safe_div(X[ni], avg_eq)152 153        ratio_bases = ['BAR', 'DBR', 'SAR', 'OMR', 'EMR', 'EAR', 'EBR', 'ROE']154        for v in ratio_bases:155            periods = {p: f'{v} ({p})' for p in ['LTM-3', 'LTM-2', 'LTM-1', 'LTM'] if f'{v} ({p})' in X.columns}156            seq = [p for p in ['LTM-3', 'LTM-2', 'LTM-1', 'LTM'] if p in periods]157            grs = []158            for i in range(1, len(seq)):159                prev, curr = seq[i-1], seq[i]160                r = safe_div(X[periods[curr]] - X[periods[prev]], X[periods[prev]])161                suffix = '-2' if (prev, curr) == ('LTM-3', 'LTM-2') else '-1' if (prev, curr) == ('LTM-2', 'LTM-1') else ''162                X[f'{v}_growth{suffix}'] = r163                grs.append(r)164            if grs:165                X[f'{v}_avg_growth'] = pd.concat(grs, axis=1).mean(axis=1)166                X[f'{v}_volatility'] = X[[periods[p] for p in seq]].std(axis=1)167            if 'LTM-3' in periods and 'LTM' in periods:168                X[f'{v}_CAGR'] = np.where(169                    X[periods['LTM-3']] > 0,170                    (X[periods['LTM']] / X[periods['LTM-3']])**(1/3) - 1,171                    np.nan172                )173 174        dep_cols = [c for c in X.columns if c.startswith('Depreciation')]175        if dep_cols:176            X.drop(columns=dep_cols, inplace=True)177 178        return X179 180 181# 최적화된 LightGBM 퀀타일 모델 실행182def quick_lightgbm_quantile(df, name, industry_cols):183    df = df.copy()184 185    if 'EBITDA (LTM)' not in df and 'EBIT (LTM)' in df and 'Depreciation (LTM)' in df:186        df['EBITDA (LTM)'] = df['EBIT (LTM)'] + df['Depreciation (LTM)']187 188    df['Listing_Age_Days'] = (pd.to_datetime('2024-12-31') - pd.to_datetime(df['Listing Date'])).dt.days189    df.drop(columns=['Listing Date'], inplace=True)190 191