ever-flow/visualization_modules
0
1 2# 산업 PER 통계 계산 함수3def compute_industry_per_stats(df, method, industry_cols):4 industries = df[industry_cols].stack().dropna().unique()5 per_values = {}6 for ind in industries:7 mask = df[industry_cols].apply(lambda r: ind in r.values, axis=1)8 if method == 'AGG':9 sub = df.loc[mask & df['Net Income (LTM)'].notna()]10 if sub.empty:11 per_values[ind] = np.nan12 continue13 q1 = sub['market_cap'].quantile(0.25)14 q3 = sub['market_cap'].quantile(0.75)15 iqr = q3 - q116 lower_bound = q1 - 2 * iqr17 upper_bound = q3 + 2 * iqr18 sub_filtered = sub[(sub['market_cap'] >= lower_bound) & (sub['market_cap'] <= upper_bound)]19 if sub_filtered.empty:20 per_values[ind] = np.nan21 else:22 total_market_cap = sub_filtered['market_cap'].sum()23 total_net_income = sub_filtered['Net Income (LTM)'].sum()24 per_values[ind] = total_market_cap / total_net_income if total_net_income > 0 else np.nan25 else:26 sub = df.loc[mask & df['Net Income (LTM)'].gt(0) & df['Net Income (LTM)'].notna()]27 if sub.empty:28 per_values[ind] = np.nan29 continue30 per_vals = sub['market_cap'] / sub['Net Income (LTM)']31 if method == 'AVG':32 per_values[ind] = per_vals.mean()33 elif method == 'MED':34 per_values[ind] = per_vals.median()35 elif method == 'HRM':36 per_values[ind] = len(per_vals) / (1.0 / per_vals).sum() if len(per_vals) > 0 else np.nan37 return per_values38 39# 산업 EV/EBITDA 통계 계산 함수40def compute_industry_ev_ebitda_stats(df, method, industry_cols):41 industries = df[industry_cols].stack().dropna().unique()42 ev_ebitda_values = {}43 for ind in industries:44 mask = df[industry_cols].apply(lambda r: ind in r.values, axis=1)45 if method == 'AGG':46 sub = df.loc[mask & df['EBITDA (LTM)'].notna() & df['Enterprise Value (FQ0)'].notna()]47 if sub.empty:48 ev_ebitda_values[ind] = np.nan49 continue50 q1 = sub['Enterprise Value (FQ0)'].quantile(0.25)51 q3 = sub['Enterprise Value (FQ0)'].quantile(0.75)52 iqr = q3 - q153 lower_bound = q1 - 2 * iqr54 upper_bound = q3 + 2 * iqr55 sub_filtered = sub[(sub['Enterprise Value (FQ0)'] >= lower_bound) & (sub['Enterprise Value (FQ0)'] <= upper_bound)]56 if sub_filtered.empty:57 ev_ebitda_values[ind] = np.nan58 else:59 total_ev = sub_filtered['Enterprise Value (FQ0)'].sum()60 total_ebitda = sub_filtered['EBITDA (LTM)'].sum()61 ev_ebitda_values[ind] = total_ev / total_ebitda if total_ebitda > 0 else np.nan62 else:63 sub = df.loc[mask & df['EBITDA (LTM)'].gt(0) & df['EBITDA (LTM)'].notna() & df['Enterprise Value (FQ0)'].notna()]64 if sub.empty:65 ev_ebitda_values[ind] = np.nan66 continue67 ev_vals = sub['Enterprise Value (FQ0)'] / sub['EBITDA (LTM)']68 if method == 'AVG':69 ev_ebitda_values[ind] = ev_vals.mean()70 elif method == 'MED':71 ev_ebitda_values[ind] = ev_vals.median()72 elif method == 'HRM':73 ev_ebitda_values[ind] = len(ev_vals) / (1.0 / ev_vals).sum() if len(ev_vals) > 0 else np.nan74 return ev_ebitda_values75 76# 도메인 및 시간 기반 피처 엔지니어링 클래스77class DomainFeatureEngineer(BaseEstimator, TransformerMixin):78 def fit(self, X, y=None):79 Xt = self._transform(X.copy())80 self.cols_ = Xt.columns.tolist()81 return self82 83 def transform(self, X):84 Xt = self._transform(X.copy())85 return Xt.reindex(columns=self.cols_, fill_value=0)86 87 def _transform(self, X):88 X = X.drop('market', axis=1, errors='ignore')89 tcols = [c for c in X.columns if '(' in c and 'LTM' in c]90 tf = {}91 for c in tcols:92 base, per = c.split(' (')93 tf.setdefault(base, {})[per.rstrip(')')] = c94 95 if 'EBIT' in tf and 'Depreciation' in tf:96 for p, e in tf['EBIT'].items():97 d = tf['Depreciation'].get(p)98 if d:99 X[f'EBITDA ({p})'] = X[e] + X[d]100 tf['EBITDA'] = {p: f'EBITDA ({p})' for p in tf['EBIT']}101 102 bases = ['Total Assets', 'Total Liabilities', 'Equity', 'Net Debt',103 'Revenue', 'EBIT', 'EBITDA', 'Net Income', 'Net Income After Minority', 'Dividends']104 105 def safe_div(a, b):106 return a.div(b.replace({0: np.nan}))107 108 for v in bases:109 periods = tf.get(v, {})110 seq = [p for p in ['LTM-3', 'LTM-2', 'LTM-1', 'LTM'] if p in periods]111 grs = []112 for i in range(1, len(seq)):113 prev, curr = seq[i-1], seq[i]114 r = safe_div(X[periods[curr]] - X[periods[prev]], X[periods[prev]])115 suffix = '-2' if (prev, curr) == ('LTM-3', 'LTM-2') else '-1' if (prev, curr) == ('LTM-2', 'LTM-1') else ''116 X[f'{v}_growth{suffix}'] = r117 grs.append(r)118 if grs:119 X[f'{v}_avg_growth'] = pd.concat(grs, axis=1).mean(axis=1)120 X[f'{v}_volatility'] = X[[periods[p] for p in seq]].std(axis=1)121 if 'LTM-3' in periods and 'LTM' in periods:122 X[f'{v}_CAGR'] = np.where(123 X[periods['LTM-3']] > 0,124 (X[periods['LTM']] / X[periods['LTM-3']])**(1/3) - 1,125 np.nan126 )127 128 def calc(a, b, name, per):129 if per in tf.get(a, {}) and per in tf.get(b, {}):130 X[f'{name} ({per})'] = safe_div(X[tf[a][per]], X[tf[b][per]])131 132 ratios = [133 ('Equity', 'Total Assets', 'BAR'),134 ('Total Liabilities', 'Equity', 'DBR'),135 ('Revenue', 'Total Assets', 'SAR'),136 ('EBIT', 'Revenue', 'OMR'),137 ('EBITDA', 'Revenue', 'EMR'),138 ('Net Income', 'Total Assets', 'EAR'),139 ('Net Income', 'Equity', 'EBR')140 ]141 for p in ['LTM', 'LTM-1', 'LTM-2', 'LTM-3']:142 for a, b, n in ratios:143 calc(a, b, n, p)144 if p in ['LTM-2', 'LTM-1', 'LTM']:145 ni = tf.get('Net Income After Minority', {}).get(p)146 eq = tf.get('Equity', {}).get(p)147 prev = {'LTM-2': 'LTM-3', 'LTM-1': 'LTM-2', 'LTM': 'LTM-1'}[p]148 ep = tf.get('Equity', {}).get(prev)149 if ni and eq and ep:150 avg_eq = (X[eq] + X[ep]) / 2151 X[f'ROE ({p})'] = safe_div(X[ni], avg_eq)152 153 ratio_bases = ['BAR', 'DBR', 'SAR', 'OMR', 'EMR', 'EAR', 'EBR', 'ROE']154 for v in ratio_bases:155 periods = {p: f'{v} ({p})' for p in ['LTM-3', 'LTM-2', 'LTM-1', 'LTM'] if f'{v} ({p})' in X.columns}156 seq = [p for p in ['LTM-3', 'LTM-2', 'LTM-1', 'LTM'] if p in periods]157 grs = []158 for i in range(1, len(seq)):159 prev, curr = seq[i-1], seq[i]160 r = safe_div(X[periods[curr]] - X[periods[prev]], X[periods[prev]])161 suffix = '-2' if (prev, curr) == ('LTM-3', 'LTM-2') else '-1' if (prev, curr) == ('LTM-2', 'LTM-1') else ''162 X[f'{v}_growth{suffix}'] = r163 grs.append(r)164 if grs:165 X[f'{v}_avg_growth'] = pd.concat(grs, axis=1).mean(axis=1)166 X[f'{v}_volatility'] = X[[periods[p] for p in seq]].std(axis=1)167 if 'LTM-3' in periods and 'LTM' in periods:168 X[f'{v}_CAGR'] = np.where(169 X[periods['LTM-3']] > 0,170 (X[periods['LTM']] / X[periods['LTM-3']])**(1/3) - 1,171 np.nan172 )173 174 dep_cols = [c for c in X.columns if c.startswith('Depreciation')]175 if dep_cols:176 X.drop(columns=dep_cols, inplace=True)177 178 return X179 180 181# 최적화된 LightGBM 퀀타일 모델 실행182def quick_lightgbm_quantile(df, name, industry_cols):183 df = df.copy()184 185 if 'EBITDA (LTM)' not in df and 'EBIT (LTM)' in df and 'Depreciation (LTM)' in df:186 df['EBITDA (LTM)'] = df['EBIT (LTM)'] + df['Depreciation (LTM)']187 188 df['Listing_Age_Days'] = (pd.to_datetime('2024-12-31') - pd.to_datetime(df['Listing Date'])).dt.days189 df.drop(columns=['Listing Date'], inplace=True)190 191 