malvec/codebert_Malvec
0
1import numpy as np2import json3import pickle4from minisom import MiniSom5 6 7class SOMAnalyzer:8 """SOM 分析器 - 載入預訓練的 SOM 並分析新檔案"""9 10 def __init__(self, som_weights_path, background_data_path):11 """12 初始化 SOM 分析器13 14 Args:15 som_weights_path: SOM 權重檔案路徑 (.npy)16 background_data_path: 背景資料檔案路徑 (.pkl),包含所有訓練資料的 embeddings 和標籤17 """18 # 載入 SOM 權重19 self.som_weights = np.load(som_weights_path)20 self.som_size = self.som_weights.shape[0]21 self.input_dim = self.som_weights.shape[2]22 23 # 建立 SOM 模型24 self.som = MiniSom(25 self.som_size, 26 self.som_size, 27 self.input_dim,28 sigma=0.5, 29 learning_rate=0.5, 30 random_seed=7731 )32 self.som._weights = self.som_weights.copy()33 34 # 載入背景資料35 with open(background_data_path, 'rb') as f:36 background = pickle.load(f)37 self.background_embeddings = background['embeddings']38 self.background_labels = background['labels']39 self.id2label = background['id2label']40 41 def find_winner_neuron(self, embedding):42 """43 找到 embedding 對應的 winner neuron44 45 Args:46 embedding: 768 維的 embedding vector47 48 Returns:49 (row, col): winner neuron 的座標50 """51 return self.som.winner(embedding)52 53 def predict_features_from_position(self, row, col, radius=1):54 """55 ✅ 根據 SOM 位置智能預測特徵值56 57 分析該位置及其周圍區域的背景樣本,統計特徵的比例來預測58 59 Args:60 row: winner neuron 的 row61 col: winner neuron 的 col62 radius: 搜尋半徑(0=只看該格,1=3x3區域,2=5x5區域)63 64 Returns:65 dict: 特徵預測結果,例如 {'dropper': 1, 'APT30': 0}66 """67 print(f" 🔍 Analyzing region around ({row}, {col}) with radius={radius}")68 69 # 收集該區域的所有樣本70 region_samples = {'dropper': [], 'APT30': []}71 72 for i, emb in enumerate(self.background_embeddings):73 sample_row, sample_col = self.som.winner(emb)74 75 # 檢查是否在指定半徑內76 if abs(sample_row - row) <= radius and abs(sample_col - col) <= radius:77 # 收集該樣本的特徵值78 if 'dropper' in self.background_labels:79 region_samples['dropper'].append(self.background_labels['dropper'][i])80 if 'APT30' in self.background_labels:81 region_samples['APT30'].append(self.background_labels['APT30'][i])82 83 # 根據該區域的統計來預測84 predictions = {}85 86 for feature_name, values in region_samples.items():87 if len(values) == 0:88 # 沒有樣本,預設為 089 predictions[feature_name] = 090 print(f" {feature_name}: No samples → 0")91 else:92 # 計算該特徵為 1 的比例93 positive_count = sum(values)94 total_count = len(values)95 positive_ratio = positive_count / total_count96 97 # 如果超過 5%,預測為 198 predictions[feature_name] = 1 if positive_ratio > 0.05 else 099 100 print(f" {feature_name}: {positive_count}/{total_count} samples = {positive_ratio:.1%} → {predictions[feature_name]}")101 102 return predictions103 104 def generate_som_json_with_marker(self, new_embedding, feature_name, feature_value, 105 filename="unknown", predicted_family=None):106 """107 生成包含新檔案標記的 SOM JSON108 109 Args:110 new_embedding: 新檔案的 768 維 embedding111 feature_name: 特徵名稱 (如 'dropper', 'APT30')112 feature_value: 該特徵的值 (0 或 1)113 filename: 檔案名稱114 predicted_family: 預測的惡意軟體家族115 116 Returns:117 dict: JSON 格式的 SOM 資料118 """119 # 找到新檔案的位置120 new_row, new_col = self.find_winner_neuron(new_embedding)121 122 # 計算背景資料的 neuron 統計123 neuron_stats = {}124 for i, emb in enumerate(self.background_embeddings):125 row, col = self.som.winner(emb)126 key = f"{row},{col}"127 128 if key not in neuron_stats:129 neuron_stats[key] = {}130 131 # 根據 feature_name 取得對應的標籤132 label = self.background_labels[feature_name][i]133 neuron_stats[key][label] = neuron_stats[key].get(label, 0) + 1134 135 # 建立 JSON 結構136 som_json = {137 "title": f"SOM neuron composition by {feature_name}",138 "som_size": self.som_size,139 "neurons": [],140 "new_file": {141 "filename": filename,142 "row": int(new_row),143 "col": int(new_col),144 "feature_name": feature_name,145 "feature_value": int(feature_value),146 "predicted_family": predicted_family147 }148 }149 150 # 轉換每個 neuron 的統計資料151 for key, counts in neuron_stats.items():152 row, col = map(int, key.split(","))153 total = sum(counts.values())154 proportions = {155 self.id2label.get(lbl, str(lbl)): cnt/total 156 for lbl, cnt in counts.items()157 }158 som_json["neurons"].append({159 "row": row,160 "col": col,161 "counts": counts,162 "proportions": proportions163 })164 165 return som_json166 167 168def prepare_background_data(df_merged, output_path):169 """170 準備背景資料(只需執行一次)171 172 Args:173 df_merged: 合併後的 DataFrame,包含 embeddings 和所有標籤174 output_path: 輸出檔案路徑175 """176 # 提取 embeddings177 emb_cols = [f"emb_{i}" for i in range(768)]178 embeddings = df_merged[emb_cols].values.astype(float)179 180 # 提取所有標籤181 labels = {182 'true_label_id': df_merged['true_label_id'].fillna(0).astype(int).values,183 'dropper': df_merged['dropper'].fillna(0).astype(int).values,184 'spreader': df_merged['spreader'].fillna(0).astype(int).values,185 'APT30': df_merged['APT30'].fillna(0).astype(int).values,186 'Codoso_Gh0st_1': df_merged['Codoso_Gh0st_1'].fillna(0).astype(int).values,187 }188 189 # 建立 id2label 映射190 label_list = sorted(df_merged['true_label'].unique())191 label2id = {l: i for i, l in enumerate(label_list)}192 id2label = {i: l for l, i in label2id.items()}193 194 # 保存195 background_data = {196 'embeddings': embeddings,197 'labels': labels,198 'id2label': id2label199 }200 201 with open(output_path, 'wb') as f:202 pickle.dump(background_data, f)203 204 print(f"✅ 背景資料已保存至: {output_path}")205 print(f" - 資料筆數: {len(embeddings)}")206 print(f" - Embedding 維度: {embeddings.shape[1]}")207 print(f" - 特徵數量: {len(labels)}")