Team Ai
Apppublic

malvec/codebert_Malvec

sourceHugging Faceupdated 11mo agoView on Hugging Face
0likes
som_analyzer.py207 linesDownload Raw Back to root
1import numpy as np2import json3import pickle4from minisom import MiniSom5 6 7class SOMAnalyzer:8    """SOM 分析器 - 載入預訓練的 SOM 並分析新檔案"""9    10    def __init__(self, som_weights_path, background_data_path):11        """12        初始化 SOM 分析器13        14        Args:15            som_weights_path: SOM 權重檔案路徑 (.npy)16            background_data_path: 背景資料檔案路徑 (.pkl),包含所有訓練資料的 embeddings 和標籤17        """18        # 載入 SOM 權重19        self.som_weights = np.load(som_weights_path)20        self.som_size = self.som_weights.shape[0]21        self.input_dim = self.som_weights.shape[2]22        23        # 建立 SOM 模型24        self.som = MiniSom(25            self.som_size, 26            self.som_size, 27            self.input_dim,28            sigma=0.5, 29            learning_rate=0.5, 30            random_seed=7731        )32        self.som._weights = self.som_weights.copy()33        34        # 載入背景資料35        with open(background_data_path, 'rb') as f:36            background = pickle.load(f)37            self.background_embeddings = background['embeddings']38            self.background_labels = background['labels']39            self.id2label = background['id2label']40    41    def find_winner_neuron(self, embedding):42        """43        找到 embedding 對應的 winner neuron44        45        Args:46            embedding: 768 維的 embedding vector47            48        Returns:49            (row, col): winner neuron 的座標50        """51        return self.som.winner(embedding)52    53    def predict_features_from_position(self, row, col, radius=1):54        """55        ✅ 根據 SOM 位置智能預測特徵值56        57        分析該位置及其周圍區域的背景樣本,統計特徵的比例來預測58        59        Args:60            row: winner neuron 的 row61            col: winner neuron 的 col62            radius: 搜尋半徑(0=只看該格,1=3x3區域,2=5x5區域)63            64        Returns:65            dict: 特徵預測結果,例如 {'dropper': 1, 'APT30': 0}66        """67        print(f"   🔍 Analyzing region around ({row}, {col}) with radius={radius}")68        69        # 收集該區域的所有樣本70        region_samples = {'dropper': [], 'APT30': []}71        72        for i, emb in enumerate(self.background_embeddings):73            sample_row, sample_col = self.som.winner(emb)74            75            # 檢查是否在指定半徑內76            if abs(sample_row - row) <= radius and abs(sample_col - col) <= radius:77                # 收集該樣本的特徵值78                if 'dropper' in self.background_labels:79                    region_samples['dropper'].append(self.background_labels['dropper'][i])80                if 'APT30' in self.background_labels:81                    region_samples['APT30'].append(self.background_labels['APT30'][i])82        83        # 根據該區域的統計來預測84        predictions = {}85        86        for feature_name, values in region_samples.items():87            if len(values) == 0:88                # 沒有樣本,預設為 089                predictions[feature_name] = 090                print(f"      {feature_name}: No samples → 0")91            else:92                # 計算該特徵為 1 的比例93                positive_count = sum(values)94                total_count = len(values)95                positive_ratio = positive_count / total_count96                97                # 如果超過 5%,預測為 198                predictions[feature_name] = 1 if positive_ratio > 0.05 else 099                100                print(f"      {feature_name}: {positive_count}/{total_count} samples = {positive_ratio:.1%} → {predictions[feature_name]}")101        102        return predictions103    104    def generate_som_json_with_marker(self, new_embedding, feature_name, feature_value, 105                                       filename="unknown", predicted_family=None):106        """107        生成包含新檔案標記的 SOM JSON108        109        Args:110            new_embedding: 新檔案的 768 維 embedding111            feature_name: 特徵名稱 (如 'dropper', 'APT30')112            feature_value: 該特徵的值 (0 或 1)113            filename: 檔案名稱114            predicted_family: 預測的惡意軟體家族115            116        Returns:117            dict: JSON 格式的 SOM 資料118        """119        # 找到新檔案的位置120        new_row, new_col = self.find_winner_neuron(new_embedding)121        122        # 計算背景資料的 neuron 統計123        neuron_stats = {}124        for i, emb in enumerate(self.background_embeddings):125            row, col = self.som.winner(emb)126            key = f"{row},{col}"127            128            if key not in neuron_stats:129                neuron_stats[key] = {}130            131            # 根據 feature_name 取得對應的標籤132            label = self.background_labels[feature_name][i]133            neuron_stats[key][label] = neuron_stats[key].get(label, 0) + 1134        135        # 建立 JSON 結構136        som_json = {137            "title": f"SOM neuron composition by {feature_name}",138            "som_size": self.som_size,139            "neurons": [],140            "new_file": {141                "filename": filename,142                "row": int(new_row),143                "col": int(new_col),144                "feature_name": feature_name,145                "feature_value": int(feature_value),146                "predicted_family": predicted_family147            }148        }149        150        # 轉換每個 neuron 的統計資料151        for key, counts in neuron_stats.items():152            row, col = map(int, key.split(","))153            total = sum(counts.values())154            proportions = {155                self.id2label.get(lbl, str(lbl)): cnt/total 156                for lbl, cnt in counts.items()157            }158            som_json["neurons"].append({159                "row": row,160                "col": col,161                "counts": counts,162                "proportions": proportions163            })164        165        return som_json166 167 168def prepare_background_data(df_merged, output_path):169    """170    準備背景資料(只需執行一次)171    172    Args:173        df_merged: 合併後的 DataFrame,包含 embeddings 和所有標籤174        output_path: 輸出檔案路徑175    """176    # 提取 embeddings177    emb_cols = [f"emb_{i}" for i in range(768)]178    embeddings = df_merged[emb_cols].values.astype(float)179    180    # 提取所有標籤181    labels = {182        'true_label_id': df_merged['true_label_id'].fillna(0).astype(int).values,183        'dropper': df_merged['dropper'].fillna(0).astype(int).values,184        'spreader': df_merged['spreader'].fillna(0).astype(int).values,185        'APT30': df_merged['APT30'].fillna(0).astype(int).values,186        'Codoso_Gh0st_1': df_merged['Codoso_Gh0st_1'].fillna(0).astype(int).values,187    }188    189    # 建立 id2label 映射190    label_list = sorted(df_merged['true_label'].unique())191    label2id = {l: i for i, l in enumerate(label_list)}192    id2label = {i: l for l, i in label2id.items()}193    194    # 保存195    background_data = {196        'embeddings': embeddings,197        'labels': labels,198        'id2label': id2label199    }200    201    with open(output_path, 'wb') as f:202        pickle.dump(background_data, f)203    204    print(f"✅ 背景資料已保存至: {output_path}")205    print(f"   - 資料筆數: {len(embeddings)}")206    print(f"   - Embedding 維度: {embeddings.shape[1]}")207    print(f"   - 特徵數量: {len(labels)}")