{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":4117,"databundleVersionId":46665,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-02T08:12:51.742296Z","iopub.execute_input":"2025-07-02T08:12:51.742589Z","iopub.status.idle":"2025-07-02T08:12:51.752993Z","shell.execute_reply.started":"2025-07-02T08:12:51.742568Z","shell.execute_reply":"2025-07-02T08:12:51.751934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ndata_path = \"/kaggle/input/malware-classification\"\n\n# List available files\nprint(os.listdir(data_path))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T08:12:51.782512Z","iopub.execute_input":"2025-07-02T08:12:51.783108Z","iopub.status.idle":"2025-07-02T08:12:51.789093Z","shell.execute_reply.started":"2025-07-02T08:12:51.783084Z","shell.execute_reply":"2025-07-02T08:12:51.787908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Path to the dataset\ndata_path = \"/kaggle/input/malware-classification\"\n\n# Load the labels\nlabels_df = pd.read_csv(f\"{data_path}/trainLabels.csv\")\n\n# Show first 5 rows\nlabels_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T09:51:38.329484Z","iopub.execute_input":"2025-07-02T09:51:38.329823Z","iopub.status.idle":"2025-07-02T09:51:38.714263Z","shell.execute_reply.started":"2025-07-02T09:51:38.329797Z","shell.execute_reply":"2025-07-02T09:51:38.713293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ndata_path = \"/kaggle/input/malware-classification\"\ndf = pd.read_csv(f\"{data_path}/trainLabels.csv\")\ndf.head(10)  # shows 10 samples with Id and Class\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T10:12:30.842605Z","iopub.execute_input":"2025-07-02T10:12:30.842958Z","iopub.status.idle":"2025-07-02T10:12:30.865577Z","shell.execute_reply.started":"2025-07-02T10:12:30.842929Z","shell.execute_reply":"2025-07-02T10:12:30.864559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Mapping class numbers to family names\nfamily_map = {\n    1: \"Ramnit\",\n    2: \"Lollipop\",\n    3: \"Kelihos_ver3\",\n    4: \"Vundo\",\n    5: \"Simda\",\n    6: \"Tracur\",\n    7: \"Kelihos_ver1\",\n    8: \"Obfuscator.ACY\",\n    9: \"Gatak\"\n}\n\n# Add a new column for family name\nlabels_df[\"Family\"] = labels_df[\"Class\"].map(family_map)\n\n# Preview updated labels\nlabels_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T08:12:51.832486Z","iopub.execute_input":"2025-07-02T08:12:51.833710Z","iopub.status.idle":"2025-07-02T08:12:51.846958Z","shell.execute_reply.started":"2025-07-02T08:12:51.833656Z","shell.execute_reply":"2025-07-02T08:12:51.845547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load labels\ndata_path = \"/kaggle/input/malware-classification\"\ndf = pd.read_csv(f\"{data_path}/trainLabels.csv\")\n\n# Optional: Map class to family names\nfamily_map = {\n    1: \"Ramnit\", 2: \"Lollipop\", 3: \"Kelihos_ver3\", 4: \"Vundo\",\n    5: \"Simda\", 6: \"Tracur\", 7: \"Kelihos_ver1\", 8: \"Obfuscator.ACY\", 9: \"Gatak\"\n}\ndf[\"Family\"] = df[\"Class\"].map(family_map)\n\n# Smart sampling: get min(10, count) rows per class\nsampled_df = df.groupby(\"Class\", group_keys=False).apply(\n    lambda x: x.sample(n=min(10, len(x)), random_state=42)\n).reset_index(drop=True)\n\n# Show count per class in the sample\nprint(sampled_df[\"Class\"].value_counts())\n\n# Preview result\nsampled_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T10:15:15.613452Z","iopub.execute_input":"2025-07-02T10:15:15.613758Z","iopub.status.idle":"2025-07-02T10:15:15.649631Z","shell.execute_reply.started":"2025-07-02T10:15:15.613738Z","shell.execute_reply":"2025-07-02T10:15:15.648639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for cls in sorted(sampled_df[\"Class\"].unique()):\n    ids = sampled_df[sampled_df[\"Class\"] == cls][\"Id\"].tolist()\n    print(f\"Class {cls} Sample IDs:\\n\", ids, \"\\n\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T10:16:39.473239Z","iopub.execute_input":"2025-07-02T10:16:39.474046Z","iopub.status.idle":"2025-07-02T10:16:39.486908Z","shell.execute_reply.started":"2025-07-02T10:16:39.474007Z","shell.execute_reply":"2025-07-02T10:16:39.485767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels_df[\"Family\"].value_counts()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T08:12:51.884184Z","iopub.execute_input":"2025-07-02T08:12:51.884945Z","iopub.status.idle":"2025-07-02T08:12:51.894060Z","shell.execute_reply.started":"2025-07-02T08:12:51.884916Z","shell.execute_reply":"2025-07-02T08:12:51.893056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(10, 6))\nsns.countplot(data=labels_df, y=\"Family\", order=labels_df[\"Family\"].value_counts().index)\nplt.title(\"Malware Family Distribution\")\nplt.xlabel(\"Number of Samples\")\nplt.ylabel(\"Malware Family\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T08:12:51.895933Z","iopub.execute_input":"2025-07-02T08:12:51.896330Z","iopub.status.idle":"2025-07-02T08:12:52.130084Z","shell.execute_reply.started":"2025-07-02T08:12:51.896297Z","shell.execute_reply":"2025-07-02T08:12:52.128853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\ndef extract_byte_histogram(file_path):\n    counts = np.zeros(256, dtype=int)\n    try:\n        with open(file_path, 'r') as f:\n            for line in f:\n                parts = line.strip().split()[1:]  # skip the address\n                for byte_str in parts:\n                    if byte_str != '??':\n                        try:\n                            byte_val = int(byte_str, 16)\n                            counts[byte_val] += 1\n                        except ValueError:\n                            continue\n    except Exception as e:\n        print(f\"Error reading {file_path}: {e}\")\n    return counts\n\n# Path to dataset\ndata_path = \"/kaggle/input/malware-classification\"\nbytes_path = os.path.join(data_path, \"train\")\n\n# Load labels\nlabel_df = pd.read_csv(os.path.join(data_path, \"trainLabels.csv\"))\nsample_ids = label_df[\"Id\"].tolist()[:100]  # First 100 samples for demo\nlabel_map = dict(zip(label_df[\"Id\"], label_df[\"Class\"]))\n\nX = []\ny = []\nfile_names = []\n\nfor file_id in tqdm(sample_ids):\n    file_path = os.path.join(bytes_path, file_id + \".bytes\")\n    if os.path.exists(file_path):\n        hist = extract_byte_histogram(file_path)\n        X.append(hist)\n        y.append(label_map[file_id])\n        file_names.append(file_id)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T09:53:48.339959Z","iopub.execute_input":"2025-07-02T09:53:48.340264Z","iopub.status.idle":"2025-07-02T09:53:48.370817Z","shell.execute_reply.started":"2025-07-02T09:53:48.340244Z","shell.execute_reply":"2025-07-02T09:53:48.369671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create the feature DataFrame\ndf_features = pd.DataFrame(X, columns=[f'byte_{i:02X}' for i in range(256)])\ndf_features[\"label\"] = y\ndf_features[\"Id\"] = file_names\n\n# Map label → malware family\nfamily_map = {\n    1: \"Ramnit\", 2: \"Lollipop\", 3: \"Kelihos_ver3\", 4: \"Vundo\",\n    5: \"Simda\", 6: \"Tracur\", 7: \"Kelihos_ver1\", 8: \"Obfuscator.ACY\", 9: \"Gatak\"\n}\ndf_features[\"Family\"] = df_features[\"label\"].map(family_map)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count NaN values\nprint(\"Total NaNs:\", df_features.isna().sum().sum())\n\n# Check data types\nprint(\"Data types:\\n\", df_features.dtypes.value_counts())\n\n# Preview suspicious rows\nprint(df_features[df_features.isna().any(axis=1)].head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T09:52:22.470155Z","iopub.execute_input":"2025-07-02T09:52:22.470479Z","iopub.status.idle":"2025-07-02T09:52:22.559006Z","shell.execute_reply.started":"2025-07-02T09:52:22.470453Z","shell.execute_reply":"2025-07-02T09:52:22.557682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check if any rows are all NaN or all zeros\nprint(df_features.isnull().sum().sum())         # Total NaNs\nprint((df_features.drop(columns=[\"label\"], errors='ignore') == 0).all(axis=1).sum())  # All-zero rows\n\n# Optionally drop NaNs\ndf_features = df_features.dropna()\n\ndf_features.head()\ndf_features.describe()\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Remove label column for correlation analysis\ncorr_matrix = df_features.drop(\"label\", axis=1).corr()\n\nplt.figure(figsize=(16, 12))\nsns.heatmap(corr_matrix, cmap='coolwarm', linewidths=0.2)\nplt.title(\"Correlation Heatmap of Byte Histogram Features\", fontsize=16)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-02T08:25:45.164427Z","iopub.execute_input":"2025-07-02T08:25:45.165673Z","iopub.status.idle":"2025-07-02T08:25:46.600101Z","shell.execute_reply.started":"2025-07-02T08:25:45.165630Z","shell.execute_reply":"2025-07-02T08:25:46.598832Z"}},"outputs":[],"execution_count":null}]}