{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":4117,"databundleVersionId":46665,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-05T07:47:07.416202Z","iopub.execute_input":"2025-04-05T07:47:07.416567Z","iopub.status.idle":"2025-04-05T07:47:07.425639Z","shell.execute_reply.started":"2025-04-05T07:47:07.41653Z","shell.execute_reply":"2025-04-05T07:47:07.424607Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile HexVectorizer.py\n\nfrom sklearn.feature_extraction.text import CountVectorizer\n\nimport os\nimport re\n\n# Byte file properties\nLINE_LEN = 16\nADDR_LEN = 9\n\n\ndef hex_to_str(hex_line):\n    \"\"\"\n    Function to strip \\r, \\n and remove address\n    of length eight character+space from hex line.\n    \"\"\"\n    return hex_line.decode().strip()[ADDR_LEN:]\n\n\ndef remove_non_hex(hex_str):\n    \"\"\"\n    Function to remove non hex characters.\n    \"\"\"\n    # Replace non hex characters with empty string.\n    hex_str = re.sub(r\"[^0-9A-F\\s]+\", \"\", hex_str, flags=re.IGNORECASE)\n    # Replace multiple spaces with single space.\n    hex_str = re.sub(r\"\\s+\", \" \", hex_str)\n\n    return hex_str.strip().lower()\n\n\nclass Vectorizer(CountVectorizer):\n    \"\"\"\n    Convert strings to vectors\n    \"\"\"\n\n    def transform_byte_file(self, ts_file):\n        \"\"\"\n        Function to convert a byte-file into a data-point/row in CSV file.\n        Each data-point will have file-name, file-size & byte-string columns.\n        \"\"\"\n\n        # Open the byte-file for reading.\n        with open(ts_file, \"rb\") as byt_f:\n            # Remove memory address in the beginning of each line and concatenate\n            # all lines in the byte-file into a single string separated by space.\n            byt_str = \" \".join([hex_to_str(line) for line in byt_f.readlines()])\n            byt_str = remove_non_hex(byt_str)\n\n            # Get the byte-file name.\n            f_path, _ = os.path.splitext(byt_f.name)  # Full path, extension.\n            _, f_name = f_path.rsplit(\"/\", 1)  # Relative path, file-name.\n\n            # Get the byte-file size.\n            file_info = os.stat(byt_f.name)\n            f_size = file_info.st_size\n\n            bow = self.transform([byt_str]).toarray()[0].tolist()\n            return [f_name, f_size] + bow\n\n    def get_feature_names(self):\n        \"\"\"\n        Function to return vocabulary as feature names.\n        \"\"\"\n        return self.get_feature_names_out().tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T07:47:45.760892Z","iopub.execute_input":"2025-04-05T07:47:45.761472Z","iopub.status.idle":"2025-04-05T07:47:45.770181Z","shell.execute_reply.started":"2025-04-05T07:47:45.761416Z","shell.execute_reply":"2025-04-05T07:47:45.769084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Data manipulation libraries.\nimport numpy as np\nimport pandas as pd\n\n# Data visualization libraries.\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom prettytable import PrettyTable\n\n# Data modeling libraries.\nimport sklearn\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler, MinMaxScaler\nfrom sklearn.manifold import TSNE\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils.class_weight import compute_sample_weight\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import log_loss\nfrom sklearn.model_selection import GridSearchCV\n\n# General Imports\nimport os\nimport csv\nimport re\nimport time\nimport math\nfrom tqdm import tqdm\nimport multiprocessing\nfrom multiprocessing import Pool\n\n# Custom modules\nfrom HexVectorizer import Vectorizer\n\n\n# Library versions.\nprint(\"NumPy version:\", np.__version__)\nprint(\"Pandas version:\", pd.__version__)\nprint(\"Matplotlib version:\", matplotlib.__version__)\nprint(\"Seaborn version:\", sns.__version__)\nprint(\"Scikit-learn version:\", sklearn.__version__)\n\n# Configure NumPy.\n# Set `Line width` to Maximum 130 characters in the output, post which it will continue in next line.\nnp.set_printoptions(linewidth=130)\n\n# Configure Pandas.\n# Set display width to maximum 130 characters in the output, post which it will continue in next line.\npd.options.display.width = 130\n# pd.options.display.max_rows = None  # Very dangerous! if dataset is large.\n\n# Configure Seaborn.\nsns.set_style(\"whitegrid\")  # Set white background with grid.\nsns.set_palette(\"deep\")  # Set color palette.\nsns.set_context(\"paper\", font_scale=1.5)  # Set font to scale 1.5 more than normal.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T07:47:46.86325Z","iopub.execute_input":"2025-04-05T07:47:46.863645Z","iopub.status.idle":"2025-04-05T07:47:46.878049Z","shell.execute_reply.started":"2025-04-05T07:47:46.863594Z","shell.execute_reply":"2025-04-05T07:47:46.876591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import multiprocessing\nfrom multiprocessing import Pool\nfrom tqdm import tqdm\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T07:47:49.128366Z","iopub.execute_input":"2025-04-05T07:47:49.128944Z","iopub.status.idle":"2025-04-05T07:47:49.134114Z","shell.execute_reply.started":"2025-04-05T07:47:49.128895Z","shell.execute_reply":"2025-04-05T07:47:49.132718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def parallelize(task, data):\n    \"\"\"\n    Function to parallelize task() for list of items passed as data.\n    \"\"\"\n    if __name__ == \"__main__\":\n        POOL_SIZE = multiprocessing.cpu_count()\n        pool = Pool(processes=POOL_SIZE)\n        print(\"Pool size:\", POOL_SIZE)\n\n        outputs = []\n        pbar = tqdm(total=len(data))\n        for output in pool.imap_unordered(task, data):\n            outputs.append(output)\n            pbar.update()\n        pbar.close()\n\n        pool.close()\n        pool.join()\n\n        return outputs\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T07:47:51.000977Z","iopub.execute_input":"2025-04-05T07:47:51.001331Z","iopub.status.idle":"2025-04-05T07:47:51.007184Z","shell.execute_reply.started":"2025-04-05T07:47:51.001296Z","shell.execute_reply":"2025-04-05T07:47:51.006119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_DIR = \"/kaggle/working/train\"\nLABELS_CSV = \"../input/malware-classification/trainLabels.csv\"\nTEST_DIR = \"../input/malware-classification/test/\"\nPREPS_DIR = \"/kaggle/input/msft-challenge\"\nBYTE_EXT = \".bytes\"\n\nBAL_CL_CSV = os.path.join(PREPS_DIR, \"trainLabels_bal.csv\")\nTR_VEC_CSV = os.path.join(PREPS_DIR, \"train_vec.csv\")\nTS_VEC_CSV = os.path.join(PREPS_DIR, \"test_vec.csv\")\n\n# Byte file properties\nLINE_LEN = 16\nADDR_LEN = 9\n\n\n# Malware class-labels.\nMALWARE_CLS = {\n    1: \"Ramnit\",\n    2: \"Lollipop\",\n    3: \"Kelihos_ver3\",\n    4: \"Vundo\",\n    5: \"Simda\",\n    6: \"Tracur\",\n    7: \"Kelihos_ver1\",\n    8: \"Obfuscator.ACY\",\n    9: \"Gatak\",\n}\nmw_codes = lambda: list(MALWARE_CLS.keys())\nmw_names = lambda: list(MALWARE_CLS.values())\n\n\nSAMPLE_SIZE = 200  # Number of samples taken from each class.\n# Function to pick random sample of size SAMPLE_SIZE from each class.\nget_sample = lambda grp: grp.sample(min(SAMPLE_SIZE, len(grp)))\n\n\ndef get_weights(cls):\n    class_weights = {\n        0: 1,\n        1: 1,\n        2: 1,\n        3: 1,\n        4: 4,\n        5: 1,\n        6: 1,\n        7: 1,\n        8: 1,\n    }\n\n    return [class_weights[cl] for cl in cls]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T07:47:52.680207Z","iopub.execute_input":"2025-04-05T07:47:52.680552Z","iopub.status.idle":"2025-04-05T07:47:52.689111Z","shell.execute_reply.started":"2025-04-05T07:47:52.680521Z","shell.execute_reply":"2025-04-05T07:47:52.687945Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file_paths = [os.path.join(TRAIN_DIR, Id + BYTE_EXT) for Id in cl_df_b[\"Id\"]]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T07:47:53.668751Z","iopub.execute_input":"2025-04-05T07:47:53.669119Z","iopub.status.idle":"2025-04-05T07:47:53.67669Z","shell.execute_reply.started":"2025-04-05T07:47:53.669087Z","shell.execute_reply":"2025-04-05T07:47:53.675324Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Class-labels DataFrame.\ncl_df = pd.read_csv(LABELS_CSV)\ncl_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T07:47:54.802038Z","iopub.execute_input":"2025-04-05T07:47:54.802381Z","iopub.status.idle":"2025-04-05T07:47:54.831545Z","shell.execute_reply.started":"2025-04-05T07:47:54.802355Z","shell.execute_reply":"2025-04-05T07:47:54.830448Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rows, cols = cl_df.shape\nprint(f\"There are around {rows} malware files available in the Train dataset.\")\ncls_count = cl_df[\"Class\"].value_counts().sort_index()\ncls_count","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T07:47:55.947276Z","iopub.execute_input":"2025-04-05T07:47:55.947708Z","iopub.status.idle":"2025-04-05T07:47:55.958286Z","shell.execute_reply.started":"2025-04-05T07:47:55.947671Z","shell.execute_reply":"2025-04-05T07:47:55.9572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BAL_CL_CSV = '/kaggle/working/balanced_classes.csv'  # Writable path\n\nif os.path.exists(BAL_CL_CSV) and os.stat(BAL_CL_CSV).st_size > 0:\n    cl_df_b = pd.read_csv(BAL_CL_CSV)\nelse:\n    cl_df_b = cl_df.groupby(\"Class\", group_keys=False).apply(get_sample)\n    cl_df_b.to_csv(BAL_CL_CSV, index=False)\n\nrows, cols = cl_df_b.shape\nprint(f\"Balanced subset will contain data of {rows} byte-files.\")\n\ncls_count = cl_df_b[\"Class\"].value_counts().sort_index()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T07:47:57.525875Z","iopub.execute_input":"2025-04-05T07:47:57.526228Z","iopub.status.idle":"2025-04-05T07:47:57.538823Z","shell.execute_reply.started":"2025-04-05T07:47:57.526199Z","shell.execute_reply":"2025-04-05T07:47:57.537592Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(6, 6))\n\nplt.pie(x=cls_count, labels=mw_names(), autopct=\"%1.0f%%\")\nplt.title(\"Malware Class-labels\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T07:47:59.145768Z","iopub.execute_input":"2025-04-05T07:47:59.146116Z","iopub.status.idle":"2025-04-05T07:47:59.346104Z","shell.execute_reply.started":"2025-04-05T07:47:59.146088Z","shell.execute_reply":"2025-04-05T07:47:59.344833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hex_chars = [\"0\", \"1\", \"2\", \"3\", \"4\", \"5\", \"6\", \"7\", \"8\", \"9\", \"a\", \"b\", \"c\", \"d\", \"e\", \"f\"]\n\nvectorizer = Vectorizer()\nvectorizer.vocabulary = [f\"{i}{j}\" for i in hex_chars for j in hex_chars]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T07:48:01.19828Z","iopub.execute_input":"2025-04-05T07:48:01.19866Z","iopub.status.idle":"2025-04-05T07:48:01.203967Z","shell.execute_reply.started":"2025-04-05T07:48:01.198589Z","shell.execute_reply":"2025-04-05T07:48:01.202629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"vectorizer.vocabulary\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T07:48:05.061944Z","iopub.execute_input":"2025-04-05T07:48:05.062292Z","iopub.status.idle":"2025-04-05T07:48:05.071222Z","shell.execute_reply.started":"2025-04-05T07:48:05.062259Z","shell.execute_reply":"2025-04-05T07:48:05.07012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_ftrs = [\"Id\", \"BytFSize\"] + vectorizer.get_feature_names()\nfile_paths = [os.path.join(TRAIN_DIR, Id) + BYTE_EXT for Id in cl_df_b[\"Id\"]]\nfile_count = len(file_paths)\n\nprint(f\"Select {file_count} byte-files from unzipped files for processing.\")\n\nif os.path.exists(TR_VEC_CSV) and os.stat(TR_VEC_CSV).st_size > 0:\n    print(\"Sample is already vectorized in:\", TR_VEC_CSV)\n    train_df = pd.read_csv(TR_VEC_CSV)\nelse:\n    print(\"Vectorization of balanced sample from original dataset...\")\n\n    # Parallelized File Processing.\n    tr_ftr_vecs = parallelize(vectorizer.transform_byte_file, file_paths)\n\n    # Convert vectors to DataFrame.\n    train_df = pd.DataFrame(tr_ftr_vecs, columns=final_ftrs)\n\n    def get_class(Id):\n        fltr = cl_df_b[\"Id\"] == Id\n        return cl_df_b.loc[fltr, \"Class\"].item()\n\n    # Append class labels.\n    train_df[\"Class\"] = train_df[\"Id\"].apply(get_class)\n\n    # Save DataFrame as CSV for future use.\n    train_df.to_csv(TR_VEC_CSV, index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T07:47:23.662955Z","iopub.execute_input":"2025-04-05T07:47:23.663366Z","iopub.status.idle":"2025-04-05T07:47:23.833893Z","shell.execute_reply.started":"2025-04-05T07:47:23.663329Z","shell.execute_reply":"2025-04-05T07:47:23.82558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T07:47:08.177839Z","iopub.status.idle":"2025-04-05T07:47:08.178263Z","shell.execute_reply":"2025-04-05T07:47:08.178108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}