{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":4117,"databundleVersionId":46665,"sourceType":"competition"},{"sourceId":6140297,"sourceType":"datasetVersion","datasetId":3080209}],"dockerImageVersionId":30407,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport shutil\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import StratifiedShuffleSplit\nimport re\nfrom sklearn.feature_extraction.text import TfidfTransformer\nimport cv2 as cv\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-12-29T16:30:32.905582Z","iopub.execute_input":"2024-12-29T16:30:32.906811Z","iopub.status.idle":"2024-12-29T16:30:34.048200Z","shell.execute_reply.started":"2024-12-29T16:30:32.906766Z","shell.execute_reply":"2024-12-29T16:30:34.047175Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_values={\n    1:'Ramnit',\n    2:'Lollipop',\n    3:'Kelihos_ver3',\n    4:'Vundo',\n    5:'Simda',\n    6:'Tracur',\n    7:'Kelihos_ver1',\n    8:'Obfuscator.ACY',\n    9:'Gatak'\n}\n\nget_vocab=False\nget_feature_dataset=False\nget_img=True\n\nmemmory_addr_pattern=re.compile('[\\dABCDEF][\\dABCDEF][\\dABCDEF]+ ')","metadata":{"execution":{"iopub.status.busy":"2024-12-29T16:30:34.050218Z","iopub.execute_input":"2024-12-29T16:30:34.050632Z","iopub.status.idle":"2024-12-29T16:30:34.056499Z","shell.execute_reply.started":"2024-12-29T16:30:34.050593Z","shell.execute_reply":"2024-12-29T16:30:34.055123Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_sample(folder:str,filenames:list=[],logs:bool=False):\n    directory_command=f'\"-o{folder}\"'\n    if logs: print(directory_command)\n    filnames_command=f'{\" \".join(filenames)}'\n    if logs: print(filnames_command)\n    command=f\"7z x /kaggle/input/malware-classification/dataSample.7z {directory_command} {filnames_command} -r\"\n    command=command+' >/dev/null 2>&1' #done to hide console output\n    if logs: print(command)\n    os.system(command)\n#     !7z x /kaggle/input/malware-classification/dataSample.7z f'\"-o {directory}\"' f'{\" \".join(filenames)}'","metadata":{"execution":{"iopub.status.busy":"2024-12-29T16:30:34.085433Z","iopub.execute_input":"2024-12-29T16:30:34.085809Z","iopub.status.idle":"2024-12-29T16:30:34.101622Z","shell.execute_reply.started":"2024-12-29T16:30:34.085772Z","shell.execute_reply":"2024-12-29T16:30:34.100382Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_data(folder:str,filenames:list,logs:bool=False,test=False):\n    if test:\n        zip_folder='test.7z'\n    else:\n        zip_folder='train.7z'\n    \n    directory_command=f'\"-o{folder}\"'\n    filnames_command=f'{\" \".join(filenames)}'\n    if logs: print(filnames_command)\n    command=f\"7z x /kaggle/input/malware-classification/{zip_folder} {directory_command} {filnames_command} -r\"\n    command=command+' >/dev/null 2>&1' #done to hide console output\n    if logs: print(command)\n    if logs: print('started extracting')\n    os.system(command)\n    if logs: print('Finished extracting')","metadata":{"execution":{"iopub.status.busy":"2024-12-29T16:30:35.430005Z","iopub.execute_input":"2024-12-29T16:30:35.430441Z","iopub.status.idle":"2024-12-29T16:30:35.437479Z","shell.execute_reply.started":"2024-12-29T16:30:35.430398Z","shell.execute_reply":"2024-12-29T16:30:35.436312Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def delete_folder(folder:str):\n    shutil.rmtree(folder)","metadata":{"execution":{"iopub.status.busy":"2024-12-29T16:30:35.808604Z","iopub.execute_input":"2024-12-29T16:30:35.809599Z","iopub.status.idle":"2024-12-29T16:30:35.814422Z","shell.execute_reply.started":"2024-12-29T16:30:35.809543Z","shell.execute_reply":"2024-12-29T16:30:35.813195Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_exe(code_bytes:str):\n    '''\n    This function will remove the memory addresses and the new lines while keeping the other values\n    '''\n    return re.sub(memmory_addr_pattern,\"\",code_bytes).replace(\"\\\\r\\\\n\", \" \").replace(\"b'\",\"\").replace(\"'\",\"\").replace('?','')","metadata":{"execution":{"iopub.status.busy":"2024-12-29T16:30:37.653270Z","iopub.execute_input":"2024-12-29T16:30:37.654732Z","iopub.status.idle":"2024-12-29T16:30:37.660992Z","shell.execute_reply.started":"2024-12-29T16:30:37.654682Z","shell.execute_reply":"2024-12-29T16:30:37.659697Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#from https://www.kaggle.com/code/dheemanthbhat/malware-classification-with-multiprocessing\nhex_chars = [\"0\", \"1\", \"2\", \"3\", \"4\", \"5\", \"6\", \"7\", \"8\", \"9\", \"a\", \"b\", \"c\", \"d\", \"e\", \"f\"]\n\nvocab_ls = [f\"{i}{j}\" for i in hex_chars for j in hex_chars]\n\nindex=0\nvocab_dic={}\nfor i in hex_chars:\n    for j in hex_chars:\n        vocab_dic[f\"{i}{j}\"]=index\n        index+=1        ","metadata":{"execution":{"iopub.status.busy":"2024-12-29T16:30:23.074246Z","iopub.status.idle":"2024-12-29T16:30:23.074775Z","shell.execute_reply.started":"2024-12-29T16:30:23.074496Z","shell.execute_reply":"2024-12-29T16:30:23.074525Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels=pd.read_csv('/kaggle/input/malware-classification/trainLabels.csv')\nlabels","metadata":{"execution":{"iopub.status.busy":"2024-12-29T16:30:23.076184Z","iopub.status.idle":"2024-12-29T16:30:23.076708Z","shell.execute_reply.started":"2024-12-29T16:30:23.076432Z","shell.execute_reply":"2024-12-29T16:30:23.076462Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels[labels['Id']=='01kcPWA9K2BOxQeS5Rju']","metadata":{"execution":{"iopub.status.busy":"2024-12-28T15:23:02.036740Z","iopub.execute_input":"2024-12-28T15:23:02.037265Z","iopub.status.idle":"2024-12-28T15:23:02.055116Z","shell.execute_reply.started":"2024-12-28T15:23:02.037197Z","shell.execute_reply":"2024-12-28T15:23:02.053924Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels['Class'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-12-28T15:23:02.056816Z","iopub.execute_input":"2024-12-28T15:23:02.057569Z","iopub.status.idle":"2024-12-28T15:23:02.075848Z","shell.execute_reply.started":"2024-12-28T15:23:02.057508Z","shell.execute_reply":"2024-12-28T15:23:02.074310Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"classes=labels.groupby('Class').count()\nclasses=classes.rename(columns={'Id':'count'})\nclasses['percentage']=classes['count']/classes['count'].sum()\nclasses","metadata":{"execution":{"iopub.status.busy":"2024-12-28T15:23:02.077324Z","iopub.execute_input":"2024-12-28T15:23:02.077635Z","iopub.status.idle":"2024-12-28T15:23:02.097693Z","shell.execute_reply.started":"2024-12-28T15:23:02.077605Z","shell.execute_reply":"2024-12-28T15:23:02.096516Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"empty_files=['a9oIzfw03ED4lTBCt52Y.bytes',\n'cf4nzsoCmudt1kwleOTI.bytes',\n'58kxhXouHzFd4g3rmInB.bytes',\n'IidxQvXrlBkWPZAfcqKT.bytes',\n'da3XhOZzQEbKVtLgMYWv.bytes',\n'fRLS3aKkijp4GH0Ds6Pv.bytes',\n'6tfw0xSL2FNHOCJBdlaA.bytes',\n'd0iHC6ANYGon7myPFzBe.bytes']","metadata":{"execution":{"iopub.status.busy":"2024-12-28T15:23:02.099181Z","iopub.execute_input":"2024-12-28T15:23:02.099612Z","iopub.status.idle":"2024-12-28T15:23:02.108810Z","shell.execute_reply.started":"2024-12-28T15:23:02.099578Z","shell.execute_reply":"2024-12-28T15:23:02.107311Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"extract_data(folder='tmp',filenames=empty_files)","metadata":{"execution":{"iopub.status.busy":"2024-12-28T15:23:02.110930Z","iopub.execute_input":"2024-12-28T15:23:02.112027Z","iopub.status.idle":"2024-12-28T15:24:58.362868Z","shell.execute_reply.started":"2024-12-28T15:23:02.111976Z","shell.execute_reply":"2024-12-28T15:24:58.361631Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\"?\" means no data => malware","metadata":{}},{"cell_type":"code","source":"tmp_files=os.listdir('tmp/train')\n\nclean_codes=[]\nfor file in tmp_files:\n    with open(f'tmp/train/{file}','rb') as f:\n        code_bytes=str(f.read())\n        clean_codes.append(code_bytes)#process_exe(code_bytes))\n        ","metadata":{"execution":{"iopub.status.busy":"2024-12-28T15:24:58.364282Z","iopub.execute_input":"2024-12-28T15:24:58.364587Z","iopub.status.idle":"2024-12-28T15:24:58.465370Z","shell.execute_reply.started":"2024-12-28T15:24:58.364557Z","shell.execute_reply":"2024-12-28T15:24:58.464359Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for code in clean_codes:\n    print(code[:100])\n    print('----------------------')","metadata":{"execution":{"iopub.status.busy":"2024-12-28T15:24:58.466571Z","iopub.execute_input":"2024-12-28T15:24:58.466872Z","iopub.status.idle":"2024-12-28T15:24:58.473791Z","shell.execute_reply.started":"2024-12-28T15:24:58.466838Z","shell.execute_reply":"2024-12-28T15:24:58.472310Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"what does \"?\" mean? and how to deal with it? add \"?\" to vocab or remove the files?","metadata":{}},{"cell_type":"code","source":"delete_folder('tmp')","metadata":{"execution":{"iopub.status.busy":"2024-12-28T15:24:58.475545Z","iopub.execute_input":"2024-12-28T15:24:58.475937Z","iopub.status.idle":"2024-12-28T15:24:58.492434Z","shell.execute_reply.started":"2024-12-28T15:24:58.475901Z","shell.execute_reply":"2024-12-28T15:24:58.490868Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels[labels['Id'].isin(list(map(lambda x: x.split('.')[0],empty_files)))]","metadata":{"execution":{"iopub.status.busy":"2024-12-28T15:24:58.493845Z","iopub.execute_input":"2024-12-28T15:24:58.494277Z","iopub.status.idle":"2024-12-28T15:24:58.513194Z","shell.execute_reply.started":"2024-12-28T15:24:58.494208Z","shell.execute_reply":"2024-12-28T15:24:58.512033Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del_files=list(map(lambda x: x.split('.')[0],empty_files))\nlabels=labels[~labels['Id'].isin(del_files)]","metadata":{"execution":{"iopub.status.busy":"2024-12-28T15:24:58.514537Z","iopub.execute_input":"2024-12-28T15:24:58.514981Z","iopub.status.idle":"2024-12-28T15:24:58.531353Z","shell.execute_reply.started":"2024-12-28T15:24:58.514935Z","shell.execute_reply":"2024-12-28T15:24:58.530134Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"there are 1541 samples for class 1. It is ok to skip these 8 samples for images.","metadata":{}},{"cell_type":"code","source":"# count vectorizer\n# vectorizer\n# tf idf\n# check:\n#     https://www.tensorflow.org/api_docs/python/tf/keras/layers/TextVectorization","metadata":{"execution":{"iopub.status.busy":"2024-12-28T15:24:58.537365Z","iopub.execute_input":"2024-12-28T15:24:58.537868Z","iopub.status.idle":"2024-12-28T15:24:58.549057Z","shell.execute_reply.started":"2024-12-28T15:24:58.537829Z","shell.execute_reply":"2024-12-28T15:24:58.547617Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"each line is 16 bytes","metadata":{}},{"cell_type":"code","source":"# re.sub(memmory_addr_pattern,\"\",code).replace(\"\\\\r\\\\n\", \" \").replace(\"b'\",\"\").replace(\"'\",\"\")","metadata":{"execution":{"iopub.status.busy":"2024-12-28T15:24:58.550647Z","iopub.execute_input":"2024-12-28T15:24:58.551000Z","iopub.status.idle":"2024-12-28T15:24:58.562345Z","shell.execute_reply.started":"2024-12-28T15:24:58.550967Z","shell.execute_reply":"2024-12-28T15:24:58.560926Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The vectorizers will process a list of strings","metadata":{}},{"cell_type":"markdown","source":"i want to get a complete vocab from all the files and use it","metadata":{}},{"cell_type":"markdown","source":"The code below can be used to get the vocab. Since the data is huge, we cannot have all in one run. we can use to get ngrams if we want.","metadata":{}},{"cell_type":"markdown","source":"process the train data","metadata":{}},{"cell_type":"code","source":"file_names=labels['Id'].tolist()\n\n#I only want the byte files\nfile_names=list(map(lambda file: file+'.bytes',file_names))\n\n# NAme of temporary folder\ntemp_folder='tmp'\n\nstart=0\nstep=2000# extract 2000 by a time\nnumber_files=len(file_names)\n\nprocess_batch=True\n\nfile_size={}\n\nngram=1\n#Get the vectorizer\nvectorizer = CountVectorizer(ngram_range=(ngram,ngram))\n\ni=1\n\nwhile(get_vocab):\n    print(f'started {i} run')\n    \n    end=start+step\n    \n    if end>(number_files):\n        end=number_files\n        \n    current_files=file_names[start:end]\n    \n    #extract the data to a temporary folder tmp\n    extract_data(folder=temp_folder,filenames=current_files)\n    print('finished extracting')\n    \n    current_tot_size=0\n    for this_file in os.listdir(temp_folder+'/train'):\n        #get the size in MBytes\n        size=os.path.getsize(f'{temp_folder}/train/{this_file}')/1e6\n        current_tot_size+=size\n        \n        #save\n        file_size[this_file]=size\n    \n    #if size greater than 10GB, then enter the loop\n    if process_batch or current_tot_size>10000 or end==number_files:\n        print('started processing')\n        #we will process and delete the extracted files\n        \n        #get files in tmp folder\n        tmp_files=os.listdir(temp_folder+'/train')\n        \n        clean_codes=[]\n        for file in tmp_files:\n            with open(f'{temp_folder}/train/{file}','rb') as f:\n                code_bytes=str(f.read())\n                clean_codes.append(process_exe(code_bytes))\n        \n        #here we can \n        \n        \n        print('fit the vectorizer')\n        #fit the vectorizer\n        vectorizer.fit(clean_codes)\n    \n        \n        print('vocab len:', len(vectorizer.vocabulary_))\n        #delete folder with its files\n        delete_folder(temp_folder)\n    \n    if end==number_files:\n        break\n    \n    start=end\n    i+=1\n\n# vectorizer.vocabulary.keys()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The file size is useful. I found an article that has suggestions for the image sizes depending on the size of the file.","metadata":{}},{"cell_type":"markdown","source":"the code below will generate a dataframe with count or TF-IDF","metadata":{}},{"cell_type":"code","source":"count_vec = CountVectorizer(ngram_range=(1,1),vocabulary=vocab_dic)\ntf_vec = TfidfVectorizer(ngram_range=(1,1),vocabulary=vocab_dic)","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:38.928192Z","iopub.execute_input":"2023-07-10T21:00:38.928622Z","iopub.status.idle":"2023-07-10T21:00:38.941313Z","shell.execute_reply.started":"2023-07-10T21:00:38.928577Z","shell.execute_reply":"2023-07-10T21:00:38.939817Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"a=count_vec.transform(['aa aa bb 01 01 03 aa bb 02','00 01 02 ff 26 ff 00'])\ndf=pd.DataFrame(columns=['filename','size']+vocab_ls)\ncounts=a.toarray()\nfor i in range(len(counts)):\n    # get filename depending on the current files being processed and\n    df.loc[i,['filename','size']]=0,1#filename,file_size[filename]\n    df.loc[i,vocab_ls]=counts[i]\ndf[['filename','size','00','01','02','03','26','aa','bb','ff']]","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:38.943595Z","iopub.execute_input":"2023-07-10T21:00:38.944024Z","iopub.status.idle":"2023-07-10T21:00:38.995606Z","shell.execute_reply.started":"2023-07-10T21:00:38.943985Z","shell.execute_reply":"2023-07-10T21:00:38.994257Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file_names=labels['Id'].tolist()\n\n#I only want the byte files\nfile_names=list(map(lambda file: file+'.bytes',file_names))\n\n# Name of temporary folder\ntemp_folder='tmp'\n\nstart=0\nstep=2000# extract 2000 by a time\nnumber_files=len(file_names)\n\nprocess_batch=True\n\ncount_df=pd.DataFrame(columns=['filename','size(MB)']+vocab_ls)\n\nfile_size={}\n\n#Get the vectorizer\nvectorizer = CountVectorizer(ngram_range=(1,1),vocabulary=vocab_dic)\nprint(len(vectorizer.vocabulary))\n\ni=1\n\nwhile(get_feature_dataset):\n    print(f'started {i} run')\n    \n    end=start+step\n    \n    if end>(number_files):\n        end=number_files\n    \n    print(f'start:{start}, end:{end}')\n        \n    current_files=file_names[start:end]\n    \n    #extract the data to a temporary folder tmp\n    extract_data(folder=temp_folder,filenames=current_files)\n    print('finished extracting')\n    \n    current_tot_size=0\n    for this_file in os.listdir(temp_folder+'/train'):\n        #get the size in MBytes\n        size=os.path.getsize(f'{temp_folder}/train/{this_file}')/1e6\n        current_tot_size+=size\n        \n        #save\n        file_size[this_file]=size\n    \n    #if size greater than 10GB, then enter the loop\n    if process_batch or current_tot_size>10000 or end==number_files:\n        print('started processing')\n        #we will process and delete the extracted files\n        \n        #get files in tmp folder\n        tmp_files=os.listdir(temp_folder+'/train')\n        print(len(tmp_files))\n        \n        cur_index=len(count_df)\n        clean_codes=[]\n        for code_index,file in enumerate(tmp_files):\n            with open(f'{temp_folder}/train/{file}','rb') as f:\n                code_bytes=str(f.read())\n        \n                #here we can get the count\n                counts=count_vec.transform([process_exe(code_bytes)]).toarray()\n                \n                #add it to the dataframe\n                count_df.loc[cur_index+code_index,['filename','size(MB)']]=file,file_size[file]#filename,file_size[filename]\n                count_df.loc[cur_index+code_index,vocab_ls]=counts#[code_index]\n        \n        #delete folder with its files\n        delete_folder(temp_folder)\n    \n    if end==number_files:\n        break\n    \n    start=end\n    i+=1","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:38.997630Z","iopub.execute_input":"2023-07-10T21:00:38.998025Z","iopub.status.idle":"2023-07-10T21:00:39.029439Z","shell.execute_reply.started":"2023-07-10T21:00:38.997986Z","shell.execute_reply":"2023-07-10T21:00:39.027352Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if(get_feature_dataset):\n    byte_df=labels.copy()\n    byte_df['Id']=byte_df['Id'].astype(str)+'.bytes'\n    byte_df['Id']=byte_df['Id'].astype(str)\n    count_df=byte_df.rename(columns={'Id':'filename'}).merge(count_df,on='filename',how='right')\n    byte_df[byte_df['Id'].isin(count_df['filename'].tolist())]","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:39.040408Z","iopub.execute_input":"2023-07-10T21:00:39.041104Z","iopub.status.idle":"2023-07-10T21:00:39.050830Z","shell.execute_reply.started":"2023-07-10T21:00:39.041060Z","shell.execute_reply":"2023-07-10T21:00:39.049393Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"count_path='/kaggle/input/malware-detection-different-representations/file_counts.csv'\nif os.path.exists(count_path) and not(get_feature_dataset):\n    count_df=pd.read_csv(count_path)","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:39.052970Z","iopub.execute_input":"2023-07-10T21:00:39.053369Z","iopub.status.idle":"2023-07-10T21:00:39.067128Z","shell.execute_reply.started":"2023-07-10T21:00:39.053331Z","shell.execute_reply":"2023-07-10T21:00:39.065895Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"count_df","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:39.068723Z","iopub.execute_input":"2023-07-10T21:00:39.070013Z","iopub.status.idle":"2023-07-10T21:00:39.091068Z","shell.execute_reply.started":"2023-07-10T21:00:39.069968Z","shell.execute_reply":"2023-07-10T21:00:39.089653Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tmp_count_df=count_df.set_index('filename')\ntmp_count_df","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:39.093107Z","iopub.execute_input":"2023-07-10T21:00:39.093566Z","iopub.status.idle":"2023-07-10T21:00:39.126877Z","shell.execute_reply.started":"2023-07-10T21:00:39.093524Z","shell.execute_reply":"2023-07-10T21:00:39.125019Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"describe_df=tmp_count_df.describe(include='all').loc['count']\ndescribe_df[describe_df<10868]","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:39.128860Z","iopub.execute_input":"2023-07-10T21:00:39.129349Z","iopub.status.idle":"2023-07-10T21:00:39.492862Z","shell.execute_reply.started":"2023-07-10T21:00:39.129307Z","shell.execute_reply":"2023-07-10T21:00:39.491327Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"the code below will generate the TF IDF representation","metadata":{}},{"cell_type":"code","source":"tf_idf_path='/kaggle/input/malware-detection-different-representations/file_tf_idf.csv'\nif (os.path.exists(tf_idf_path) and (not get_feature_dataset)):\n    tfidf_trans_df=pd.read_csv(tf_idf_path)\nelse:\n    transformer = TfidfTransformer()\n    tfidf_trans = transformer.fit_transform(tmp_count_df[vocab_ls])\n    tfidf_trans_df = pd.DataFrame(tfidf_trans.toarray(), index = tmp_count_df.index, columns=vocab_ls)\n    tfidf_trans_df=tfidf_trans_df.reset_index()\n    tfidf_trans_df['size(MB)']=tmp_count_df['size(MB)'].values\n    tfidf_trans_df['Class']=tmp_count_df['Class'].values\n    tfidf_trans_df=tfidf_trans_df[['filename','size(MB)','Class']+vocab_ls]","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:39.494694Z","iopub.execute_input":"2023-07-10T21:00:39.495156Z","iopub.status.idle":"2023-07-10T21:00:40.977284Z","shell.execute_reply.started":"2023-07-10T21:00:39.495110Z","shell.execute_reply":"2023-07-10T21:00:40.975901Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tfidf_trans_df","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:40.978855Z","iopub.execute_input":"2023-07-10T21:00:40.979240Z","iopub.status.idle":"2023-07-10T21:00:41.023892Z","shell.execute_reply.started":"2023-07-10T21:00:40.979203Z","shell.execute_reply":"2023-07-10T21:00:41.022423Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1>Images</h1>","metadata":{}},{"cell_type":"markdown","source":"tested 1000 files to images. saved as PNG and read the image. calculated the sum of the difference between the read image and the original one. the difference for all is 0.","metadata":{}},{"cell_type":"code","source":"#create the directories\n\nif get_img:\n    os.mkdir('vit')#main directory for images\n    os.mkdir('vit/train/')\n    os.mkdir('vit/val/')\n    os.mkdir('vit/test/')\n    \n    os.mkdir('cnn')#main directory for images\n    os.mkdir('cnn/train/')\n    os.mkdir('cnn/val/')\n    os.mkdir('cnn/test/')\n    \n#     for label,class_name in list(class_values.keys()):#items()):\n    for label in sorted(list(class_values.keys())):#items()):\n        class_name=class_values[label]\n        os.mkdir(f'vit/train/{label}-{class_name}')\n        os.mkdir(f'vit/val/{label}-{class_name}')\n        os.mkdir(f'vit/test/{label}-{class_name}')\n        \n        os.mkdir(f'cnn/train/{label}-{class_name}')\n        os.mkdir(f'cnn/val/{label}-{class_name}')\n        os.mkdir(f'cnn/test/{label}-{class_name}')\n        ","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:17:07.608028Z","iopub.execute_input":"2023-07-10T21:17:07.608439Z","iopub.status.idle":"2023-07-10T21:17:07.623679Z","shell.execute_reply.started":"2023-07-10T21:17:07.608403Z","shell.execute_reply":"2023-07-10T21:17:07.622092Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h4>Now divide the dataset to train-val-test<h4/>","metadata":{}},{"cell_type":"code","source":"def data_split(file_names,file_labels,**kwargs):\n    splitter=StratifiedShuffleSplit(**kwargs)\n\n    train_index, test_index=next(splitter.split(file_names, file_labels))\n\n    train_files=file_names[train_index]\n    train_labels=file_labels[train_index]\n\n    test_files=file_names[test_index]\n    test_labels=file_labels[test_index]\n    \n    return train_files,train_labels,test_files,test_labels","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:01:49.434637Z","iopub.execute_input":"2023-07-10T21:01:49.436154Z","iopub.status.idle":"2023-07-10T21:01:49.445506Z","shell.execute_reply.started":"2023-07-10T21:01:49.436073Z","shell.execute_reply":"2023-07-10T21:01:49.443853Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file_names=labels['Id'].values\nfile_labels=labels['Class'].values\nrandom_state=39\n\ntrain_files,train_labels,test_files,test_labels=data_split(file_names,file_labels,n_splits=1, test_size=0.3, random_state=random_state)\n\nval_files,val_labels,test_files,test_labels=data_split(test_files,test_labels,n_splits=1, test_size=0.5, random_state=random_state)\n\nprint(f\"  Train: size={len(train_files)}, examples={train_files}\\n\")\nprint(f\"  Test:  size={len(val_files)}, examples={val_files}\\n\")\nprint(f\"  Test:  size={len(test_files)}, examples={test_files}\\n\")","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:01:49.631480Z","iopub.execute_input":"2023-07-10T21:01:49.631976Z","iopub.status.idle":"2023-07-10T21:01:49.654287Z","shell.execute_reply.started":"2023-07-10T21:01:49.631933Z","shell.execute_reply":"2023-07-10T21:01:49.652871Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def code_to_image(code_bytes:str,filename:str,check:bool=False):\n    img= np.frombuffer(bytes.fromhex(code_bytes), dtype='uint8')#dtype='uint8'\n    length=img.shape[0]\n\n    if length==0:\n        # we should not have arrays with length 0\n        raise f'NO BYTES: {filename} has a length of 0'\n\n    #get the width\n    width=int(np.ceil(np.sqrt(length)))\n    height=width\n\n    #find the length so that sqrt is whole\n    new_length=np.square(width,dtype=int)\n\n    if check:print('width x height:',f'{width}x{height}')\n    if check:print('length:',length)\n    if check:print('new_length:',new_length)\n        \n    #Not all shapes work. So i am padding with 0s at the end\n    #calculate pad to be added\n    n_pad=int(new_length-length)\n    if check:print('n_pad:',n_pad)\n    \n    # pad the sequence length\n    img_padded=np.pad(img,(0,n_pad))\n    img = img_padded.reshape((width,-1))\n    \n    return img","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:01:49.795686Z","iopub.execute_input":"2023-07-10T21:01:49.796147Z","iopub.status.idle":"2023-07-10T21:01:49.808125Z","shell.execute_reply.started":"2023-07-10T21:01:49.796105Z","shell.execute_reply":"2023-07-10T21:01:49.806280Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_images(current_files:np.array,current_labels:np.array,save_folder:str,batch_size=128,check=False,resize_method='bilinear',resize_shape=[512,512]):\n#     current_files=list(train_files[train_index].copy())\n#     current_labels=list(train_labels[train_index].copy())\n    \n    print('started extracting . . .')\n    extract_data(folder=save_folder,filenames=list(map(lambda x: f'{x}.bytes',current_files)))\n    print('finished extracting\\n')\n    \n    print('started processing . . .')\n    \n    #get files in tmp folder\n    tmp_files=os.listdir(save_folder+'/train')\n    \n    train_images=[]\n    img_labels=[]\n    for file in tmp_files:\n        with open(f'{save_folder}/train/{file}','rb') as f:\n            code_bytes=str(f.read())\n            code_bytes=process_exe(code_bytes)\n#             clean_codes.append((code_bytes,file))\n            \n            # get the file class\n            filename=file.split('.')[0]\n            if check:print('filename:',filename)\n            \n            file_class=labels[labels['Id']==filename]['Class'].values[0]\n            if check:print('file_class:',file_class)\n            \n            img=code_to_image(code_bytes,filename,check)\n            img=img/normalizer#/255#normalize\n            img=np.expand_dims(img,axis=-1)\n            \n#             img=tf.image.resize(img,resize_shape,method=resize_method)\n            \n#             print(type(img))\n#             print(img)\n            #add to a list of images and another with the label.\n            train_images.append(img)\n            img_labels.append(file_class-1)\n    \n    #get the data as a numpy array\n    tmp_array=np.zeros(shape=(len(tmp_files),resize_shape[0],resize_shape[1],1))\n    for i in range(len(train_images)):\n        tmp_array[i,:,:,:]=train_images[i]\n\n    train_images=tmp_array\n    \n    dataset=tf.data.Dataset.from_tensor_slices((train_images,tf.one_hot(indices=img_labels,depth=9)))\n    dataset=dataset.batch(batch_size=batch_size)\n    \n    dataset = dataset.prefetch(tf.data.AUTOTUNE)\n    \n    delete_folder(save_folder)\n    \n    return dataset,tmp_files","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:01:49.990993Z","iopub.execute_input":"2023-07-10T21:01:49.991433Z","iopub.status.idle":"2023-07-10T21:01:50.010541Z","shell.execute_reply.started":"2023-07-10T21:01:49.991396Z","shell.execute_reply":"2023-07-10T21:01:50.008734Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file_names=labels['Id'].tolist()\n\n#I only want the byte files\nfile_names=list(map(lambda file: file+'.bytes',file_names))\n\n# aAme of temporary folder\ntemp_folder='tmp'\n\nimage_extention='jpeg'\n\nstart=0\nstep=1000# extract 10 by a time\nnumber_files=len(file_names)\n\nprocess_batch=True\n\ncheck=False\n\nfile_size={}\n\nresize_shape=(224,224)\n\n# ngram=1\n# #Get the vectorizer\n# vectorizer = CountVectorizer(ngram_range=(ngram,ngram))\n\ni=1\n\nwidth_ls=[]\nlength_ls=[]\nclass_ls=[]\nwhile(get_img):\n    print(f'started {i} run')\n    \n    end=start+step\n    \n    if end>(number_files):\n        end=number_files\n        \n    current_files=file_names[start:end]\n    \n    #extract the data to a temporary folder tmp\n    extract_data(folder=temp_folder,filenames=current_files)\n    print('finished extracting')\n    print('started processing')\n    #we will process and delete the extracted files\n\n    #get files in tmp folder\n    tmp_files=os.listdir(temp_folder+'/train')\n\n    clean_codes=[]\n    for file in tmp_files:\n        with open(f'{temp_folder}/train/{file}','rb') as f:\n            code_bytes=str(f.read())\n            code_bytes=process_exe(code_bytes)\n            \n            # get the file class\n            filename=file.split('.')[0]\n            if check:print('filename:',filename)\n            \n            file_class=labels[labels['Id']==filename]['Class'].values[0]\n            file_class_label=class_values[file_class]\n            if check:print('file_class:',file_class)\n                \n            path='vit/'\n            \n            if filename in train_files:\n                path+='train/'\n            elif  filename in test_files:\n                path+='test/'\n            else:\n                path+='val/'\n            \n            path+=f'{file_class}-{file_class_label}'\n            \n#             print('path',path+'/'+filename)\n            \n            img=code_to_image(code_bytes,filename,check)\n            length,width=img.shape\n            \n# #             img=img/normalizer#/255#normalize\n#             img=np.expand_dims(img,axis=-1)\n            \n#             img=np.repeat(img,3,-1)\n            \n            #convert to PIL image\n            img=Image.fromarray(img,'L')\n            \n            #resize using bilinear method\n            img = img.resize(resize_shape, Image.BILINEAR)\n            \n            # img = img.save(f\"{path}/{filename}.jpeg\")\n            \n            \n            length_ls.append(length)\n            width_ls.append(width)\n            class_ls.append(file_class)\n            \n            #save image\n#             try:\n#                 cv.imwrite(path,img)\n#             except:\n#                 print(path)\n#                 print(img.shape)\n                \n#             img = Image.fromarray(img,mode='L')\n#             img.save(path)\n            \n            if check:print('---------------------------------------------------')\n#             read_img=cv.imread('test.png',cv.IMREAD_GRAYSCALE)\n#             diff=np.abs(read_img-img)\n            \n#             error=np.sum(diff,axis=None)\n#             if error>0: print('sum:', error)\n            \n        \n    #delete folder with its files\n    delete_folder(temp_folder)\n    \n    if end==number_files:\n        break\n    \n    start=end\n    i+=1","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:07:19.054944Z","iopub.execute_input":"2023-07-10T21:07:19.055443Z","iopub.status.idle":"2023-07-10T21:09:38.704722Z","shell.execute_reply.started":"2023-07-10T21:07:19.055399Z","shell.execute_reply":"2023-07-10T21:09:38.702820Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file_names=labels['Id'].tolist()\n\n#I only want the byte files\nfile_names=list(map(lambda file: file+'.bytes',file_names))\n\n# aAme of temporary folder\ntemp_folder='tmp'\n\nimage_extention='jpeg'\n\nstart=0\nstep=1000# extract 10 by a time\nnumber_files=len(file_names)\n\nprocess_batch=True\n\ncheck=False\n\nfile_size={}\n\nresize_shape=(256,256)\n\n# ngram=1\n# #Get the vectorizer\n# vectorizer = CountVectorizer(ngram_range=(ngram,ngram))\n\ni=1\n\nwidth_ls=[]\nlength_ls=[]\nclass_ls=[]\nwhile(get_img):\n    print(f'started {i} run')\n    \n    end=start+step\n    \n    if end>(number_files):\n        end=number_files\n        \n    current_files=file_names[start:end]\n    \n    #extract the data to a temporary folder tmp\n    extract_data(folder=temp_folder,filenames=current_files)\n    print('finished extracting')\n    print('started processing')\n    #we will process and delete the extracted files\n\n    #get files in tmp folder\n    tmp_files=os.listdir(temp_folder+'/train')\n\n    clean_codes=[]\n    for file in tmp_files:\n        with open(f'{temp_folder}/train/{file}','rb') as f:\n            code_bytes=str(f.read())\n            code_bytes=process_exe(code_bytes)\n            \n            # get the file class\n            filename=file.split('.')[0]\n            if check:print('filename:',filename)\n            \n            file_class=labels[labels['Id']==filename]['Class'].values[0]\n            file_class_label=class_values[file_class]\n            if check:print('file_class:',file_class)\n                \n            path='cnn/'\n            \n            if filename in train_files:\n                path+='train/'\n            elif  filename in test_files:\n                path+='test/'\n            else:\n                path+='val/'\n            \n            path+=f'{file_class}-{file_class_label}'\n            \n#             print('path',path+'/'+filename)\n            \n            img=code_to_image(code_bytes,filename,check)\n            length,width=img.shape\n            \n# #             img=img/normalizer#/255#normalize\n#             img=np.expand_dims(img,axis=-1)\n            \n#             img=np.repeat(img,3,-1)\n            \n            #convert to PIL image\n            img=Image.fromarray(img,'L')\n            \n            #resize using bilinear method\n            img = img.resize(resize_shape, Image.BILINEAR)\n            \n            img = img.save(f\"{path}/{filename}.jpeg\")\n            \n            \n            length_ls.append(length)\n            width_ls.append(width)\n            class_ls.append(file_class)\n            \n            #save image\n#             try:\n#                 cv.imwrite(path,img)\n#             except:\n#                 print(path)\n#                 print(img.shape)\n                \n#             img = Image.fromarray(img,mode='L')\n#             img.save(path)\n            \n            if check:print('---------------------------------------------------')\n#             read_img=cv.imread('test.png',cv.IMREAD_GRAYSCALE)\n#             diff=np.abs(read_img-img)\n            \n#             error=np.sum(diff,axis=None)\n#             if error>0: print('sum:', error)\n            \n        \n    #delete folder with its files\n    delete_folder(temp_folder)\n    \n    if end==number_files:\n        break\n    \n    start=end\n    i+=1","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:41.188609Z","iopub.status.idle":"2023-07-10T21:00:41.189108Z","shell.execute_reply.started":"2023-07-10T21:00:41.188868Z","shell.execute_reply":"2023-07-10T21:00:41.188894Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# file_names=labels['Id'].tolist()\n\n# #I only want the byte files\n# file_names=list(map(lambda file: file+'.bytes',file_names))\n\n# # aAme of temporary folder\n# temp_folder='tmp'\n\n# image_extention='jpeg'\n\n# start=0\n# step=2000# extract 10 by a time\n# number_files=len(file_names)\n\n# process_batch=True\n\n# check=False\n\n# file_size={}\n\n# ngram=1\n# #Get the vectorizer\n# vectorizer = CountVectorizer(ngram_range=(ngram,ngram))\n\n# i=1\n\n# width_ls=[]\n# length_ls=[]\n# class_ls=[]\n# while(get_img):\n#     print(f'started {i} run')\n    \n#     end=start+step\n    \n#     if end>(number_files):\n#         end=number_files\n        \n#     current_files=file_names[start:end]\n    \n#     #extract the data to a temporary folder tmp\n#     extract_data(folder=temp_folder,filenames=current_files)\n#     print('finished extracting')\n    \n#     current_tot_size=0\n#     for this_file in os.listdir(temp_folder+'/train'):\n#         #get the size in MBytes\n#         size=os.path.getsize(f'{temp_folder}/train/{this_file}')/1e6\n#         current_tot_size+=size\n        \n#         #save\n#         file_size[this_file]=size\n    \n#     #if size greater than 10GB, then enter the loop\n#     if process_batch or current_tot_size>10000 or end==number_files:\n#         print('started processing')\n#         #we will process and delete the extracted files\n        \n#         #get files in tmp folder\n#         tmp_files=os.listdir(temp_folder+'/train')\n        \n#         clean_codes=[]\n#         for file in tmp_files:\n#             with open(f'{temp_folder}/train/{file}','rb') as f:\n#                 code_bytes=str(f.read())\n#                 code_bytes=process_exe(code_bytes)\n#                 clean_codes.append((code_bytes,file))\n        \n        \n#         #convert to images\n#         for (code,file) in clean_codes:\n#             # get the file class\n#             filename=file.split('.')[0]\n#             if check:print('filename:',filename)\n            \n#             file_class=labels[labels['Id']==filename]['Class'].values[0]\n#             if check:print('file_class:',file_class)\n            \n#             path=f'img/{file_class}/{filename}.{image_extention}'\n#             if check:print('path:',path)\n            \n#             img= np.frombuffer(bytes.fromhex(code), dtype='uint8')\n#             length=img.shape[0]\n            \n#             if length==0:\n#                 continue\n                \n#             #get the width\n#             width=int(np.ceil(np.sqrt(length)))\n#             height=width\n            \n#             #find the length so that sqrt is whole\n#             new_length=np.square(width,dtype=int)\n            \n#             if check:print('width x height:',f'{width}x{height}')\n#             if check:print('length:',length)\n#             if check:print('new_length:',new_length)\n#             #Not all shapes work. So i am padding with 0s at the end\n#             n_pad=int(new_length-length)\n#             if check:print('n_pad:',n_pad)\n#             img_padded=np.pad(img,(0,n_pad))\n#             img = img_padded.reshape((width,-1))\n            \n#             length_ls.append(length)\n#             width_ls.append(width)\n#             class_ls.append(file_class)\n            \n#             #save image\n# #             try:\n# #                 cv.imwrite(path,img)\n# #             except:\n# #                 print(path)\n# #                 print(img.shape)\n                \n# #             img = Image.fromarray(img,mode='L')\n# #             img.save(path)\n            \n#             if check:print('---------------------------------------------------')\n# #             read_img=cv.imread('test.png',cv.IMREAD_GRAYSCALE)\n# #             diff=np.abs(read_img-img)\n            \n# #             error=np.sum(diff,axis=None)\n# #             if error>0: print('sum:', error)\n            \n        \n#         #delete folder with its files\n#         delete_folder(temp_folder)\n    \n#     if end==number_files:\n#         break\n    \n#     start=end\n#     i+=1","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:41.191377Z","iopub.status.idle":"2023-07-10T21:00:41.192371Z","shell.execute_reply.started":"2023-07-10T21:00:41.192028Z","shell.execute_reply":"2023-07-10T21:00:41.192066Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# length_ls=np.random.randint(0,20,40)\n# width_ls=np.random.randint(0,10,40)\n# class_ls=np.random.randint(1,10,40)\ndata_df=pd.DataFrame(data={'length':length_ls, 'width':width_ls,'class':class_ls})","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:41.193923Z","iopub.status.idle":"2023-07-10T21:00:41.195193Z","shell.execute_reply.started":"2023-07-10T21:00:41.194921Z","shell.execute_reply":"2023-07-10T21:00:41.194950Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.histplot(data=data_df,x='length',kde=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:41.196631Z","iopub.status.idle":"2023-07-10T21:00:41.198126Z","shell.execute_reply.started":"2023-07-10T21:00:41.197850Z","shell.execute_reply":"2023-07-10T21:00:41.197880Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# sns.histplot(data=data_df,x='length',kde=True)\nfig, axs = plt.subplots(3, 3, figsize=(10,10))\nfor class_label,group_df in data_df.copy().groupby('class'):\n    label=int(class_label)\n    new_col_name=f'{class_values[label]} ({label}) length'\n    group_df=group_df.rename(columns={'length':new_col_name})\n    sns.histplot(data=group_df,x=new_col_name,kde=True,ax=axs[(label-1)//3,(label-1)%3])","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:41.199582Z","iopub.status.idle":"2023-07-10T21:00:41.200407Z","shell.execute_reply.started":"2023-07-10T21:00:41.200158Z","shell.execute_reply":"2023-07-10T21:00:41.200183Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# sns.histplot(data=data_df,x='length',hue='class',multiple='stack',kde=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:41.201766Z","iopub.status.idle":"2023-07-10T21:00:41.202621Z","shell.execute_reply.started":"2023-07-10T21:00:41.202340Z","shell.execute_reply":"2023-07-10T21:00:41.202397Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.histplot(data=data_df,x='width',kde=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:41.203979Z","iopub.status.idle":"2023-07-10T21:00:41.204953Z","shell.execute_reply.started":"2023-07-10T21:00:41.204541Z","shell.execute_reply":"2023-07-10T21:00:41.204568Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# sns.histplot(data=data_df,x='width',hue='class',multiple='stack',kde=True)\n\nfig, axs = plt.subplots(3, 3, figsize=(15,15))\nfor class_label,group_df in data_df.copy().groupby('class'):\n    label=int(class_label)\n    new_col_name=f'{class_values[label]} ({label}) width'\n    group_df=group_df.rename(columns={'width':new_col_name})\n    sns.histplot(data=group_df,x=new_col_name,kde=True,ax=axs[(label-1)//3,(label-1)%3])","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:41.206477Z","iopub.status.idle":"2023-07-10T21:00:41.207252Z","shell.execute_reply.started":"2023-07-10T21:00:41.207002Z","shell.execute_reply":"2023-07-10T21:00:41.207030Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_df.to_csv('data_df.csv')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h2>Now for test data</h2>","metadata":{}},{"cell_type":"code","source":"# file_names=labels['Id'].tolist()\n\n# #I only want the byte files\n# file_names=list(map(lambda file: file+'.bytes',file_names))\n\n# # Name of temporary folder\n# temp_folder='tmp'\n\n# start=0\n# step=2000# extract 2000 by a time\n# number_files=len(file_names)\n\n# process_batch=True\n\n# count_test_df=pd.DataFrame(columns=['filename','size(MB)']+vocab_ls)\n\n# file_size={}\n\n# #Get the vectorizer\n# vectorizer = CountVectorizer(ngram_range=(1,1),vocabulary=vocab_dic)\n# print(len(vectorizer.vocabulary))\n\n# i=1\n\n# while(get_feature_dataset):\n#     print(f'started {i} run')\n    \n#     end=start+step\n    \n#     if end>(number_files):\n#         end=number_files\n    \n#     print(f'start:{start}, end:{end}')\n        \n#     current_files=file_names[start:end]\n    \n#     #extract the data to a temporary folder tmp\n#     extract_data(folder=temp_folder,filenames=current_files)\n#     print('finished extracting')\n    \n#     current_tot_size=0\n#     for this_file in os.listdir(temp_folder+'/train'):\n#         #get the size in MBytes\n#         size=os.path.getsize(f'{temp_folder}/train/{this_file}')/1e6\n#         current_tot_size+=size\n        \n#         #save\n#         file_size[this_file]=size\n    \n#     #if size greater than 10GB, then enter the loop\n#     if process_batch or current_tot_size>10000 or end==number_files:\n#         print('started processing')\n#         #we will process and delete the extracted files\n        \n#         #get files in tmp folder\n#         tmp_files=os.listdir(temp_folder+'/train')\n#         print(len(tmp_files))\n        \n#         cur_index=len(count_test_df)\n#         clean_codes=[]\n#         for code_index,file in enumerate(tmp_files):\n#             with open(f'{temp_folder}/train/{file}','rb') as f:\n#                 code_bytes=str(f.read())\n        \n#                 #here we can get the count\n#                 counts=count_vec.transform([process_exe(code_bytes)]).toarray()\n                \n#                 #add it to the dataframe\n#                 count_test_df.loc[cur_index+code_index,['filename','size(MB)']]=file,file_size[file]#filename,file_size[filename]\n#                 count_test_df.loc[cur_index+code_index,vocab_ls]=counts#[code_index]\n        \n#         #delete folder with its files\n#         delete_folder(temp_folder)\n    \n#     if end==number_files:\n#         break\n    \n#     start=end\n#     i+=1","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:41.209115Z","iopub.status.idle":"2023-07-10T21:00:41.209889Z","shell.execute_reply.started":"2023-07-10T21:00:41.209634Z","shell.execute_reply":"2023-07-10T21:00:41.209661Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#             print(f\"  Train: size={len(train_files)}, examples={train_files}\\n\")\n#             print(f\"  Test:  size={len(val_files)}, examples={val_files}\\n\")\n#             print(f\"  Test:  size={len(test_files)}, examples={test_files}\\n\")","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:41.211179Z","iopub.status.idle":"2023-07-10T21:00:41.211928Z","shell.execute_reply.started":"2023-07-10T21:00:41.211676Z","shell.execute_reply":"2023-07-10T21:00:41.211702Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"directory='BoW'\nif not os.path.exists(directory): os.mkdir(directory)\ncount_df[count_df['filename'].isin(list(map(lambda x: x+'.bytes',train_files)))].to_csv(f'{directory}/train.csv',index=False)\ncount_df[count_df['filename'].isin(list(map(lambda x: x+'.bytes',val_files)))].to_csv(f'{directory}/val.csv',index=False)\ncount_df[count_df['filename'].isin(list(map(lambda x: x+'.bytes',test_files)))].to_csv(f'{directory}/test.csv',index=False)\ncount_df.to_csv(f'{directory}/file_counts.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:41.213230Z","iopub.status.idle":"2023-07-10T21:00:41.213981Z","shell.execute_reply.started":"2023-07-10T21:00:41.213719Z","shell.execute_reply":"2023-07-10T21:00:41.213745Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"directory='tfidf'\nif not os.path.exists(directory): os.mkdir(directory)\ntfidf_trans_df[tfidf_trans_df['filename'].isin(list(map(lambda x: x+'.bytes',train_files)))].to_csv(f'{directory}/train.csv',index=False)\ntfidf_trans_df[tfidf_trans_df['filename'].isin(list(map(lambda x: x+'.bytes',val_files)))].to_csv(f'{directory}/val.csv',index=False)\ntfidf_trans_df[tfidf_trans_df['filename'].isin(list(map(lambda x: x+'.bytes',test_files)))].to_csv(f'{directory}/test.csv',index=False)\ntfidf_trans_df.to_csv(f'{directory}/file_tf_idf.csv',index=False)\ntfidf_trans_df.to_csv('file_tf_idf.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:41.215262Z","iopub.status.idle":"2023-07-10T21:00:41.216011Z","shell.execute_reply.started":"2023-07-10T21:00:41.215748Z","shell.execute_reply":"2023-07-10T21:00:41.215774Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_rounds=1","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:41.217305Z","iopub.status.idle":"2023-07-10T21:00:41.218049Z","shell.execute_reply.started":"2023-07-10T21:00:41.217804Z","shell.execute_reply":"2023-07-10T21:00:41.217830Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"i can use vectorizer from sklearn but there is also a vectorizer from tensorflow(TextVectorization).\ntensorflow uses list while sklearn uses a dic. Hence, i will create both.","metadata":{}},{"cell_type":"code","source":"labels.groupby('Class').count()","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:41.219310Z","iopub.status.idle":"2023-07-10T21:00:41.220080Z","shell.execute_reply.started":"2023-07-10T21:00:41.219828Z","shell.execute_reply":"2023-07-10T21:00:41.219855Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels.groupby('Class').count()*100/len(labels)","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:41.221360Z","iopub.status.idle":"2023-07-10T21:00:41.222121Z","shell.execute_reply.started":"2023-07-10T21:00:41.221866Z","shell.execute_reply":"2023-07-10T21:00:41.221893Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train dataset:\n#     dataset1 dataset2 ....\n#     RF1      RF2      ....\n    \n    \n#     sample1\n    \n#     output1 output2 ....\n    \n#     final results = average output1 output2 ","metadata":{"execution":{"iopub.status.busy":"2023-07-10T21:00:41.223424Z","iopub.status.idle":"2023-07-10T21:00:41.224176Z","shell.execute_reply.started":"2023-07-10T21:00:41.223926Z","shell.execute_reply":"2023-07-10T21:00:41.223953Z"},"trusted":true},"outputs":[],"execution_count":null}]}