{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1><center>Machine Learning Project Phase 1</center></h1>\n<h2><center>Malware Classification using machine learning</center></h2>\n<h3>Submitted by:</h3>\n<pre>Chandrakiran J            AMENU4CSE20017\nGeo Jolly Cheeramvelil    AMENU4CSE20025\nNoel Siby                 AMENU4CSE20051\nPankaj P                  AMENU4CSE20053\nSreenadh Venugopal        AMENU4CSE20068\n</pre>\n    \n","metadata":{}},{"cell_type":"code","source":"import os\n\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\n\n%matplotlib inline\nplt.style.use('ggplot')\nimport datetime\nimport gc\nimport time\n\nimport lightgbm as lgb\nimport plotly.offline as py\nfrom sklearn import metrics\nfrom sklearn.linear_model import LogisticRegression, LogisticRegressionCV\nfrom sklearn.metrics import mean_squared_error, roc_auc_score\nfrom sklearn.model_selection import KFold, StratifiedKFold\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\n\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\nimport plotly.tools as tls\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-10T14:02:00.127781Z","iopub.execute_input":"2022-12-10T14:02:00.128930Z","iopub.status.idle":"2022-12-10T14:02:00.145274Z","shell.execute_reply.started":"2022-12-10T14:02:00.128880Z","shell.execute_reply":"2022-12-10T14:02:00.143784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Loading and Data Preprocessing\n\n- load objects as categories, object is fine as well.\n- Binary values are switched to int8\n- Binary values with missing values are switched to float16 (int does not understand NaN), it is possible to use category here as well.\n- 64 bits encoding are all switched to 32, or 16 if possible\n\n<div style=\"color:blue;font-size:20px;\">[Notes]</div>\nThe data is quite big here and all of it cannot be loaded at once with a simple read_csv call.\n\nA solution is to specify types, to gain memory (for example switching from float64 to float32)","metadata":{}},{"cell_type":"code","source":"dtypes = {\n        'MachineIdentifier':                                    'category',\n        'ProductName':                                          'category',\n        'EngineVersion':                                        'category',\n        'AppVersion':                                           'category',\n        'AvSigVersion':                                         'category',\n        'IsBeta':                                               'int8',\n        'RtpStateBitfield':                                     'float16',\n        'IsSxsPassiveMode':                                     'int8',\n        'DefaultBrowsersIdentifier':                            'float16',\n        'AVProductStatesIdentifier':                            'float32',\n        'AVProductsInstalled':                                  'float16',\n        'AVProductsEnabled':                                    'float16',\n        'HasTpm':                                               'int8',\n        'CountryIdentifier':                                    'int16',\n        'CityIdentifier':                                       'float32',\n        'OrganizationIdentifier':                               'float16',\n        'GeoNameIdentifier':                                    'float16',\n        'LocaleEnglishNameIdentifier':                          'int8',\n        'Platform':                                             'category',\n        'Processor':                                            'category',\n        'OsVer':                                                'category',\n        'OsBuild':                                              'int16',\n        'OsSuite':                                              'int16',\n        'OsPlatformSubRelease':                                 'category',\n        'OsBuildLab':                                           'category',\n        'SkuEdition':                                           'category',\n        'IsProtected':                                          'float16',\n        'AutoSampleOptIn':                                      'int8',\n        'PuaMode':                                              'category',\n        'SMode':                                                'float16',\n        'IeVerIdentifier':                                      'float16',\n        'SmartScreen':                                          'category',\n        'Firewall':                                             'float16',\n        'UacLuaenable':                                         'float32',\n        'Census_MDC2FormFactor':                                'category',\n        'Census_DeviceFamily':                                  'category',\n        'Census_OEMNameIdentifier':                             'float16',\n        'Census_OEMModelIdentifier':                            'float32',\n        'Census_ProcessorCoreCount':                            'float16',\n        'Census_ProcessorManufacturerIdentifier':               'float16',\n        'Census_ProcessorModelIdentifier':                      'float16',\n        'Census_ProcessorClass':                                'category',\n        'Census_PrimaryDiskTotalCapacity':                      'float32',\n        'Census_PrimaryDiskTypeName':                           'category',\n        'Census_SystemVolumeTotalCapacity':                     'float32',\n        'Census_HasOpticalDiskDrive':                           'int8',\n        'Census_TotalPhysicalRAM':                              'float32',\n        'Census_ChassisTypeName':                               'category',\n        'Census_InternalPrimaryDiagonalDisplaySizeInInches':    'float16',\n        'Census_InternalPrimaryDisplayResolutionHorizontal':    'float16',\n        'Census_InternalPrimaryDisplayResolutionVertical':      'float16',\n        'Census_PowerPlatformRoleName':                         'category',\n        'Census_InternalBatteryType':                           'category',\n        'Census_InternalBatteryNumberOfCharges':                'float32',\n        'Census_OSVersion':                                     'category',\n        'Census_OSArchitecture':                                'category',\n        'Census_OSBranch':                                      'category',\n        'Census_OSBuildNumber':                                 'int16',\n        'Census_OSBuildRevision':                               'int32',\n        'Census_OSEdition':                                     'category',\n        'Census_OSSkuName':                                     'category',\n        'Census_OSInstallTypeName':                             'category',\n        'Census_OSInstallLanguageIdentifier':                   'float16',\n        'Census_OSUILocaleIdentifier':                          'int16',\n        'Census_OSWUAutoUpdateOptionsName':                     'category',\n        'Census_IsPortableOperatingSystem':                     'int8',\n        'Census_GenuineStateName':                              'category',\n        'Census_ActivationChannel':                             'category',\n        'Census_IsFlightingInternal':                           'float16',\n        'Census_IsFlightsDisabled':                             'float16',\n        'Census_FlightRing':                                    'category',\n        'Census_ThresholdOptIn':                                'float16',\n        'Census_FirmwareManufacturerIdentifier':                'float16',\n        'Census_FirmwareVersionIdentifier':                     'float32',\n        'Census_IsSecureBootEnabled':                           'int8',\n        'Census_IsWIMBootEnabled':                              'float16',\n        'Census_IsVirtualDevice':                               'float16',\n        'Census_IsTouchEnabled':                                'int8',\n        'Census_IsPenCapable':                                  'int8',\n        'Census_IsAlwaysOnAlwaysConnectedCapable':              'float16',\n        'Wdft_IsGamer':                                         'float16',\n        'Wdft_RegionIdentifier':                                'float16',\n        'HasDetections':                                        'int8'\n        }","metadata":{"execution":{"iopub.status.busy":"2022-12-10T12:57:02.488586Z","iopub.execute_input":"2022-12-10T12:57:02.488971Z","iopub.status.idle":"2022-12-10T12:57:02.508358Z","shell.execute_reply.started":"2022-12-10T12:57:02.488924Z","shell.execute_reply":"2022-12-10T12:57:02.506996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage(deep=True).sum() / 1024**2    \n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)    \n    end_mem = df.memory_usage(deep=True).sum() / 1024**2\n    if verbose: \n        print(\n            'Mem. usage decreased to {:5.2f} Mb ({:.1f}% reduction)'.format(\n                end_mem, 100 * (start_mem - end_mem) / start_mem\n            )\n        )\n    return df\n\nnumerics = ['int8', 'int16', 'int32', 'int64', 'float16', 'float32', 'float64']\nnumerical_columns = [c for c,v in dtypes.items() if v in numerics]\ncategorical_columns = [c for c,v in dtypes.items() if v not in numerics]","metadata":{"execution":{"iopub.status.busy":"2022-12-10T12:57:02.513255Z","iopub.execute_input":"2022-12-10T12:57:02.513843Z","iopub.status.idle":"2022-12-10T12:57:02.538354Z","shell.execute_reply.started":"2022-12-10T12:57:02.513788Z","shell.execute_reply":"2022-12-10T12:57:02.536048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Read training data**","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv(\n    '../input/microsoft-malware-prediction/train.csv', \n    dtype=dtypes\n)","metadata":{"execution":{"iopub.status.busy":"2022-12-10T12:57:02.541377Z","iopub.execute_input":"2022-12-10T12:57:02.541903Z","iopub.status.idle":"2022-12-10T13:00:39.435475Z","shell.execute_reply.started":"2022-12-10T12:57:02.541850Z","shell.execute_reply":"2022-12-10T13:00:39.432812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reducing memory usage\ntrain_data = reduce_mem_usage(train_data)","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:00:39.438232Z","iopub.execute_input":"2022-12-10T13:00:39.439401Z","iopub.status.idle":"2022-12-10T13:00:57.619609Z","shell.execute_reply.started":"2022-12-10T13:00:39.439346Z","shell.execute_reply":"2022-12-10T13:00:57.617558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Analysing the training data**","metadata":{}},{"cell_type":"code","source":"stats = []\nfor col in train_data.columns:\n    stats.append((\n        col, train_data[col].nunique(), \n        train_data[col].isnull().sum() * 100 / train_data.shape[0], \n        train_data[col].value_counts(\n            normalize=True, dropna=False\n        ).values[0] * 100, \n        train_data[col].dtype\n    ))\n    \nstats_df = pd.DataFrame(stats, columns=['Feature', 'Unique_values', 'Percentage of missing values', 'Percentage of values in the biggest category', 'type'])\nstats_df.sort_values('Percentage of missing values', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:00:57.622639Z","iopub.execute_input":"2022-12-10T13:00:57.623263Z","iopub.status.idle":"2022-12-10T13:01:19.157790Z","shell.execute_reply.started":"2022-12-10T13:00:57.623205Z","shell.execute_reply":"2022-12-10T13:01:19.156483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Top things to note from this summary\n - `PuaMode` and `Census_ProcessorClass` have 99%+ missing values, which means that these columns are useless and should be dropped(makes no sense in filling data).\n - `DefaultBrowsersIdentifier` column 95% values belong to one category which makes this column useless and thus it should be dropped.\n -  There are 26 columns in total in which one category contains 90% values. These should be removed since it cause a high imbalance in the dataset.\n - All columns except `Census_SystemVolumeTotalCapacity` are categorial. Other than that there are 3 columns with most the values missing. It should also be dropped,","metadata":{}},{"cell_type":"code","source":"cleaned_cols = list(train_data.columns)\nfor col in train_data.columns:\n    rate = train_data[col].value_counts(normalize=True, dropna=False).values[0]\n    if rate > 0.9:\n        cleaned_cols.remove(col)\n\ntrain_data = train_data[cleaned_cols]","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:01:19.162424Z","iopub.execute_input":"2022-12-10T13:01:19.162891Z","iopub.status.idle":"2022-12-10T13:01:31.091525Z","shell.execute_reply.started":"2022-12-10T13:01:19.162850Z","shell.execute_reply":"2022-12-10T13:01:31.089969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Visualization","metadata":{}},{"cell_type":"code","source":"def plot_categorical_feature(col, only_bars=False, top_n=10, by_touch=False):\n    top_n = top_n if train_data[col].nunique() > top_n else train_data[col].nunique()\n    print(f\"{col} has {train_data[col].nunique()} unique values and type: {train_data[col].dtype}.\")\n    print(train_data[col].value_counts(normalize=True, dropna=False).head())\n    if not by_touch:\n        if not only_bars:\n            df = train_data.groupby([col]).agg({'HasDetections': ['count', 'mean']})\n            df = df.sort_values(('HasDetections', 'count'), ascending=False).head(top_n).sort_index()\n            data = [go.Bar(x=df.index, y=df['HasDetections']['count'].values, name='counts'),\n                    go.Scatter(x=df.index, y=df['HasDetections']['mean'], name='Detections rate', yaxis='y2')]\n\n            layout = go.Layout(dict(title = f\"Counts of {col} by top-{top_n} categories and mean target value\",\n                                xaxis = dict(title = f'{col}',\n                                             showgrid=False,\n                                             zeroline=False,\n                                             showline=False,),\n                                yaxis = dict(title = 'Counts',\n                                             showgrid=False,\n                                             zeroline=False,\n                                             showline=False,),\n                                yaxis2=dict(title='Detections rate', overlaying='y', side='right')),\n                           legend=dict(orientation=\"v\"))\n\n        else:\n            top_cat = list(train_data[col].value_counts(dropna=False).index[:top_n])\n            df0 = train_data.loc[(train_data[col].isin(top_cat)) & (train_data['HasDetections'] == 1), col].value_counts().head(10).sort_index()\n            df1 = train_data.loc[(train_data[col].isin(top_cat)) & (train_data['HasDetections'] == 0), col].value_counts().head(10).sort_index()\n            data = [go.Bar(x=df0.index, y=df0.values, name='Has Detections'),\n                    go.Bar(x=df1.index, y=df1.values, name='No Detections')]\n\n            layout = go.Layout(dict(title = f\"Counts of {col} by top-{top_n} categories\",\n                                xaxis = dict(title = f'{col}',\n                                             showgrid=False,\n                                             zeroline=False,\n                                             showline=False,),\n                                yaxis = dict(title = 'Counts',\n                                             showgrid=False,\n                                             zeroline=False,\n                                             showline=False,),\n                                ),\n                           legend=dict(orientation=\"v\"), barmode='group')\n        py.iplot(dict(data=data, layout=layout))\n        \n    else:\n        top_n = 10\n        top_cat = list(train_data[col].value_counts(dropna=False).index[:top_n])\n        df = train_data.loc[train_data[col].isin(top_cat)]\n\n        df1 = train_data.loc[train_data['Census_IsTouchEnabled'] == 1]\n        df0 = train_data.loc[train_data['Census_IsTouchEnabled'] == 0]\n\n        df0_ = df0.groupby([col]).agg({'HasDetections': ['count', 'mean']})\n        df0_ = df0_.sort_values(('HasDetections', 'count'), ascending=False).head(top_n).sort_index()\n        df1_ = df1.groupby([col]).agg({'HasDetections': ['count', 'mean']})\n        df1_ = df1_.sort_values(('HasDetections', 'count'), ascending=False).head(top_n).sort_index()\n        data1 = [go.Bar(x=df0_.index, y=df0_['HasDetections']['count'].values, name='Nontouch device counts'),\n                go.Scatter(x=df0_.index, y=df0_['HasDetections']['mean'], name='Detections rate for nontouch devices', yaxis='y2')]\n        data2 = [go.Bar(x=df1_.index, y=df1_['HasDetections']['count'].values, name='Touch device counts'),\n                go.Scatter(x=df1_.index, y=df1_['HasDetections']['mean'], name='Detections rate for touch devices', yaxis='y2')]\n\n        layout = go.Layout(dict(title = f\"Counts of {col} by top-{top_n} categories for nontouch devices\",\n                            xaxis = dict(title = f'{col}',\n                                         showgrid=False,\n                                         zeroline=False,\n                                         showline=False,\n                                         type='category'),\n                            yaxis = dict(title = 'Counts',\n                                         showgrid=False,\n                                         zeroline=False,\n                                         showline=False,),\n                                    yaxis2=dict(title='Detections rate', overlaying='y', side='right'),\n                            ),\n                       legend=dict(orientation=\"v\"), barmode='group')\n\n        py.iplot(dict(data=data1, layout=layout))\n        layout['title'] = f\"Counts of {col} by top-{top_n} categories for touch devices\"\n        py.iplot(dict(data=data2, layout=layout))","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:01:31.094009Z","iopub.execute_input":"2022-12-10T13:01:31.094428Z","iopub.status.idle":"2022-12-10T13:01:31.125364Z","shell.execute_reply.started":"2022-12-10T13:01:31.094390Z","shell.execute_reply":"2022-12-10T13:01:31.124005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1. **Target**","metadata":{}},{"cell_type":"code","source":"plot_categorical_feature('HasDetections', True)","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:01:31.127688Z","iopub.execute_input":"2022-12-10T13:01:31.128214Z","iopub.status.idle":"2022-12-10T13:01:32.720458Z","shell.execute_reply.started":"2022-12-10T13:01:31.128170Z","shell.execute_reply":"2022-12-10T13:01:32.719474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color:blue\">[Summary]</span> Figure shows the data is balanced","metadata":{}},{"cell_type":"markdown","source":"2. **Conclusions from `Census_IsTouchEnabled`**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(5,5))\nplot_categorical_feature('Census_IsTouchEnabled', True)","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:01:32.724334Z","iopub.execute_input":"2022-12-10T13:01:32.725292Z","iopub.status.idle":"2022-12-10T13:01:33.392193Z","shell.execute_reply.started":"2022-12-10T13:01:32.725250Z","shell.execute_reply":"2022-12-10T13:01:33.390790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color:blue\">[Summary]</span> The rate of infections is lower for touch devices, but not by much. For instance, rate of infection on non-touch devices like computer is more compared to touch devices like phone.","metadata":{}},{"cell_type":"markdown","source":"3. **Conclusions from `AvSigVersion`**","metadata":{}},{"cell_type":"code","source":"plot_categorical_feature('AvSigVersion')","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:01:33.393533Z","iopub.execute_input":"2022-12-10T13:01:33.393910Z","iopub.status.idle":"2022-12-10T13:01:33.952294Z","shell.execute_reply.started":"2022-12-10T13:01:33.393876Z","shell.execute_reply":"2022-12-10T13:01:33.951043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"4. **Conclusions from `AvProductsInstalled`**","metadata":{}},{"cell_type":"code","source":"plot_categorical_feature('AVProductsInstalled')","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:01:33.954355Z","iopub.execute_input":"2022-12-10T13:01:33.955029Z","iopub.status.idle":"2022-12-10T13:01:34.965826Z","shell.execute_reply.started":"2022-12-10T13:01:33.954971Z","shell.execute_reply":"2022-12-10T13:01:34.964456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color:blue\">[Summary]</span> If a computer has an antivirus, it is less likely to be infected. But having two antiviruses has an opposite effect. We can draw one conclusion that maybe those who install 2 antiviruses tend to have less experience with working on PC","metadata":{}},{"cell_type":"markdown","source":"5. **Conclusions from `OsBuildLab`**","metadata":{}},{"cell_type":"code","source":"plot_categorical_feature('OsBuildLab', True)","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:01:34.967714Z","iopub.execute_input":"2022-12-10T13:01:34.968150Z","iopub.status.idle":"2022-12-10T13:01:35.670770Z","shell.execute_reply.started":"2022-12-10T13:01:34.968109Z","shell.execute_reply":"2022-12-10T13:01:35.669441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color:blue\">[Summary]</span> We can see that most used version is `17134.1.amd64fre.rs4_release.180410-1804` and has a higher rate of detection.","metadata":{}},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"train_data['AVProductStatesIdentifier'] = train_data['AVProductStatesIdentifier'].astype('category')\ntrain_data['AVProductsInstalled'] = train_data['AVProductsInstalled'].astype('category')","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:01:35.672897Z","iopub.execute_input":"2022-12-10T13:01:35.673475Z","iopub.status.idle":"2022-12-10T13:01:36.501539Z","shell.execute_reply.started":"2022-12-10T13:01:35.673419Z","shell.execute_reply":"2022-12-10T13:01:36.500203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['OsBuildLab'] = train_data['OsBuildLab'].cat.add_categories(['0.0.0.0.0-0'])\ntrain_data['OsBuildLab'] = train_data['OsBuildLab'].fillna('0.0.0.0.0-0')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineering(df):\n    df['EngineVersion_2'] = df['EngineVersion'].apply(lambda x: x.split('.')[2]).astype('category')\n    df['EngineVersion_3'] = df['EngineVersion'].apply(lambda x: x.split('.')[3]).astype('category')\n    df['AppVersion_1'] = df['AppVersion'].apply(lambda x: x.split('.')[1]).astype('category')\n    df['AppVersion_2'] = df['AppVersion'].apply(lambda x: x.split('.')[2]).astype('category')\n    df['AppVersion_3'] = df['AppVersion'].apply(lambda x: x.split('.')[3]).astype('category')\n    df['AvSigVersion_0'] = df['AvSigVersion'].apply(lambda x: x.split('.')[0]).astype('category')\n    df['AvSigVersion_1'] = df['AvSigVersion'].apply(lambda x: x.split('.')[1]).astype('category')\n    df['AvSigVersion_2'] = df['AvSigVersion'].apply(lambda x: x.split('.')[2]).astype('category')\n    df['OsBuildLab_0'] = df['OsBuildLab'].apply(lambda x: x.split('.')[0]).astype('category')\n    df['OsBuildLab_1'] = df['OsBuildLab'].apply(lambda x: x.split('.')[1]).astype('category')\n    df['OsBuildLab_2'] = df['OsBuildLab'].apply(lambda x: x.split('.')[2]).astype('category')\n    df['OsBuildLab_3'] = df['OsBuildLab'].apply(lambda x: x.split('.')[3]).astype('category')\n    df['Census_OSVersion_0'] = df['Census_OSVersion'].apply(lambda x: x.split('.')[0]).astype('category')\n    df['Census_OSVersion_1'] = df['Census_OSVersion'].apply(lambda x: x.split('.')[1]).astype('category')\n    df['Census_OSVersion_2'] = df['Census_OSVersion'].apply(lambda x: x.split('.')[2]).astype('category')\n    df['Census_OSVersion_3'] = df['Census_OSVersion'].apply(lambda x: x.split('.')[3]).astype('category')\n\n    df['aspect_ratio'] = df[\n        'Census_InternalPrimaryDisplayResolutionHorizontal'\n    ]/ df['Census_InternalPrimaryDisplayResolutionVertical']\n    df['monitor_dims'] = df[\n        'Census_InternalPrimaryDisplayResolutionHorizontal'].astype(str) + '*' + df[\n        'Census_InternalPrimaryDisplayResolutionVertical'].astype('str')\n    \n    df['monitor_dims'] = df['monitor_dims'].astype('category')\n    df['Census_IsFlightingInternal'] = df['Census_IsFlightingInternal'].fillna(1)\n    df['Census_ThresholdOptIn'] = df['Census_ThresholdOptIn'].fillna(1)\n    df['Census_IsWIMBootEnabled'] = df['Census_IsWIMBootEnabled'].fillna(1)\n    df['Wdft_IsGamer'] = df['Wdft_IsGamer'].fillna(0)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:02:58.394613Z","iopub.execute_input":"2022-12-10T13:02:58.395142Z","iopub.status.idle":"2022-12-10T13:02:58.415590Z","shell.execute_reply.started":"2022-12-10T13:02:58.395107Z","shell.execute_reply":"2022-12-10T13:02:58.414138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_train = feature_engineering(train_data)","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:03:01.345291Z","iopub.execute_input":"2022-12-10T13:03:01.345783Z","iopub.status.idle":"2022-12-10T13:03:30.699526Z","shell.execute_reply.started":"2022-12-10T13:03:01.345741Z","shell.execute_reply":"2022-12-10T13:03:30.698388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = [\n    col \n    for col in train_data.columns \n    if col not in ['MachineIdentifier', 'Census_SystemVolumeTotalCapacity', 'HasDetections'] \n    and \n    str(train_data[col].dtype) == 'category'\n]","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:03:30.701309Z","iopub.execute_input":"2022-12-10T13:03:30.701651Z","iopub.status.idle":"2022-12-10T13:03:30.712898Z","shell.execute_reply.started":"2022-12-10T13:03:30.701621Z","shell.execute_reply":"2022-12-10T13:03:30.711328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def frequency_encoding(variable):\n    t = pd.concat([_train[variable]]).value_counts().reset_index()\n    t = t.reset_index()\n    t.loc[t[variable] == 1, 'level_0'] = np.nan\n    t.set_index('index', inplace=True)\n    max_label = t['level_0'].max() + 1\n    t.fillna(max_label, inplace=True)\n    return t.to_dict()['level_0']\n\nfor col in tqdm(cat_cols):\n    freq_enc_dict = frequency_encoding(col)\n    _train[col] = _train[col].map(lambda x: freq_enc_dict.get(x, np.nan))","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:07:07.739608Z","iopub.execute_input":"2022-12-10T13:07:07.740304Z","iopub.status.idle":"2022-12-10T13:07:14.376228Z","shell.execute_reply.started":"2022-12-10T13:07:07.740262Z","shell.execute_reply":"2022-12-10T13:07:14.374968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nindexer = {}\nfor col in tqdm(cat_cols):\n    _, indexer[col] = pd.factorize(_train[col].astype(str), sort=True)\n    \nfor col in tqdm(cat_cols):\n    _train[col] = indexer[col].get_indexer(_train[col].astype(str))","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:09:14.550785Z","iopub.execute_input":"2022-12-10T13:09:14.552064Z","iopub.status.idle":"2022-12-10T13:15:08.735035Z","shell.execute_reply.started":"2022-12-10T13:09:14.551997Z","shell.execute_reply":"2022-12-10T13:15:08.733754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del indexer","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:15:08.737660Z","iopub.execute_input":"2022-12-10T13:15:08.738060Z","iopub.status.idle":"2022-12-10T13:15:08.752809Z","shell.execute_reply.started":"2022-12-10T13:15:08.738028Z","shell.execute_reply":"2022-12-10T13:15:08.751176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"_train.dropna(inplace=True)\ny = _train['HasDetections']\nX = _train.drop(['HasDetections', 'MachineIdentifier'], axis=1)\ngc.collect()\nX.sort_values('AvSigVersion')\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=1)","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:44:17.461105Z","iopub.execute_input":"2022-12-10T13:44:17.461611Z","iopub.status.idle":"2022-12-10T13:45:01.275116Z","shell.execute_reply.started":"2022-12-10T13:44:17.461571Z","shell.execute_reply":"2022-12-10T13:45:01.273812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_fold = 5\nfolds = StratifiedKFold(n_splits=n_fold, shuffle=True, random_state=15)","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:45:16.126731Z","iopub.execute_input":"2022-12-10T13:45:16.127183Z","iopub.status.idle":"2022-12-10T13:45:16.134830Z","shell.execute_reply.started":"2022-12-10T13:45:16.127145Z","shell.execute_reply":"2022-12-10T13:45:16.132963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from numba import jit\n\n@jit\ndef fast_auc(y_true, y_prob):\n    y_true = np.asarray(y_true)\n    y_true = y_true[np.argsort(y_prob)]\n    nfalse = 0\n    auc = 0\n    n = len(y_true)\n    for i in range(n):\n        y_i = y_true[i]\n        nfalse += (1 - y_i)\n        auc += y_i * nfalse\n    auc /= (nfalse * (n - nfalse))\n    return auc\n\ndef eval_auc(preds, dtrain):\n    labels = dtrain.get_label()\n    return 'auc', fast_auc(labels, preds), True","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:45:31.098756Z","iopub.execute_input":"2022-12-10T13:45:31.099263Z","iopub.status.idle":"2022-12-10T13:45:32.016241Z","shell.execute_reply.started":"2022-12-10T13:45:31.099222Z","shell.execute_reply":"2022-12-10T13:45:32.014985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {'num_leaves': 256,\n         'min_data_in_leaf': 42,\n         'objective': 'binary',\n         'max_depth': 5,\n         'learning_rate': 0.05,\n         \"boosting\": \"gbdt\",\n         \"feature_fraction\": 0.8,\n         \"bagging_freq\": 5,\n         \"bagging_fraction\": 0.8,\n         \"bagging_seed\": 11,\n         \"lambda_l1\": 0.15,\n         \"lambda_l2\": 0.15,\n         \"random_state\": 42,          \n         \"verbosity\": -1}","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:45:33.257461Z","iopub.execute_input":"2022-12-10T13:45:33.257887Z","iopub.status.idle":"2022-12-10T13:45:33.264504Z","shell.execute_reply.started":"2022-12-10T13:45:33.257848Z","shell.execute_reply":"2022-12-10T13:45:33.263245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del stats_df, freq_enc_dict","metadata":{"execution":{"iopub.status.busy":"2022-12-10T13:45:35.719643Z","iopub.execute_input":"2022-12-10T13:45:35.720507Z","iopub.status.idle":"2022-12-10T13:45:35.801837Z","shell.execute_reply.started":"2022-12-10T13:45:35.720454Z","shell.execute_reply":"2022-12-10T13:45:35.800746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_model(X, y, params=None, folds=folds):\n    scores = []\n    feature_importance = pd.DataFrame()\n    accuracy = []\n    for fold_n, (train_index, valid_index) in enumerate(folds.split(X, y)):\n        gc.collect()\n        print('Fold', fold_n + 1)\n        X_train, X_valid = X.iloc[train_index], X.iloc[valid_index]\n        y_train, y_valid = y.iloc[train_index], y.iloc[valid_index]\n        \n        train_data = lgb.Dataset(X_train, label=y_train, categorical_feature = cat_cols)\n        valid_data = lgb.Dataset(X_valid, label=y_valid, categorical_feature = cat_cols)\n\n        model = lgb.train(params,\n                train_data,\n                num_boost_round=100,\n                valid_sets = [train_data, valid_data],\n                verbose_eval=10,\n                early_stopping_rounds = 20,\n                feval=eval_auc)\n\n        del train_data, valid_data\n\n        y_pred_valid = model.predict(X_valid, num_iteration=model.best_iteration)\n        del X_valid\n        gc.collect()\n        auc = metrics.roc_auc_score(y_valid, y_pred_valid)\n        accuracy.append(((np.round(y_pred_valid)==y_valid).sum())/len(y_valid))\n        print(f'Accuracy:{accuracy[-1]}')\n        false_positive_rate, true_positive_rate, thresolds = metrics.roc_curve(y_valid, y_pred_valid)\n        plt.axis('scaled')\n        plt.xlim([0, 1])\n        plt.ylim([0, 1])\n        plt.title(\"AUC & ROC Curve\")\n        plt.plot(false_positive_rate, true_positive_rate, 'g')\n        plt.fill_between(false_positive_rate, true_positive_rate, facecolor='lightgreen', alpha=0.7)\n        plt.text(0.95, 0.05, 'AUC = %0.4f' % auc, ha='right', fontsize=12, weight='bold', color='blue')\n        plt.xlabel(\"False Positive Rate\")\n        plt.ylabel(\"True Positive Rate\")\n        plt.show()\n            \n        scores.append(fast_auc(y_valid, y_pred_valid))\n        print('Fold roc_auc:', roc_auc_score(y_valid, y_pred_valid))\n        print('')\n        \n        \n        fold_importance = pd.DataFrame()\n        fold_importance[\"feature\"] = X.columns\n        fold_importance[\"importance\"] = model.feature_importance()\n        fold_importance[\"fold\"] = fold_n + 1\n        feature_importance = pd.concat([feature_importance, fold_importance], axis=0)\n    \n    print('CV mean score: {0:.4f}, std: {1:.4f}.'.format(np.mean(scores), np.std(scores)))\n    \n    feature_importance[\"importance\"] /= n_fold\n    cols = feature_importance[[\"feature\", \"importance\"]].groupby(\"feature\").mean().sort_values(\n        by=\"importance\", ascending=False)[:50].index\n\n    best_features = feature_importance.loc[feature_importance.feature.isin(cols)]\n\n    plt.figure(figsize=(16, 12));\n    sns.barplot(x=\"importance\", y=\"feature\", data=best_features.sort_values(by=\"importance\", ascending=False));\n    plt.title('LGB Features (avg over folds)');\n    return accuracy","metadata":{"execution":{"iopub.status.busy":"2022-12-10T17:09:35.502457Z","iopub.execute_input":"2022-12-10T17:09:35.503311Z","iopub.status.idle":"2022-12-10T17:09:35.523277Z","shell.execute_reply.started":"2022-12-10T17:09:35.503260Z","shell.execute_reply":"2022-12-10T17:09:35.521400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy = train_model(X=X_train, y=y_train, params=params)","metadata":{"execution":{"iopub.status.busy":"2022-12-10T17:09:42.265036Z","iopub.execute_input":"2022-12-10T17:09:42.265544Z","iopub.status.idle":"2022-12-10T17:40:13.052602Z","shell.execute_reply.started":"2022-12-10T17:09:42.265499Z","shell.execute_reply":"2022-12-10T17:40:13.051270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.xlabel('Fold')\nplt.ylabel('Accuracy')\nplt.xticks(np.arange(5), labels=np.arange(1,6))\nplt.plot(accuracy);","metadata":{"execution":{"iopub.status.busy":"2022-12-10T17:44:48.703090Z","iopub.execute_input":"2022-12-10T17:44:48.703550Z","iopub.status.idle":"2022-12-10T17:44:48.935000Z","shell.execute_reply.started":"2022-12-10T17:44:48.703517Z","shell.execute_reply":"2022-12-10T17:44:48.933405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Links\n1. [Microsoft Malware Detection Dataset](https://www.kaggle.com/competitions/microsoft-malware-prediction)\n2. [Visualisation - Notebook](https://www.kaggle.com/code/junikki/ml-project-phase-1)","metadata":{}}]}