{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nimport warnings\nimport optuna\nfrom sklearn.preprocessing import LabelEncoder \nfrom sklearn.impute import SimpleImputer\nfrom sklearn.ensemble import RandomForestClassifier\nfrom xgboost import XGBClassifier\nfrom itertools import cycle, islice\nfrom sklearn.metrics import roc_auc_score, roc_curve\nfrom lightgbm import LGBMClassifier, early_stopping, log_evaluation\nfrom sklearn.model_selection import train_test_split, KFold, cross_val_score ,StratifiedKFold","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-04T13:42:27.488268Z","iopub.execute_input":"2022-08-04T13:42:27.489031Z","iopub.status.idle":"2022-08-04T13:42:27.498986Z","shell.execute_reply.started":"2022-08-04T13:42:27.488980Z","shell.execute_reply":"2022-08-04T13:42:27.497969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"color_pal = plt.rcParams[\"axes.prop_cycle\"].by_key()[\"color\"]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T10:58:21.355402Z","iopub.execute_input":"2022-08-04T10:58:21.355851Z","iopub.status.idle":"2022-08-04T10:58:21.361349Z","shell.execute_reply.started":"2022-08-04T10:58:21.355813Z","shell.execute_reply":"2022-08-04T10:58:21.360185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pd.read_csv(\"../input/tabular-playground-series-aug-2022/train.csv\",index_col='id')\ntest=pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv',index_col='id')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:42:31.439925Z","iopub.execute_input":"2022-08-04T13:42:31.440382Z","iopub.status.idle":"2022-08-04T13:42:31.693416Z","shell.execute_reply.started":"2022-08-04T13:42:31.440347Z","shell.execute_reply":"2022-08-04T13:42:31.692061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:42:42.175196Z","iopub.execute_input":"2022-08-04T13:42:42.175697Z","iopub.status.idle":"2022-08-04T13:42:42.212125Z","shell.execute_reply.started":"2022-08-04T13:42:42.175656Z","shell.execute_reply":"2022-08-04T13:42:42.210651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape , test.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:42:45.090273Z","iopub.execute_input":"2022-08-04T13:42:45.091298Z","iopub.status.idle":"2022-08-04T13:42:45.098149Z","shell.execute_reply.started":"2022-08-04T13:42:45.091249Z","shell.execute_reply":"2022-08-04T13:42:45.097100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:42:52.208228Z","iopub.execute_input":"2022-08-04T13:42:52.209130Z","iopub.status.idle":"2022-08-04T13:42:52.240386Z","shell.execute_reply.started":"2022-08-04T13:42:52.209078Z","shell.execute_reply":"2022-08-04T13:42:52.239229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Some Basic informations about the train dataframe","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:43:00.479768Z","iopub.execute_input":"2022-08-04T13:43:00.480210Z","iopub.status.idle":"2022-08-04T13:43:00.502152Z","shell.execute_reply.started":"2022-08-04T13:43:00.480174Z","shell.execute_reply":"2022-08-04T13:43:00.501087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Lets look at our Categorical Columns**\n\n****There are three categorical columns presented in our DataFrame****","metadata":{}},{"cell_type":"code","source":"categorical_columns= [c for c in train.columns if train[c].dtype == 'object' and c!='product_code']\ncategorical_columns","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:43:08.563073Z","iopub.execute_input":"2022-08-04T13:43:08.563585Z","iopub.status.idle":"2022-08-04T13:43:08.571983Z","shell.execute_reply.started":"2022-08-04T13:43:08.563543Z","shell.execute_reply":"2022-08-04T13:43:08.570715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.concat([train[categorical_columns].isna().sum().rename('Missing values in train'),\n          test[categorical_columns].isna().sum().rename('Missing values in test')],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:00:22.726012Z","iopub.execute_input":"2022-08-04T11:00:22.727434Z","iopub.status.idle":"2022-08-04T11:00:22.749322Z","shell.execute_reply.started":"2022-08-04T11:00:22.727369Z","shell.execute_reply":"2022-08-04T11:00:22.747726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We have found that there is no missing values in our categorical columns**","metadata":{}},{"cell_type":"markdown","source":"###  Is there any null value presented in the DataFrame ?? Lets find this !!","metadata":{}},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:43:34.126026Z","iopub.execute_input":"2022-08-04T13:43:34.126506Z","iopub.status.idle":"2022-08-04T13:43:34.142552Z","shell.execute_reply.started":"2022-08-04T13:43:34.126467Z","shell.execute_reply":"2022-08-04T13:43:34.141279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Float Columns - 16 Float Columns**\n\n***We have to found what are the missing values in the Float Columns.***","metadata":{}},{"cell_type":"code","source":"float_columns = [f for f in train.columns if train[f].dtype == 'float']\nfloat_columns","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:43:43.596501Z","iopub.execute_input":"2022-08-04T13:43:43.597719Z","iopub.status.idle":"2022-08-04T13:43:43.605329Z","shell.execute_reply.started":"2022-08-04T13:43:43.597676Z","shell.execute_reply":"2022-08-04T13:43:43.604328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.concat([train[float_columns].isna().sum().rename('Missing values in train'),\n           test[float_columns].isna().sum().rename('Missing values in test')],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:02:54.260392Z","iopub.execute_input":"2022-08-04T11:02:54.260815Z","iopub.status.idle":"2022-08-04T11:02:54.283805Z","shell.execute_reply.started":"2022-08-04T11:02:54.260781Z","shell.execute_reply":"2022-08-04T11:02:54.281878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Lets look at our Integer Columns.\n***How many integer columns we have?***\n\n****Is there any missing values in the integer column? Lets find !!****","metadata":{}},{"cell_type":"code","source":"integer_columns = [c for c in train.columns if train[c].dtype == 'int' and c!= 'failure']\ninteger_columns","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:43:53.330959Z","iopub.execute_input":"2022-08-04T13:43:53.331415Z","iopub.status.idle":"2022-08-04T13:43:53.339922Z","shell.execute_reply.started":"2022-08-04T13:43:53.331379Z","shell.execute_reply":"2022-08-04T13:43:53.338617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***So, basically there are 5 integer columns . We have never consider our target column as an integer column*** \n\n***Luckily we do not have any missing values in our integer columns***","metadata":{}},{"cell_type":"code","source":"pd.concat([train[integer_columns].isna().sum().rename('Missing values in train'),\n          test[integer_columns].isna().sum().rename('Missing values in test')],axis= 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:03:45.465117Z","iopub.execute_input":"2022-08-04T11:03:45.465558Z","iopub.status.idle":"2022-08-04T11:03:45.483037Z","shell.execute_reply.started":"2022-08-04T11:03:45.465522Z","shell.execute_reply":"2022-08-04T11:03:45.482171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" ## Lets look at our target Column - 'Failure'\n \n ### Here 'Failure' column contains two values. 0 & 1.\n \n #### Of the 26570 products tested, 79 % are good and 21 % fail.","metadata":{"execution":{"iopub.status.busy":"2022-08-03T05:22:30.035321Z","iopub.execute_input":"2022-08-03T05:22:30.035734Z","iopub.status.idle":"2022-08-03T05:22:30.039468Z","shell.execute_reply.started":"2022-08-03T05:22:30.035703Z","shell.execute_reply":"2022-08-03T05:22:30.038757Z"}}},{"cell_type":"code","source":"print(train['failure'].value_counts()/len(train))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:31:08.246759Z","iopub.execute_input":"2022-08-04T13:31:08.247953Z","iopub.status.idle":"2022-08-04T13:31:08.259071Z","shell.execute_reply.started":"2022-08-04T13:31:08.247896Z","shell.execute_reply":"2022-08-04T13:31:08.257580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*****Let us look at our product code column*****","metadata":{}},{"cell_type":"code","source":"train['product_code'].value_counts() , test['product_code'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:44:08.843083Z","iopub.execute_input":"2022-08-04T13:44:08.843585Z","iopub.status.idle":"2022-08-04T13:44:08.859315Z","shell.execute_reply.started":"2022-08-04T13:44:08.843546Z","shell.execute_reply":"2022-08-04T13:44:08.858146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Test set has totally different product codes from train dataset**\n\n**Product C has slightly more samples than the others but generally they re balanced. Failure ratios about 21%.**","metadata":{}},{"cell_type":"code","source":"train.groupby('product_code')['failure'].value_counts(normalize=True).mul(100) ","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:04:14.545905Z","iopub.execute_input":"2022-08-04T11:04:14.546767Z","iopub.status.idle":"2022-08-04T11:04:14.566104Z","shell.execute_reply.started":"2022-08-04T11:04:14.546722Z","shell.execute_reply":"2022-08-04T11:04:14.565191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(1,2, figsize=(12,5),sharey=True)\n\ntrain.groupby('product_code')['loading'].count().plot(kind='bar', ax=axs[0],title=\"Train Product Type\",color='red')\ntest.groupby('product_code')['loading'].count().plot(kind='bar', ax=axs[1],title=\"Test Procut Type\",color='blue')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:04:25.147716Z","iopub.execute_input":"2022-08-04T11:04:25.148406Z","iopub.status.idle":"2022-08-04T11:04:25.516082Z","shell.execute_reply.started":"2022-08-04T11:04:25.148363Z","shell.execute_reply":"2022-08-04T11:04:25.514439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"colors = list(islice(cycle(['y','g']),None, len(train.product_code.unique()) *2))\ncolors","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:04:32.953456Z","iopub.execute_input":"2022-08-04T11:04:32.953915Z","iopub.status.idle":"2022-08-04T11:04:32.965822Z","shell.execute_reply.started":"2022-08-04T11:04:32.953877Z","shell.execute_reply":"2022-08-04T11:04:32.964622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.groupby(['product_code','failure'])['loading'].count().plot(kind='bar',color=colors,figsize=(12, 5),title= 'Failure Ratio by Product Category')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:04:36.449539Z","iopub.execute_input":"2022-08-04T11:04:36.449981Z","iopub.status.idle":"2022-08-04T11:04:36.697120Z","shell.execute_reply.started":"2022-08-04T11:04:36.449941Z","shell.execute_reply":"2022-08-04T11:04:36.695900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Loading Column***","metadata":{}},{"cell_type":"code","source":"train[train['failure']==0].groupby(['product_code']).agg({'loading': ['mean', 'min', 'max']}).plot(kind='bar',figsize=(12, 7), title=\"Success Data by Product Type\")","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:04:42.290599Z","iopub.execute_input":"2022-08-04T11:04:42.291103Z","iopub.status.idle":"2022-08-04T11:04:42.529149Z","shell.execute_reply.started":"2022-08-04T11:04:42.291062Z","shell.execute_reply":"2022-08-04T11:04:42.528040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train['failure']==1].groupby(['product_code']).agg({'loading': ['mean', 'min', 'max']}).plot(kind='bar',figsize=(12, 7), title=\"Failure Data by Product\")","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:04:48.179465Z","iopub.execute_input":"2022-08-04T11:04:48.179880Z","iopub.status.idle":"2022-08-04T11:04:48.417167Z","shell.execute_reply.started":"2022-08-04T11:04:48.179846Z","shell.execute_reply":"2022-08-04T11:04:48.415915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train['failure']==1].groupby(['product_code']).agg({'loading': ['mean', 'min', 'max']})","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:04:58.110807Z","iopub.execute_input":"2022-08-04T11:04:58.111276Z","iopub.status.idle":"2022-08-04T11:04:58.134395Z","shell.execute_reply.started":"2022-08-04T11:04:58.111235Z","shell.execute_reply":"2022-08-04T11:04:58.133226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train['failure']==0].groupby(['product_code']).agg({'loading': ['mean', 'min', 'max']})","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:05:04.581520Z","iopub.execute_input":"2022-08-04T11:05:04.582027Z","iopub.status.idle":"2022-08-04T11:05:04.609475Z","shell.execute_reply.started":"2022-08-04T11:05:04.581943Z","shell.execute_reply":"2022-08-04T11:05:04.608022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Loading column have some missing data**","metadata":{}},{"cell_type":"code","source":"train[train['loading'].isna()].groupby(['product_code','failure'])['attribute_0'].count()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:05:11.388931Z","iopub.execute_input":"2022-08-04T11:05:11.389375Z","iopub.status.idle":"2022-08-04T11:05:11.403654Z","shell.execute_reply.started":"2022-08-04T11:05:11.389335Z","shell.execute_reply":"2022-08-04T11:05:11.402693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[test['loading'].isna()].groupby(['product_code'])['attribute_0'].count()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:05:18.078756Z","iopub.execute_input":"2022-08-04T11:05:18.079218Z","iopub.status.idle":"2022-08-04T11:05:18.091641Z","shell.execute_reply.started":"2022-08-04T11:05:18.079176Z","shell.execute_reply":"2022-08-04T11:05:18.090037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Analyzing the attiribute columns***\n\n","metadata":{}},{"cell_type":"code","source":"train.groupby(['product_code','attribute_0','attribute_1','attribute_2','attribute_3']).size()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:05:20.889045Z","iopub.execute_input":"2022-08-04T11:05:20.890125Z","iopub.status.idle":"2022-08-04T11:05:20.909918Z","shell.execute_reply.started":"2022-08-04T11:05:20.890084Z","shell.execute_reply":"2022-08-04T11:05:20.908931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.groupby(['product_code','attribute_0','attribute_1','attribute_2','attribute_3']).size()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:06:09.005152Z","iopub.execute_input":"2022-08-04T11:06:09.005545Z","iopub.status.idle":"2022-08-04T11:06:09.025481Z","shell.execute_reply.started":"2022-08-04T11:06:09.005512Z","shell.execute_reply":"2022-08-04T11:06:09.023975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***There are four attribute columns . The first two are about materials and the last two are just categorical numbers.***\n\n\n**We can say,  Product code was defined by the Attributes combinations. Each product code has only one attribute combinations.**","metadata":{}},{"cell_type":"markdown","source":"## Measurement Columns 0-17 ##","metadata":{}},{"cell_type":"markdown","source":"**There are total 18 Measurement columns are in our database. Mesurements except for the first two mesurement have some missing values**","metadata":{}},{"cell_type":"code","source":"measurements = [a for a in train.columns if a.startswith('measurement')]\ntrain[measurements].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:45:06.144445Z","iopub.execute_input":"2022-08-04T13:45:06.144929Z","iopub.status.idle":"2022-08-04T13:45:06.161267Z","shell.execute_reply.started":"2022-08-04T13:45:06.144889Z","shell.execute_reply":"2022-08-04T13:45:06.160027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def missing_plot(df,num,name):\n    a = train[measurements].notna().sum().plot.bar(title=f'Mesurement Missing Values Distribution {name}',ax=axs[num])\n    MAX = max(train[measurements].notna().sum())\n    MIN = min(train[measurements].notna().sum())\n    a.set_ybound(MIN-300,MAX+300)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:06:20.474677Z","iopub.execute_input":"2022-08-04T11:06:20.475157Z","iopub.status.idle":"2022-08-04T11:06:20.483614Z","shell.execute_reply.started":"2022-08-04T11:06:20.475101Z","shell.execute_reply":"2022-08-04T11:06:20.482504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(1,2, figsize=(12,5),sharey=True)\n\nmissing_plot(train,0,'Train')\nmissing_plot(test,1,'Test')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:06:22.821435Z","iopub.execute_input":"2022-08-04T11:06:22.821867Z","iopub.status.idle":"2022-08-04T11:06:23.274467Z","shell.execute_reply.started":"2022-08-04T11:06:22.821828Z","shell.execute_reply":"2022-08-04T11:06:23.273188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, axs = plt.subplots(4, 4, figsize=(12,12))\nfor f, ax in zip(float_columns, axs.ravel()):\n    mi = min(train[f].min(), test[f].min())\n    ma = max(train[f].max(), test[f].max())\n    bins = np.linspace(mi, ma, 50)\n    ax.hist(train[f], bins=bins, alpha=0.5, density=True, label='train')\n    ax.hist(test[f], bins=bins, alpha=0.5, density=True, label='test')\n    ax.set_xlabel(f)\n    if ax == axs[0, 0]: ax.legend(loc='lower left')\n        \n    ax2 = ax.twinx()\n    total, _ = np.histogram(train[f], bins=bins)\n    failures, _ = np.histogram(train[f][train.failure == 1], bins=bins)\n    with warnings.catch_warnings(): # ignore divide by zero for empty bins\n        warnings.filterwarnings('ignore', category=RuntimeWarning)\n        ax2.scatter((bins[1:] + bins[:-1]) / 2, failures / total,\n                    color='m', s=10, label='failure probability')\n    ax2.set_ylim(0, 0.5)\n    ax2.tick_params(axis='y', colors='m')\n    if ax == axs[0, 0]: ax2.legend(loc='upper right')\nplt.tight_layout(w_pad=1)\nplt.suptitle('Train and test distributions of the continuous features', fontsize=20, y=1.02)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:06:27.932299Z","iopub.execute_input":"2022-08-04T11:06:27.932708Z","iopub.status.idle":"2022-08-04T11:06:34.927329Z","shell.execute_reply.started":"2022-08-04T11:06:27.932675Z","shell.execute_reply":"2022-08-04T11:06:34.925823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" ## Model Building\n\n***Now moving on towards the modeling portion***","metadata":{}},{"cell_type":"code","source":"train['skf_fold']=-1","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:45:14.569550Z","iopub.execute_input":"2022-08-04T13:45:14.570413Z","iopub.status.idle":"2022-08-04T13:45:14.577216Z","shell.execute_reply.started":"2022-08-04T13:45:14.570366Z","shell.execute_reply":"2022-08-04T13:45:14.575991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skf = StratifiedKFold (n_splits=5 , shuffle=True , random_state=42)\nfor fold, (train_indicies,valid_indicies) in enumerate (skf.split(X=train , y=train.failure.values )):\n    train.loc[valid_indicies,'skf_fold'] = fold","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:45:17.047880Z","iopub.execute_input":"2022-08-04T13:45:17.048352Z","iopub.status.idle":"2022-08-04T13:45:17.068041Z","shell.execute_reply.started":"2022-08-04T13:45:17.048310Z","shell.execute_reply":"2022-08-04T13:45:17.066989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.to_csv('train_folds.csv',index=False) # Saving the folds data into a new a new csv file","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:45:23.754197Z","iopub.execute_input":"2022-08-04T13:45:23.755309Z","iopub.status.idle":"2022-08-04T13:45:24.176435Z","shell.execute_reply.started":"2022-08-04T13:45:23.755258Z","shell.execute_reply":"2022-08-04T13:45:24.175220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_folds_data =  pd.read_csv('./train_folds.csv')\ntrain_folds_data.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:45:25.648243Z","iopub.execute_input":"2022-08-04T13:45:25.648935Z","iopub.status.idle":"2022-08-04T13:45:25.788578Z","shell.execute_reply.started":"2022-08-04T13:45:25.648891Z","shell.execute_reply":"2022-08-04T13:45:25.787315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_columns=[f for f in train_folds_data.columns if f=='failure']\nuseful_features=[c for c in train_folds_data.columns if c not in (['product_code','failure','skf_fold'])]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:45:35.142376Z","iopub.execute_input":"2022-08-04T13:45:35.143274Z","iopub.status.idle":"2022-08-04T13:45:35.150334Z","shell.execute_reply.started":"2022-08-04T13:45:35.143226Z","shell.execute_reply":"2022-08-04T13:45:35.149065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_columns","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:45:38.651875Z","iopub.execute_input":"2022-08-04T13:45:38.652695Z","iopub.status.idle":"2022-08-04T13:45:38.660874Z","shell.execute_reply.started":"2022-08-04T13:45:38.652627Z","shell.execute_reply":"2022-08-04T13:45:38.659609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***We already found that there are some missing values available in the DataFrame , so we have to fill them with some valus.***\n\n***We have used simple imputer for imputing missing values***\n\n*****Most of the features are symmetrically distributed so we will replace missing values with mean value*****","metadata":{}},{"cell_type":"code","source":"scores = []\nfinal_predictions = []\nfor fold in range(5):\n\n\n    x_train = train_folds_data[train_folds_data.skf_fold!=fold].reset_index(drop=True)\n    x_valid  = train_folds_data[train_folds_data.skf_fold==fold].reset_index(drop=True)\n    x_test = test.copy()\n\n    # Hyperparameter tuning using optuna\n\n    # learning_rate = trial.suggest_float(\"learning_rate\",1e-2,0.25,log=True)\n    # reg_lambda = trial.suggest_loguniform('reg_lambda',1e-8,100.0)\n    # reg_alpha =  trial.suggest_loguniform('reg_alpha',1e-8,100.0)\n    # subsample =  trial.suggest_float('subsample', 0.1, 1.0)\n    # colsample_bytree =  trial.suggest_float('colsample_bytree', 0.1, 1.0)\n    # max_depth =   trial.suggest_int('max_depth', 1, 7)\n\n\n    y_train = x_train[target_columns]\n    y_valid  = x_valid[target_columns]\n\n\n    x_train = x_train[useful_features]\n    x_valid  = x_valid[useful_features]\n    x_test = x_test[useful_features]\n    \n    \n    # Converting the Categorial variables into dummies\n    x_train =pd.get_dummies(x_train,columns=categorical_columns, drop_first=True)\n    x_valid =pd.get_dummies(x_valid,columns=categorical_columns, drop_first=True)\n    x_test=pd.get_dummies(x_test,columns=categorical_columns, drop_first=True)\n\n    \n    #handling missing values  by using KNN imputer\n    \n    imputer = SimpleImputer(missing_values=np.nan, strategy='mean')\n    x_train= imputer.fit_transform(x_train)\n    x_valid=imputer.transform(x_valid)\n    x_test=imputer.transform(x_test)\n     \n    \n    \n\n\n\n    model = XGBClassifier(random_state=fold, n_estimators = 8)\n    #                              learning_rate=learning_rate,reg_lambda= reg_lambda,\n    #                              reg_alpha=reg_alpha,colsample_bytree=colsample_bytree,\n    #                              subsample=subsample,max_depth=max_depth)\n\n    model.fit(x_train,y_train)\n\n    preds_valid = model.predict_proba(x_valid)\n    \n    preds_valid = [z[1] for z in preds_valid]\n                               \n    preds_test = model.predict_proba(x_test) \n    \n    preds_test = [z[1] for z in preds_test]\n    \n    final_predictions.append(preds_test)\n\n    roc_score = roc_auc_score(y_valid,preds_valid)\n\n    scores.append(roc_score)\n\nprint(scores)\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:46:14.200227Z","iopub.execute_input":"2022-08-04T13:46:14.201416Z","iopub.status.idle":"2022-08-04T13:46:16.570563Z","shell.execute_reply.started":"2022-08-04T13:46:14.201361Z","shell.execute_reply":"2022-08-04T13:46:16.569198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.mean(final_predictions,axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:46:37.475164Z","iopub.execute_input":"2022-08-04T13:46:37.476159Z","iopub.status.idle":"2022-08-04T13:46:37.490193Z","shell.execute_reply.started":"2022-08-04T13:46:37.476103Z","shell.execute_reply":"2022-08-04T13:46:37.489095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:46:47.402600Z","iopub.execute_input":"2022-08-04T13:46:47.403519Z","iopub.status.idle":"2022-08-04T13:46:47.496267Z","shell.execute_reply.started":"2022-08-04T13:46:47.403468Z","shell.execute_reply":"2022-08-04T13:46:47.494817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['failure']=np.mean(final_predictions,axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:47:12.581761Z","iopub.execute_input":"2022-08-04T13:47:12.582236Z","iopub.status.idle":"2022-08-04T13:47:12.595166Z","shell.execute_reply.started":"2022-08-04T13:47:12.582200Z","shell.execute_reply":"2022-08-04T13:47:12.593995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[['id','failure']].to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:47:14.545157Z","iopub.execute_input":"2022-08-04T13:47:14.545672Z","iopub.status.idle":"2022-08-04T13:47:14.595264Z","shell.execute_reply.started":"2022-08-04T13:47:14.545630Z","shell.execute_reply":"2022-08-04T13:47:14.593761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}