{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\npd.set_option('max_columns',35)\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import SimpleImputer,KNNImputer,IterativeImputer\nfrom catboost import CatBoostRegressor\nimport xgboost\nimport lightgbm as lgb\nfrom lightgbm import LGBMClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import StratifiedKFold\n\nimport plotly.express as px\nimport plotly.graph_objs as go\nfrom plotly.tools import FigureFactory as FF\nfrom plotly.subplots import make_subplots","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-04T14:10:01.657754Z","iopub.execute_input":"2022-08-04T14:10:01.658473Z","iopub.status.idle":"2022-08-04T14:10:01.665990Z","shell.execute_reply.started":"2022-08-04T14:10:01.658435Z","shell.execute_reply":"2022-08-04T14:10:01.664964Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Reading the training set\ndf=pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ndf.drop('id',axis=1,inplace=True)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:36:19.291126Z","iopub.execute_input":"2022-08-04T13:36:19.291749Z","iopub.status.idle":"2022-08-04T13:36:19.401271Z","shell.execute_reply.started":"2022-08-04T13:36:19.291712Z","shell.execute_reply":"2022-08-04T13:36:19.400126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Reading the test set\ntest = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')\ntest.drop('id',axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:36:23.627720Z","iopub.execute_input":"2022-08-04T13:36:23.628564Z","iopub.status.idle":"2022-08-04T13:36:23.726768Z","shell.execute_reply.started":"2022-08-04T13:36:23.628523Z","shell.execute_reply":"2022-08-04T13:36:23.725344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We segregate the columns based on the data type, so that we can analyse the data better. We segregate it to float, int and categorical columns","metadata":{}},{"cell_type":"code","source":"cat_cols=[col for col in df.columns if df[col].dtype=='object']\nfloat_cols = [col for col in df.columns if df[col].dtype=='float64']\nint_cols = [col for col in df.columns if df[col].dtype=='int64']\n\nprint(f\"The categorical columns are:\\n {cat_cols}\")\nprint(f\"The continuous value columns are:\\n {float_cols}\")\nprint(f\"The discrete value columns are:\\n {int_cols}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:38:01.540800Z","iopub.execute_input":"2022-08-04T13:38:01.541514Z","iopub.status.idle":"2022-08-04T13:38:01.549518Z","shell.execute_reply.started":"2022-08-04T13:38:01.541475Z","shell.execute_reply":"2022-08-04T13:38:01.547706Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Target Distribution","metadata":{}},{"cell_type":"code","source":"fig=go.Figure(data=[go.Bar(y=df['failure'].value_counts().index, \n                     x=df['failure'].value_counts().values,\n                     orientation=\"h\",\n                    marker=dict(color=[n for n in range(14)], \n                                line_color='rgb(0,0,0)', \n                                line_width = 2,\n                                coloraxis=\"coloraxis\")\n                    )\n                    ])\nfig.update_layout(width=1000,height=600, title_text=\"Target distribution\",title_x=0.5)\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:39:18.416590Z","iopub.execute_input":"2022-08-04T13:39:18.417250Z","iopub.status.idle":"2022-08-04T13:39:18.436605Z","shell.execute_reply.started":"2022-08-04T13:39:18.417209Z","shell.execute_reply":"2022-08-04T13:39:18.435361Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Float Column Distributions","metadata":{}},{"cell_type":"code","source":"fig,ax = plt.subplots(nrows=4,ncols=4,figsize=(20,20))\n\nsns.kdeplot(data=df[float_cols[0]],ax=ax[0][0],shade=True)\nsns.kdeplot(data=df[float_cols[1]],ax=ax[0][1],shade=True)\nsns.kdeplot(data=df[float_cols[2]],ax=ax[0][2],shade=True)\nsns.kdeplot(data=df[float_cols[3]],ax=ax[0][3],shade=True)\nsns.kdeplot(data=df[float_cols[4]],ax=ax[1][0],shade=True)\nsns.kdeplot(data=df[float_cols[5]],ax=ax[1][1],shade=True)\nsns.kdeplot(data=df[float_cols[6]],ax=ax[1][2],shade=True)\nsns.kdeplot(data=df[float_cols[7]],ax=ax[1][3],shade=True)\nsns.kdeplot(data=df[float_cols[8]],ax=ax[2][0],shade=True)\nsns.kdeplot(data=df[float_cols[9]],ax=ax[2][1],shade=True)\nsns.kdeplot(data=df[float_cols[10]],ax=ax[2][2],shade=True)\nsns.kdeplot(data=df[float_cols[11]],ax=ax[2][3],shade=True)\nsns.kdeplot(data=df[float_cols[12]],ax=ax[3][0],shade=True)\nsns.kdeplot(data=df[float_cols[13]],ax=ax[3][1],shade=True)\nsns.kdeplot(data=df[float_cols[14]],ax=ax[3][2],shade=True)\nsns.kdeplot(data=df[float_cols[15]],ax=ax[3][3],shade=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:40:55.082205Z","iopub.execute_input":"2022-08-04T13:40:55.082584Z","iopub.status.idle":"2022-08-04T13:40:59.831650Z","shell.execute_reply.started":"2022-08-04T13:40:55.082551Z","shell.execute_reply":"2022-08-04T13:40:59.830661Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have almost a normal distribution for all the continuous features, which is a good news","metadata":{}},{"cell_type":"markdown","source":"## Integer column correlations","metadata":{}},{"cell_type":"code","source":"corr= df[int_cols[1:]].corr()\n\n# Getting the Upper Triangle of the co-relation matrix\nmatrix = np.triu(corr)\n\nplt.figure(figsize=(10,6))\nsns.heatmap(corr,annot=True,mask=matrix)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:41:38.562700Z","iopub.execute_input":"2022-08-04T13:41:38.563423Z","iopub.status.idle":"2022-08-04T13:41:38.836903Z","shell.execute_reply.started":"2022-08-04T13:41:38.563385Z","shell.execute_reply":"2022-08-04T13:41:38.835870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Category column distributions","metadata":{}},{"cell_type":"code","source":"def unique_values(data):\n    for col in cat_cols:\n        print('*'*50)\n        print(f\"The unique values in {col} are:\")\n        print(data[col].unique())","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:43:14.924703Z","iopub.execute_input":"2022-08-04T13:43:14.925429Z","iopub.status.idle":"2022-08-04T13:43:14.930890Z","shell.execute_reply.started":"2022-08-04T13:43:14.925381Z","shell.execute_reply":"2022-08-04T13:43:14.929868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_values(df)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:43:17.258425Z","iopub.execute_input":"2022-08-04T13:43:17.259468Z","iopub.status.idle":"2022-08-04T13:43:17.271560Z","shell.execute_reply.started":"2022-08-04T13:43:17.259419Z","shell.execute_reply":"2022-08-04T13:43:17.270059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_values(test)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:43:33.828064Z","iopub.execute_input":"2022-08-04T13:43:33.829101Z","iopub.status.idle":"2022-08-04T13:43:33.841607Z","shell.execute_reply.started":"2022-08-04T13:43:33.829061Z","shell.execute_reply":"2022-08-04T13:43:33.840538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We see that the unique values in training and test set for the features 'product code' and 'attribute_1' are different. That is the training set will not have the unique values of product code and attribute_1.\n\nHence for simplicity we avoid these columns while handling the missing values","metadata":{}},{"cell_type":"markdown","source":"## Full feature correlation","metadata":{}},{"cell_type":"code","source":"corr= df.corr()\n\n# Getting the Upper Triangle of the co-relation matrix\nmatrix = np.triu(corr)\n\nplt.figure(figsize=(20,20))\nsns.heatmap(corr,annot=True,mask=matrix)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:47:12.963096Z","iopub.execute_input":"2022-08-04T13:47:12.963506Z","iopub.status.idle":"2022-08-04T13:47:14.326671Z","shell.execute_reply.started":"2022-08-04T13:47:12.963473Z","shell.execute_reply":"2022-08-04T13:47:14.325760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the correlations, we see that no feature has a good linear relationship with the failure variable. Loading feature has a decent positive correlation, but hard luck with all the other features.\n\nWe need to check for any non linear relation and feature interactions ","metadata":{}},{"cell_type":"markdown","source":"## Missing Values","metadata":{}},{"cell_type":"markdown","source":"Let's explore the distribution of the missing values, and accordingly we will check for a strategy to impute the missing values","metadata":{}},{"cell_type":"code","source":"fig=go.Figure(data=[go.Bar(y=df.isna().sum().sort_values(ascending=False).index[:-1], \n                     x=df.isna().sum().sort_values(ascending=False).values[:-1],\n                     orientation=\"h\",\n                    marker=dict(color=[n for n in range(14)], \n                                line_color='rgb(0,0,0)', \n                                line_width = 2,\n                                coloraxis=\"coloraxis\")\n                    )\n                    ])\nfig.update_layout(showlegend=False, title_text=\"Missing values distribution\", title_x=0.5)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T13:49:35.872928Z","iopub.execute_input":"2022-08-04T13:49:35.873674Z","iopub.status.idle":"2022-08-04T13:49:35.900711Z","shell.execute_reply.started":"2022-08-04T13:49:35.873637Z","shell.execute_reply":"2022-08-04T13:49:35.899789Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We see that the only missing values in the features are from the float columns i.e continuous features\n\nHence, we will use only the float columns as continous variables only has missing values, and we'll use iterative imputer with XGBoost regressor to compute the missing values","metadata":{}},{"cell_type":"code","source":"def missing(train,test):\n    df_miss = df[float_cols]\n    test_miss = test[float_cols]\n    return df_miss,test_miss","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:04:40.979403Z","iopub.execute_input":"2022-08-04T14:04:40.979780Z","iopub.status.idle":"2022-08-04T14:04:40.984629Z","shell.execute_reply.started":"2022-08-04T14:04:40.979747Z","shell.execute_reply":"2022-08-04T14:04:40.983545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_miss,test_miss = missing(df,test)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:04:42.140737Z","iopub.execute_input":"2022-08-04T14:04:42.141708Z","iopub.status.idle":"2022-08-04T14:04:42.149429Z","shell.execute_reply.started":"2022-08-04T14:04:42.141664Z","shell.execute_reply":"2022-08-04T14:04:42.148359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imp = IterativeImputer(\n    estimator=xgboost.XGBRegressor(\n        n_estimators=150,\n        random_state=333,\n        tree_method='gpu_hist',\n    ),\n    missing_values=np.nan,\n    max_iter=20,\n    initial_strategy='mean',\n    imputation_order='ascending',\n    verbose=2,\n    random_state=333\n)\n\n#Fitting to training set, and transforming train and test sets\ndf_miss[:] = imp.fit_transform(df_miss)\ntest_miss[:] = imp.transform(test_miss)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:04:43.082611Z","iopub.execute_input":"2022-08-04T14:04:43.083297Z","iopub.status.idle":"2022-08-04T14:07:55.401994Z","shell.execute_reply.started":"2022-08-04T14:04:43.083261Z","shell.execute_reply":"2022-08-04T14:07:55.400911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.concat([df_miss,df[int_cols],df[cat_cols]],axis=1)\ntrain.to_csv('training_set.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:10:23.925876Z","iopub.execute_input":"2022-08-04T14:10:23.926248Z","iopub.status.idle":"2022-08-04T14:10:24.258755Z","shell.execute_reply.started":"2022-08-04T14:10:23.926216Z","shell.execute_reply":"2022-08-04T14:10:24.257713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.concat([test_miss,test[int_cols[:-1]],test[cat_cols]],axis=1)\ntest.to_csv('test_set.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:11:31.232750Z","iopub.execute_input":"2022-08-04T14:11:31.233679Z","iopub.status.idle":"2022-08-04T14:11:31.498160Z","shell.execute_reply.started":"2022-08-04T14:11:31.233643Z","shell.execute_reply":"2022-08-04T14:11:31.497093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have a bit more exploration to do in the data, which will be done shortly","metadata":{}},{"cell_type":"markdown","source":"![](https://cdn-icons-png.flaticon.com/512/5578/5578703.png)","metadata":{}}]}