{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Tabular Playground Series - Aug 2022","metadata":{}},{"cell_type":"markdown","source":"\n### Table of Contents : \n\n  * [Data Manipulation](#sec1)\n       * [Importing Dataset](#sec1.1)\n       * [Dataset View](#sec1.2)\n       * [Dataset Information](#sec1.3)\n       * [Summary Statistics](#sec1.4)\n       * [Checking for unique values in integer type attribute](#sec1.5)\n       * [Checking for missing values in each column](#sec1.6)\n       * [percentage of missing values in each column](#sec1.7)\n       \n  * [Data Visualization](#sec2)\n       * [Missing Value Plot](#sec2.1)\n       * [Density Plot of Continuous Variables](#sec2.2)\n       * [Heatmap](#sec2.3)\n       * [Analysing categorical features with Pie chart](#sec2.4)\n       \n  * [Modeling](#sec3)\n       * [Simple Imputer for filling missing values](#sec3.2)\n       * [Applying ANN on training dataset](#sec3.3)\n       * [Precting Values from the test dataset](#sec3.3)\n       \n   * [Importing Submission File](#sec4)","metadata":{}},{"cell_type":"markdown","source":"## Data Manipulation <a class=\"anchor\" id=\"sec1\"></a>","metadata":{}},{"cell_type":"markdown","source":"### Importing libraries ","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport seaborn as sns","metadata":{"execution":{"iopub.execute_input":"2022-06-26T17:41:23.063598Z","iopub.status.busy":"2022-06-26T17:41:23.062685Z","iopub.status.idle":"2022-06-26T17:41:24.269647Z","shell.execute_reply":"2022-06-26T17:41:24.268587Z","shell.execute_reply.started":"2022-06-26T17:41:23.063475Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option(\"display.max_rows\", 100, \"display.max_columns\", 100)","metadata":{"execution":{"iopub.execute_input":"2022-06-26T17:41:24.271611Z","iopub.status.busy":"2022-06-26T17:41:24.271303Z","iopub.status.idle":"2022-06-26T17:41:24.277026Z","shell.execute_reply":"2022-06-26T17:41:24.275305Z","shell.execute_reply.started":"2022-06-26T17:41:24.271580Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Importing dataset <a class=\"anchor\" id=\"sec1.1\"></a>","metadata":{}},{"cell_type":"code","source":"df=pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ntest=pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')","metadata":{"execution":{"iopub.execute_input":"2022-06-26T17:41:24.279125Z","iopub.status.busy":"2022-06-26T17:41:24.278507Z","iopub.status.idle":"2022-06-26T17:41:43.473951Z","shell.execute_reply":"2022-06-26T17:41:43.472982Z","shell.execute_reply.started":"2022-06-26T17:41:24.279071Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dataset View <a class=\"anchor\" id=\"sec1.2\"></a>","metadata":{}},{"cell_type":"code","source":"df.head(10)","metadata":{"execution":{"iopub.execute_input":"2022-06-26T17:41:43.476239Z","iopub.status.busy":"2022-06-26T17:41:43.475646Z","iopub.status.idle":"2022-06-26T17:41:43.555474Z","shell.execute_reply":"2022-06-26T17:41:43.554555Z","shell.execute_reply.started":"2022-06-26T17:41:43.476199Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dataset Information <a class=\"anchor\" id=\"sec1.3\"></a>","metadata":{}},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.execute_input":"2022-06-26T17:47:54.643855Z","iopub.status.busy":"2022-06-26T17:47:54.643188Z","iopub.status.idle":"2022-06-26T17:47:54.840888Z","shell.execute_reply":"2022-06-26T17:47:54.839829Z","shell.execute_reply.started":"2022-06-26T17:47:54.643793Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Summary Statistics <a class=\"anchor\" id=\"sec1.4\"></a>","metadata":{}},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.execute_input":"2022-06-26T17:47:58.316431Z","iopub.status.busy":"2022-06-26T17:47:58.315709Z","iopub.status.idle":"2022-06-26T17:48:02.722944Z","shell.execute_reply":"2022-06-26T17:48:02.721892Z","shell.execute_reply.started":"2022-06-26T17:47:58.316391Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Checking for unique values in all attribute <a class=\"anchor\" id=\"sec1.5\"></a>","metadata":{}},{"cell_type":"code","source":"df.nunique().sort_values(ascending=True)","metadata":{"execution":{"iopub.execute_input":"2022-06-26T17:48:02.725033Z","iopub.status.busy":"2022-06-26T17:48:02.724595Z","iopub.status.idle":"2022-06-26T17:48:02.955890Z","shell.execute_reply":"2022-06-26T17:48:02.954946Z","shell.execute_reply.started":"2022-06-26T17:48:02.724998Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Checking for missing values in each column <a class=\"anchor\" id=\"sec1.6\"></a>","metadata":{}},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"execution":{"iopub.execute_input":"2022-06-26T17:48:06.308672Z","iopub.status.busy":"2022-06-26T17:48:06.307498Z","iopub.status.idle":"2022-06-26T17:48:06.463777Z","shell.execute_reply":"2022-06-26T17:48:06.463024Z","shell.execute_reply.started":"2022-06-26T17:48:06.308615Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### percentage of missing values in each column <a class=\"anchor\" id=\"sec1.7\"></a>","metadata":{}},{"cell_type":"code","source":"pd.options.display.float_format = '{:,.2f} %'.format\n(df.isnull().sum()/len(df))*100","metadata":{"execution":{"iopub.execute_input":"2022-06-26T17:48:09.395398Z","iopub.status.busy":"2022-06-26T17:48:09.395016Z","iopub.status.idle":"2022-06-26T17:48:09.540602Z","shell.execute_reply":"2022-06-26T17:48:09.539782Z","shell.execute_reply.started":"2022-06-26T17:48:09.395367Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.options.display.float_format = '{:,.2f}'.format","metadata":{"execution":{"iopub.execute_input":"2022-06-26T17:48:12.941490Z","iopub.status.busy":"2022-06-26T17:48:12.941096Z","iopub.status.idle":"2022-06-26T17:48:12.948274Z","shell.execute_reply":"2022-06-26T17:48:12.945292Z","shell.execute_reply.started":"2022-06-26T17:48:12.941457Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Visualization <a class=\"anchor\" id=\"sec2\"></a>","metadata":{}},{"cell_type":"markdown","source":"### Missing Value Plot <a class=\"anchor\" id=\"sec2.1\"></a>","metadata":{}},{"cell_type":"code","source":"import missingno as msno","metadata":{"execution":{"iopub.execute_input":"2022-06-26T17:48:15.188028Z","iopub.status.busy":"2022-06-26T17:48:15.187004Z","iopub.status.idle":"2022-06-26T17:48:15.201079Z","shell.execute_reply":"2022-06-26T17:48:15.199965Z","shell.execute_reply.started":"2022-06-26T17:48:15.187988Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msno.matrix(df,labels=[df.columns],figsize=(30,16),fontsize=12)","metadata":{"execution":{"iopub.execute_input":"2022-06-26T18:09:56.178561Z","iopub.status.busy":"2022-06-26T18:09:56.178150Z","iopub.status.idle":"2022-06-26T18:10:28.558475Z","shell.execute_reply":"2022-06-26T18:10:28.557376Z","shell.execute_reply.started":"2022-06-26T18:09:56.178530Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Checking the data distribution of each Continuous variable  <a class=\"anchor\" id=\"sec2.2\"></a>","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(18, 18))\nfor i, col in enumerate(df.select_dtypes(include=['float64']).columns):\n    plt.rcParams['axes.facecolor'] = 'black'\n    ax = plt.subplot(5,5, i+1)\n    sns.histplot(data=df, x=col, ax=ax,color='red',kde=True)\nplt.suptitle('Data distribution of continuous variables')\nplt.tight_layout()","metadata":{"execution":{"iopub.execute_input":"2022-06-26T17:53:12.485445Z","iopub.status.busy":"2022-06-26T17:53:12.485140Z","iopub.status.idle":"2022-06-26T17:57:43.369464Z","shell.execute_reply":"2022-06-26T17:57:43.368351Z","shell.execute_reply.started":"2022-06-26T17:53:12.485417Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we can see that there are a lot of attributes which are positively or negatively distributed.so we will use power transformation to make these attributes symmetrical.","metadata":{}},{"cell_type":"markdown","source":"### Heatmap <a class=\"anchor\" id=\"sec2.3\"></a>","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(18,18))\nsns.heatmap(df.corr(),annot=True)\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2022-06-26T18:01:04.115266Z","iopub.status.busy":"2022-06-26T18:01:04.114802Z","iopub.status.idle":"2022-06-26T18:01:14.516078Z","shell.execute_reply":"2022-06-26T18:01:14.515106Z","shell.execute_reply.started":"2022-06-26T18:01:04.115230Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Analysing Categorical features <a class=\"anchor\" id=\"sec2.4\"></a>","metadata":{}},{"cell_type":"markdown","source":"#### Failure","metadata":{}},{"cell_type":"code","source":"target_var=pd.crosstab(index=df['failure'],columns='% observations')\nplt.pie(target_var['% observations'],labels=target_var['% observations'].index,autopct='%.0f%%')\nplt.title('Failure')\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_var=['product_code','attribute_0', 'attribute_1','attribute_2', 'attribute_3']\ntrain_val=[]\ntest_val=[]\nfor column in df[cat_var].columns:\n    train_val.append(pd.crosstab(index=df[column], columns='per_obs', normalize='columns')*100)\nfor column in test[cat_var].columns:\n    test_val.append(pd.crosstab(index=test[column], columns='per_obs', normalize='columns')*100)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### pie Chart for all categorical variables in the training dataset","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(18, 18),dpi=100)\nfor i in range(0,len(train_val)):\n    ax = plt.subplot(3,2,i+1)\n    ax.pie(train_val[i].per_obs,labels=train_val[i].index,autopct='%.0f%%')\n    ax.set_title(train_val[i].index.name)\nplt.suptitle('pie Chart for all categorical variables in the training dataset')\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Pie Chart for all categorical variables in the testing dataset","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(18, 18),dpi=100)\nfor i in range(0,len(test_val)):\n    ax = plt.subplot(3,2,i+1)\n    ax.pie(test_val[i].per_obs,labels=test_val[i].index,autopct='%.0f%%')\n    ax.set_title(test_val[i].index.name)\nplt.suptitle('pie Chart for all categorical variables in the testing dataset')\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Product_code","metadata":{}},{"cell_type":"code","source":"print(df['product_code'].unique())\nprint(test['product_code'].unique())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"here we can see that in both the training and testing dataset have different unique values.so, we will delete this feature from both the datasets","metadata":{}},{"cell_type":"code","source":"del df['product_code']\ndel test['product_code']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Attribute_0","metadata":{}},{"cell_type":"code","source":"print(df['attribute_0'].unique())\nprint(test['attribute_0'].unique())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Both the dataset have same unique values in each column.we will encode this feature","metadata":{}},{"cell_type":"code","source":"df['attribute_0']=df['attribute_0'].map({'material_7':0,'material_5':1})\ntest['attribute_0']=test['attribute_0'].map({'material_7':0,'material_5':1})","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Attribute_1","metadata":{}},{"cell_type":"code","source":"print(df['attribute_1'].unique())\nprint(test['attribute_1'].unique())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"here we can see that in both the training and testing dataset have different unique values.so, we will delete this feature from both the datasets.","metadata":{}},{"cell_type":"code","source":"del df['attribute_1']\ndel test['attribute_1']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Attribute_2","metadata":{}},{"cell_type":"code","source":"print(df['attribute_2'].unique())\nprint(test['attribute_2'].unique())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"here we can see that in both the training and testing dataset have different unique values.so, we will delete this feature from both the datasets.","metadata":{}},{"cell_type":"code","source":"del df['attribute_2']\ndel test['attribute_2']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Attribute_3","metadata":{}},{"cell_type":"code","source":"print(df['attribute_3'].unique())\nprint(test['attribute_3'].unique())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"here we can see that in both the training and testing dataset have different unique values.so, we will delete this feature from both the datasets.","metadata":{}},{"cell_type":"code","source":"del df['attribute_3']\ndel test['attribute_3']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train and Test split","metadata":{}},{"cell_type":"code","source":"X_train=df.iloc[:,1:-1]\nX_test=test.iloc[:,1:]\ny_train=df.iloc[:,-1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modeling <a class=\"anchor\" id=\"sec3\"></a>","metadata":{}},{"cell_type":"markdown","source":"### Simple Imputer for filling missing values<a class=\"anchor\" id=\"sec3.1\"></a>","metadata":{}},{"cell_type":"markdown","source":"Most of the features are symmetrically distributed so we will replace missing values with mean value","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imputer = SimpleImputer(missing_values=np.nan, strategy='mean')\nX_train= imputer.fit_transform(X_train)\nX_test=imputer.transform(X_test)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Applying ANN on training dataset <a class=\"anchor\" id=\"sec3.2\"></a>","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ann = tf.keras.models.Sequential()\nann.add(tf.keras.layers.Dense(units=6, activation='relu'))\nann.add(tf.keras.layers.Dense(units=6, activation='relu'))\nann.add(tf.keras.layers.Dense(units=1, activation='sigmoid'))\nann.compile(optimizer = 'adam', loss = 'binary_crossentropy', metrics = ['accuracy'])\nann.fit(X_train,y_train,verbose=2,batch_size=32,epochs=15)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Precting Values from the test dataset <a class=\"anchor\" id=\"sec3.3\"></a>","metadata":{}},{"cell_type":"code","source":"y_pred=ann.predict(X_test)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Importing Submission file <a class=\"anchor\" id=\"sec4\"></a>","metadata":{}},{"cell_type":"code","source":"sub=pd.read_csv('../input/tabular-playground-series-aug-2022/sample_submission.csv')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['failure']=y_pred","metadata":{"execution":{"iopub.execute_input":"2022-06-21T07:41:38.599391Z","iopub.status.busy":"2022-06-21T07:41:38.598964Z","iopub.status.idle":"2022-06-21T07:41:38.640552Z","shell.execute_reply":"2022-06-21T07:41:38.639913Z","shell.execute_reply.started":"2022-06-21T07:41:38.599357Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('final_submission5.csv',index=False)","metadata":{},"execution_count":null,"outputs":[]}]}