{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nimport plotly.express as px\nimport plotly.graph_objects as go\nwarnings.filterwarnings(\"ignore\")\nplt.style.use(\"fivethirtyeight\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-01T10:41:43.930848Z","iopub.execute_input":"2022-08-01T10:41:43.931385Z","iopub.status.idle":"2022-08-01T10:41:45.107006Z","shell.execute_reply.started":"2022-08-01T10:41:43.931339Z","shell.execute_reply":"2022-08-01T10:41:45.106059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Reading Train , Test and Submission Data","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv(\"../input/tabular-playground-series-aug-2022/train.csv\")\ntest_data = pd.read_csv(\"../input/tabular-playground-series-aug-2022/test.csv\")\nsubmission = pd.read_csv(\"../input/tabular-playground-series-aug-2022/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T09:21:30.034959Z","iopub.execute_input":"2022-08-01T09:21:30.035386Z","iopub.status.idle":"2022-08-01T09:21:30.290262Z","shell.execute_reply.started":"2022-08-01T09:21:30.035348Z","shell.execute_reply":"2022-08-01T09:21:30.289436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data","metadata":{"execution":{"iopub.status.busy":"2022-08-01T09:21:36.677056Z","iopub.execute_input":"2022-08-01T09:21:36.677576Z","iopub.status.idle":"2022-08-01T09:21:36.729552Z","shell.execute_reply.started":"2022-08-01T09:21:36.677531Z","shell.execute_reply":"2022-08-01T09:21:36.728753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:49:02.550378Z","iopub.execute_input":"2022-08-01T10:49:02.550833Z","iopub.status.idle":"2022-08-01T10:49:02.599754Z","shell.execute_reply.started":"2022-08-01T10:49:02.550796Z","shell.execute_reply":"2022-08-01T10:49:02.598666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info(show_counts=True, memory_usage=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T09:22:18.065437Z","iopub.execute_input":"2022-08-01T09:22:18.065829Z","iopub.status.idle":"2022-08-01T09:22:18.093754Z","shell.execute_reply.started":"2022-08-01T09:22:18.065798Z","shell.execute_reply":"2022-08-01T09:22:18.090769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_data.select_dtypes(\"float64\").columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:00:20.363434Z","iopub.execute_input":"2022-08-01T11:00:20.364443Z","iopub.status.idle":"2022-08-01T11:00:20.374897Z","shell.execute_reply.started":"2022-08-01T11:00:20.364402Z","shell.execute_reply":"2022-08-01T11:00:20.373800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 1. It is fairly a small dataset.It has **26570** rows and **26** columns.\n#### 2. It has 3 object columns, 7 integer columns and 16 float columns","metadata":{}},{"cell_type":"code","source":"train_data.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-08-01T09:22:42.622037Z","iopub.execute_input":"2022-08-01T09:22:42.622514Z","iopub.status.idle":"2022-08-01T09:22:42.731152Z","shell.execute_reply.started":"2022-08-01T09:22:42.622475Z","shell.execute_reply":"2022-08-01T09:22:42.729793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### There is a variation in the minimum and maximum values in the features. If we are going with any distance based algorithm then we need to standardise it so that there should be not the variation between the values","metadata":{}},{"cell_type":"code","source":"sns.countplot(x=train_data[\"failure\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T09:37:10.496929Z","iopub.execute_input":"2022-08-01T09:37:10.497435Z","iopub.status.idle":"2022-08-01T09:37:10.640175Z","shell.execute_reply.started":"2022-08-01T09:37:10.497387Z","shell.execute_reply":"2022-08-01T09:37:10.638713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### The target feature is skewed :- have more positive values then the negative values. We should be careful with this otherwise our model will lead to overfitting\n\n### We can either used a good cross-validation technique or up-sampling the neagtive class to make the samples equal in the target column","metadata":{}},{"cell_type":"code","source":"train_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T09:24:29.966535Z","iopub.execute_input":"2022-08-01T09:24:29.966925Z","iopub.status.idle":"2022-08-01T09:24:29.986723Z","shell.execute_reply.started":"2022-08-01T09:24:29.966893Z","shell.execute_reply":"2022-08-01T09:24:29.985530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### There are 16 features having missing values. We need to impute them before training or else we can go with any tree based algorithm on which missing value does not effect","metadata":{}},{"cell_type":"code","source":"list_of_missing_values = [feature for feature in train_data.columns if train_data[feature].isnull().sum() > 0]\nmissing_df = pd.DataFrame({\"Feature Name\":list_of_missing_values,\n                          \"Number of Missing Values\" : train_data[list_of_missing_values].isnull().sum()}).reset_index(drop=True)\n\npx.bar(missing_df,x=\"Feature Name\",y=\"Number of Missing Values\",\\\n       title=\"Number of Missing Values in the Columns having Missing values\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:41:54.368025Z","iopub.execute_input":"2022-08-01T10:41:54.369008Z","iopub.status.idle":"2022-08-01T10:41:55.375554Z","shell.execute_reply.started":"2022-08-01T10:41:54.368968Z","shell.execute_reply":"2022-08-01T10:41:55.374354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"object_cols = [feature for feature in train_data.columns if train_data[feature].dtypes == \"object\"]\nobject_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-01T09:26:06.643375Z","iopub.execute_input":"2022-08-01T09:26:06.644253Z","iopub.status.idle":"2022-08-01T09:26:06.651911Z","shell.execute_reply.started":"2022-08-01T09:26:06.644207Z","shell.execute_reply":"2022-08-01T09:26:06.650672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_cols = [feature for feature in train_data.columns if feature not in object_cols \n                  and feature not in [\"failure\",\"id\"]]\nnumerical_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-01T09:52:02.464326Z","iopub.execute_input":"2022-08-01T09:52:02.465385Z","iopub.status.idle":"2022-08-01T09:52:02.472686Z","shell.execute_reply.started":"2022-08-01T09:52:02.465344Z","shell.execute_reply":"2022-08-01T09:52:02.471671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data[object_cols].head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T09:27:02.284150Z","iopub.execute_input":"2022-08-01T09:27:02.284671Z","iopub.status.idle":"2022-08-01T09:27:02.297129Z","shell.execute_reply.started":"2022-08-01T09:27:02.284640Z","shell.execute_reply":"2022-08-01T09:27:02.295699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for feature in object_cols:\n    print(f\">>>>>> UNIQUE VALUES IN {feature} IS <<<<<<<<\\n\")\n    print(train_data[feature].value_counts())\n    print()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T09:29:05.205924Z","iopub.execute_input":"2022-08-01T09:29:05.206332Z","iopub.status.idle":"2022-08-01T09:29:05.219826Z","shell.execute_reply.started":"2022-08-01T09:29:05.206278Z","shell.execute_reply":"2022-08-01T09:29:05.218602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for feature in object_cols:\n    print(f\">>>>>> UNIQUE VALUES IN {feature} IS <<<<<<<<\\n\")\n    print(test_data[feature].value_counts())\n    print()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:49:33.535497Z","iopub.execute_input":"2022-08-01T10:49:33.536290Z","iopub.status.idle":"2022-08-01T10:49:33.548519Z","shell.execute_reply.started":"2022-08-01T10:49:33.536252Z","shell.execute_reply":"2022-08-01T10:49:33.547526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Now we can see from the abive output there are some production code which are present in train data and not present in test data. This feature will lead to some incorrect results as we don't know nothing about unknown features.\n\n## Ambrose discussion post --> https://www.kaggle.com/competitions/tabular-playground-series-aug-2022/discussion/341070","metadata":{}},{"cell_type":"code","source":"for feature in object_cols:\n    sns.countplot(x=feature,hue=\"failure\",data=train_data)\n    plt.title(f\"Count Plot for {feature}\")\n    plt.ylabel(\"Count of each unique values\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T09:43:11.396869Z","iopub.execute_input":"2022-08-01T09:43:11.397253Z","iopub.status.idle":"2022-08-01T09:43:12.051588Z","shell.execute_reply.started":"2022-08-01T09:43:11.397222Z","shell.execute_reply":"2022-08-01T09:43:12.050356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data[numerical_cols].head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T09:34:03.362843Z","iopub.execute_input":"2022-08-01T09:34:03.363198Z","iopub.status.idle":"2022-08-01T09:34:03.394150Z","shell.execute_reply.started":"2022-08-01T09:34:03.363166Z","shell.execute_reply":"2022-08-01T09:34:03.393016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,20))\nfor i in enumerate(numerical_cols):\n    plt.subplot(6,4,i[0]+1)\n    sns.distplot(x=train_data[i[1]])\n    plt.xlabel(i[1])\nplt.suptitle('Distributions of the Numerical Features', fontsize=20, y=1.02)    \nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:02:24.469192Z","iopub.execute_input":"2022-08-01T10:02:24.469596Z","iopub.status.idle":"2022-08-01T10:02:32.632781Z","shell.execute_reply.started":"2022-08-01T10:02:24.469564Z","shell.execute_reply":"2022-08-01T10:02:32.631606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Insight from above figure**\nMost of numerical features follow gaussian distribution. loading, measurement_0, measurement_1, measurement_2 are slightly skewed towards right. We can transformation techniques on these features and then observe the performance before and after transformation","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,8))\nsns.heatmap(data=train_data.corr())\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T09:55:20.031121Z","iopub.execute_input":"2022-08-01T09:55:20.031918Z","iopub.status.idle":"2022-08-01T09:55:20.715733Z","shell.execute_reply.started":"2022-08-01T09:55:20.031878Z","shell.execute_reply":"2022-08-01T09:55:20.714789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.corr()[\"failure\"].sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:29:12.805773Z","iopub.execute_input":"2022-08-01T11:29:12.806613Z","iopub.status.idle":"2022-08-01T11:29:12.859078Z","shell.execute_reply.started":"2022-08-01T11:29:12.806574Z","shell.execute_reply":"2022-08-01T11:29:12.858019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### There are some features which shows good relation in the estimating the target feature\n* loading\n* measurement_17\n* measurement_5\n* measurement_8\n* measurement_7","metadata":{}},{"cell_type":"markdown","source":"#### We can use this in our machine learning model to estimate the target in the test dataset\n\n\n\n\n#### Stay tuned for the modelling notebook","metadata":{}}]}