{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport category_encoders\nfrom scipy import stats","metadata":{"execution":{"iopub.status.busy":"2022-08-04T05:47:49.939588Z","iopub.execute_input":"2022-08-04T05:47:49.939990Z","iopub.status.idle":"2022-08-04T05:47:49.945951Z","shell.execute_reply.started":"2022-08-04T05:47:49.939959Z","shell.execute_reply":"2022-08-04T05:47:49.944491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Read datasets**","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/train.csv', index_col='id')\ntest = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/test.csv', index_col='id')\nsubmit = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/sample_submission.csv')\nprint('Train:', train.shape)\nprint('Test:', test.shape)\nprint('Submit:', submit.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T04:57:19.708599Z","iopub.execute_input":"2022-08-04T04:57:19.708996Z","iopub.status.idle":"2022-08-04T04:57:19.997610Z","shell.execute_reply.started":"2022-08-04T04:57:19.708964Z","shell.execute_reply":"2022-08-04T04:57:19.996251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T04:57:23.222166Z","iopub.execute_input":"2022-08-04T04:57:23.222874Z","iopub.status.idle":"2022-08-04T04:57:23.257608Z","shell.execute_reply.started":"2022-08-04T04:57:23.222825Z","shell.execute_reply":"2022-08-04T04:57:23.256386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T04:57:25.676525Z","iopub.execute_input":"2022-08-04T04:57:25.677283Z","iopub.status.idle":"2022-08-04T04:57:25.702646Z","shell.execute_reply.started":"2022-08-04T04:57:25.677245Z","shell.execute_reply":"2022-08-04T04:57:25.701898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:26:09.893420Z","iopub.execute_input":"2022-08-02T17:26:09.893834Z","iopub.status.idle":"2022-08-02T17:26:09.912940Z","shell.execute_reply.started":"2022-08-02T17:26:09.893801Z","shell.execute_reply":"2022-08-02T17:26:09.911706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:48:47.019911Z","iopub.execute_input":"2022-08-02T17:48:47.020346Z","iopub.status.idle":"2022-08-02T17:48:47.039002Z","shell.execute_reply.started":"2022-08-02T17:48:47.020310Z","shell.execute_reply":"2022-08-02T17:48:47.037611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_features = [col for col in train.columns if train[col].dtypes == object]\nint_features = [col for col in train.columns if (train[col].dtypes == 'int64' and col != 'failure')]\nfloat_features = [col for col in train.columns if train[col].dtypes == 'float64']\nprint('Checking:', len(cat_features) + len(int_features) + len(float_features) == len(train.columns)-1)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T04:57:30.438706Z","iopub.execute_input":"2022-08-04T04:57:30.439840Z","iopub.status.idle":"2022-08-04T04:57:30.452659Z","shell.execute_reply.started":"2022-08-04T04:57:30.439795Z","shell.execute_reply":"2022-08-04T04:57:30.451501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr = train.drop(columns='failure')\ntr['tr/te'] = 'train'\nte = test.copy()\nte['tr/te'] = 'test'\ndf = pd.concat([tr, te], axis=0, join='inner')\nlen(df)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T04:57:32.845159Z","iopub.execute_input":"2022-08-04T04:57:32.845635Z","iopub.status.idle":"2022-08-04T04:57:32.885389Z","shell.execute_reply.started":"2022-08-04T04:57:32.845591Z","shell.execute_reply":"2022-08-04T04:57:32.884593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Checking nulls**","metadata":{}},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:26:16.801455Z","iopub.execute_input":"2022-08-02T17:26:16.801905Z","iopub.status.idle":"2022-08-02T17:26:16.820221Z","shell.execute_reply.started":"2022-08-02T17:26:16.801871Z","shell.execute_reply":"2022-08-02T17:26:16.819281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:26:20.071560Z","iopub.execute_input":"2022-08-02T17:26:20.071956Z","iopub.status.idle":"2022-08-02T17:26:20.083914Z","shell.execute_reply.started":"2022-08-02T17:26:20.071924Z","shell.execute_reply":"2022-08-02T17:26:20.083117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Missing categorical values:', train[cat_features].isnull().sum().sum())\nprint('Missing integer values:', train[int_features].isnull().sum().sum())\nprint('Missing float values:', train[float_features].isnull().sum().sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:52:30.340461Z","iopub.execute_input":"2022-08-02T17:52:30.340877Z","iopub.status.idle":"2022-08-02T17:52:30.360369Z","shell.execute_reply.started":"2022-08-02T17:52:30.340846Z","shell.execute_reply":"2022-08-02T17:52:30.359498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Missing categorical values:', test[cat_features].isnull().sum().sum())\nprint('Missing integer values:', test[int_features].isnull().sum().sum())\nprint('Missing float values:', test[float_features].isnull().sum().sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:52:30.488648Z","iopub.execute_input":"2022-08-02T17:52:30.489306Z","iopub.status.idle":"2022-08-02T17:52:30.508878Z","shell.execute_reply.started":"2022-08-02T17:52:30.489256Z","shell.execute_reply":"2022-08-02T17:52:30.506941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_tr = train[float_features].isnull().sum(axis=1)\nsumma = 0\nfor i in sorted(temp_tr.unique()):\n    n_miss = len(train[temp_tr == i])\n    print(i, 'missing values:', n_miss, '    \\tPercent:', round(100*n_miss/len(train),2))\n    summa += n_miss*i\nprint('Checking:', summa == train[float_features].isnull().sum().sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-04T04:57:38.675414Z","iopub.execute_input":"2022-08-04T04:57:38.675804Z","iopub.status.idle":"2022-08-04T04:57:38.699637Z","shell.execute_reply.started":"2022-08-04T04:57:38.675772Z","shell.execute_reply":"2022-08-04T04:57:38.698769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_te = test[float_features].isnull().sum(axis=1)\nsumma = 0\nfor i in sorted(temp_te.unique()):\n    n_miss = len(test[temp_te == i])\n    print(i, 'missing values:', n_miss, '   \\tPercent:', round(100*n_miss/len(test),2))\n    summa += n_miss*i\nprint('Checking:', summa == test[float_features].isnull().sum().sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-04T04:57:41.102896Z","iopub.execute_input":"2022-08-04T04:57:41.103257Z","iopub.status.idle":"2022-08-04T04:57:41.123479Z","shell.execute_reply.started":"2022-08-04T04:57:41.103228Z","shell.execute_reply":"2022-08-04T04:57:41.122770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Conclusion 1:** float_features columns have some missing values, measurement_17 column has maximum nulls, but less than 9% of all data. About a half of the float values have no missing values. Maximum 6 missing values in one row.","metadata":{}},{"cell_type":"markdown","source":"# **Statistical Analysis**","metadata":{}},{"cell_type":"markdown","source":"**Integer features**","metadata":{}},{"cell_type":"code","source":"train[int_features].describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:50:51.508789Z","iopub.execute_input":"2022-08-03T04:50:51.509246Z","iopub.status.idle":"2022-08-03T04:50:51.552936Z","shell.execute_reply.started":"2022-08-03T04:50:51.509199Z","shell.execute_reply":"2022-08-03T04:50:51.551803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[int_features].describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:15:26.838066Z","iopub.execute_input":"2022-08-03T08:15:26.838735Z","iopub.status.idle":"2022-08-03T08:15:26.873355Z","shell.execute_reply.started":"2022-08-03T08:15:26.838699Z","shell.execute_reply":"2022-08-03T08:15:26.872111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(3,2, sharex=False, sharey=False, figsize=(12,8), constrained_layout=True)\nfig.suptitle('Integer features (boxplot)', fontsize=25)\n\nfor i, col in enumerate(int_features):\n    sns.boxplot(data=df[int_features + [df.columns[-1]]], ax=ax[i//2,i%2], y=col, x='tr/te')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:00:30.830545Z","iopub.execute_input":"2022-08-03T16:00:30.830981Z","iopub.status.idle":"2022-08-03T16:00:32.039990Z","shell.execute_reply.started":"2022-08-03T16:00:30.830947Z","shell.execute_reply":"2022-08-03T16:00:32.038747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(3,2, sharex=False, sharey=False, figsize=(14,10), constrained_layout=True)\nfig.suptitle('Integer features (countplot)', fontsize=25)\n\nfor i, col in enumerate(int_features):\n    sns.countplot(data=df[int_features + [df.columns[-1]]], ax=ax[i//2,i%2], x=col, hue='tr/te')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:00:32.912036Z","iopub.execute_input":"2022-08-03T16:00:32.913162Z","iopub.status.idle":"2022-08-03T16:00:35.491185Z","shell.execute_reply.started":"2022-08-03T16:00:32.913119Z","shell.execute_reply":"2022-08-03T16:00:35.489973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Float features**","metadata":{}},{"cell_type":"code","source":"train[float_features].describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:00:35.642419Z","iopub.execute_input":"2022-08-03T16:00:35.643639Z","iopub.status.idle":"2022-08-03T16:00:35.753715Z","shell.execute_reply.started":"2022-08-03T16:00:35.643584Z","shell.execute_reply":"2022-08-03T16:00:35.752523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[float_features].describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:00:42.597498Z","iopub.execute_input":"2022-08-03T16:00:42.597973Z","iopub.status.idle":"2022-08-03T16:00:42.698873Z","shell.execute_reply.started":"2022-08-03T16:00:42.597936Z","shell.execute_reply":"2022-08-03T16:00:42.697594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(4,4, sharex=False, sharey=False, figsize=(14,12), constrained_layout=True)\nfig.suptitle('Float features', fontsize=25)\n\nfor i, col in enumerate(float_features):\n    sns.boxplot(data=df[float_features + [df.columns[-1]]], ax=ax[i//4,i%4], y=col, x='tr/te')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:00:46.022550Z","iopub.execute_input":"2022-08-03T16:00:46.023004Z","iopub.status.idle":"2022-08-03T16:00:49.155323Z","shell.execute_reply.started":"2022-08-03T16:00:46.022968Z","shell.execute_reply":"2022-08-03T16:00:49.153814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Categorical features**","metadata":{}},{"cell_type":"code","source":"train[cat_features].describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:00:49.944598Z","iopub.execute_input":"2022-08-03T16:00:49.945040Z","iopub.status.idle":"2022-08-03T16:00:49.981276Z","shell.execute_reply.started":"2022-08-03T16:00:49.945005Z","shell.execute_reply":"2022-08-03T16:00:49.979597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[cat_features].describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:00:50.933102Z","iopub.execute_input":"2022-08-03T16:00:50.933927Z","iopub.status.idle":"2022-08-03T16:00:50.963763Z","shell.execute_reply.started":"2022-08-03T16:00:50.933885Z","shell.execute_reply":"2022-08-03T16:00:50.962835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(2,2, sharex=False, sharey=False, figsize=(12,8), constrained_layout=True)\nfig.suptitle('Categorical features', fontsize=25)\n\nfor i, col in enumerate(cat_features):\n    sns.countplot(data=df[cat_features + [df.columns[-1]]], ax=ax[i//2,i%2], x=col, hue='tr/te')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:00:53.966207Z","iopub.execute_input":"2022-08-03T16:00:53.967051Z","iopub.status.idle":"2022-08-03T16:00:55.313261Z","shell.execute_reply.started":"2022-08-03T16:00:53.967010Z","shell.execute_reply":"2022-08-03T16:00:55.311975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Unique Values**","metadata":{}},{"cell_type":"code","source":"for column in train.columns:\n    temp = train[column].unique()\n    print(column, ' \\t\\t(', len(temp), '): ', temp, sep='')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:00:58.996651Z","iopub.execute_input":"2022-08-03T16:00:58.997655Z","iopub.status.idle":"2022-08-03T16:00:59.033199Z","shell.execute_reply.started":"2022-08-03T16:00:58.997604Z","shell.execute_reply":"2022-08-03T16:00:59.031883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for column in test.columns:\n    temp = test[column].unique()\n    print(column, ' \\t\\t(', len(temp), '): ', temp, sep='')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:01:02.259095Z","iopub.execute_input":"2022-08-03T16:01:02.260530Z","iopub.status.idle":"2022-08-03T16:01:02.292810Z","shell.execute_reply.started":"2022-08-03T16:01:02.260473Z","shell.execute_reply":"2022-08-03T16:01:02.291385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Conclusion 2:** All the attribute_0-3 look like categorical features. Measurement_0-2 can be ordinal. All the number features are greater than or equal to zero. Statistical variables of the train and test float features are close each other. They should have the same distributions. ","metadata":{}},{"cell_type":"markdown","source":"**Some other visualizations**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8,6))\nplt.title('Checking a Balance of target', fontsize=16)\nsns.countplot(x='failure', data=train)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:01:07.613608Z","iopub.execute_input":"2022-08-03T16:01:07.614157Z","iopub.status.idle":"2022-08-03T16:01:07.847416Z","shell.execute_reply.started":"2022-08-03T16:01:07.614109Z","shell.execute_reply":"2022-08-03T16:01:07.846385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Conclusion 3:** The dataset is unbalanced.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(4,2, sharex=False, sharey=False, figsize=(14,12), constrained_layout=True)\nfig.suptitle('Frequency Analysis', fontsize=25)\n\ntemp_features = cat_features + int_features\nfor i, col in enumerate(temp_features):\n    sns.countplot(data=train[temp_features + ['failure']], ax=ax[i//2,i%2], x=col, hue='failure')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:01:09.983389Z","iopub.execute_input":"2022-08-03T16:01:09.984230Z","iopub.status.idle":"2022-08-03T16:01:12.580364Z","shell.execute_reply.started":"2022-08-03T16:01:09.984188Z","shell.execute_reply":"2022-08-03T16:01:12.579104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Conclusion 4:** Categorical features do not change the proportion of unbalancing.","metadata":{}},{"cell_type":"code","source":"enc_hot = category_encoders.one_hot.OneHotEncoder(cols=cat_features+int_features[:2]).fit(train)\ntrain_hot = enc_hot.transform(train)\n\nenc_ord = category_encoders.ordinal.OrdinalEncoder(cols=cat_features+int_features[:2],\n    mapping=[{'col': 'product_code', 'mapping': {None: 0, 'A': 1, 'B': 2, 'C': 3, 'D': 4, 'E': 5}},\n             {'col': 'attribute_0', 'mapping': {None: 0, 'material_5': 5, 'material_7': 7}},\n             {'col': 'attribute_1', 'mapping': {None: 0, 'material_5': 5, 'material_6': 6, 'material_8': 8}}]).fit(train)\ntrain_ord = enc_ord.transform(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T04:58:21.333577Z","iopub.execute_input":"2022-08-04T04:58:21.334655Z","iopub.status.idle":"2022-08-04T04:58:21.724882Z","shell.execute_reply.started":"2022-08-04T04:58:21.334617Z","shell.execute_reply":"2022-08-04T04:58:21.723704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_hot.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T04:58:25.943433Z","iopub.execute_input":"2022-08-04T04:58:25.943875Z","iopub.status.idle":"2022-08-04T04:58:25.964426Z","shell.execute_reply.started":"2022-08-04T04:58:25.943837Z","shell.execute_reply":"2022-08-04T04:58:25.963246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_hot.corr()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T18:08:27.223503Z","iopub.execute_input":"2022-08-03T18:08:27.223914Z","iopub.status.idle":"2022-08-03T18:08:27.379727Z","shell.execute_reply.started":"2022-08-03T18:08:27.223878Z","shell.execute_reply":"2022-08-03T18:08:27.378685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,10))\nsns.heatmap(train_hot.corr())\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T17:51:15.722645Z","iopub.execute_input":"2022-08-03T17:51:15.723476Z","iopub.status.idle":"2022-08-03T17:51:16.906090Z","shell.execute_reply.started":"2022-08-03T17:51:15.723434Z","shell.execute_reply":"2022-08-03T17:51:16.904754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"enc_hot_test = category_encoders.one_hot.OneHotEncoder(cols=cat_features+int_features[:2]).fit(test)\ntest_hot = enc_hot_test.transform(test)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T04:58:38.526199Z","iopub.execute_input":"2022-08-04T04:58:38.526613Z","iopub.status.idle":"2022-08-04T04:58:38.752092Z","shell.execute_reply.started":"2022-08-04T04:58:38.526577Z","shell.execute_reply":"2022-08-04T04:58:38.750890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_hot.corr()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T17:51:30.745959Z","iopub.execute_input":"2022-08-03T17:51:30.746392Z","iopub.status.idle":"2022-08-03T17:51:30.869187Z","shell.execute_reply.started":"2022-08-03T17:51:30.746358Z","shell.execute_reply":"2022-08-03T17:51:30.867759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,10))\nsns.heatmap(test_hot.corr())\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T18:08:49.275552Z","iopub.execute_input":"2022-08-03T18:08:49.275913Z","iopub.status.idle":"2022-08-03T18:08:50.245281Z","shell.execute_reply.started":"2022-08-03T18:08:49.275885Z","shell.execute_reply":"2022-08-03T18:08:50.243938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ord.corr()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:53:19.521631Z","iopub.execute_input":"2022-08-03T16:53:19.522091Z","iopub.status.idle":"2022-08-03T16:53:19.620958Z","shell.execute_reply.started":"2022-08-03T16:53:19.522055Z","shell.execute_reply":"2022-08-03T16:53:19.619529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,10))\nsns.heatmap(train_ord.corr())\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T17:52:39.001539Z","iopub.execute_input":"2022-08-03T17:52:39.001967Z","iopub.status.idle":"2022-08-03T17:52:39.892716Z","shell.execute_reply.started":"2022-08-03T17:52:39.001932Z","shell.execute_reply":"2022-08-03T17:52:39.891363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Conclusion 5:** Correlation heatmap (OHE) shows the linear dependence between the product_code and the attributes. It means we can just drop product_code. Anyway the product_code is helpless for us because train and test have different values. We can also see the next linear dependences:\n\nattribute_0 and attribute_2_2 (material 8), \n\nattribute_1_3 (material 6) and attribute_3_4 (material 9), \n\nattribute_2_1 (material 9) and attribute_3_1 (material 5). \n\nIf we look at correlations in test set we will see the linear dependence between:\n\nattribute_0 and attribute_1_1 (material 6),\n\nattribute_1_2 (material 7) and attribute_2_3 (material 7) and attribute_3_3 (material 9),\n\nattribute 2_1 (material 6) and attribute_3_1 (material 4),\n\nattribute_1_3 (material 5) and attribute_3_4 (material 5).\n\nWe have to untangle these relationships in order to use attributes as features in our models because train and test can have different materials in their attributes. For a start the attribute_0 looks quite well. It has the same values in train and test and the linear correlation with other attributes.\n\nThe measurements_3-9 don't have significant correlation between other features excluding measurement_17. The other features have weak correlations with each other and the measurements_0-2 a bit stronger. The target has some correlation with the loading.","metadata":{}},{"cell_type":"code","source":"%%time\nsns.pairplot(df.iloc[:,6:], hue='tr/te')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T05:02:49.739220Z","iopub.execute_input":"2022-08-04T05:02:49.739626Z","iopub.status.idle":"2022-08-04T05:17:00.733251Z","shell.execute_reply.started":"2022-08-04T05:02:49.739594Z","shell.execute_reply":"2022-08-04T05:17:00.730314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsns.pairplot(train.iloc[:,6:], hue='failure')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T05:20:40.901108Z","iopub.execute_input":"2022-08-04T05:20:40.901522Z","iopub.status.idle":"2022-08-04T05:29:25.332017Z","shell.execute_reply.started":"2022-08-04T05:20:40.901485Z","shell.execute_reply":"2022-08-04T05:29:25.329707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Resume:** we have unbalanced dataset with different types of features for a binary classification problem. Float values in our dataset also have some missings. We have disharmony between train and test in categorical features as well. Some of those categorical features have strong linear dependence between each other.","metadata":{}},{"cell_type":"markdown","source":"**What's next?**\n\n1. Filling missing values. We can remember TPS June and try different imputing techniques for this. If we untangle categorical data we can use them for imputing the measurements_10_16. The measurements_3_9 should be mean-imputed and then we can use them for imputing the measurement_17. All of those imputations should be tested with ML models.\n2. For untangling categorical features we can remember TPS July and try clustering them in order to avoid attributes and materials and get just clear numbers for different categories. Again it should be checked using ML models.\n3. We should not use accuracy for checking models or imputing and clustering techniques. We can use precision, recall, F1 and ROC AUC.","metadata":{}},{"cell_type":"markdown","source":"# To be continued ( I hope :) )","metadata":{}}]}