{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook is me just exploring some random ideas. Most have provided no useful features with meaningful correlation to the failure target.","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-13T16:32:32.651059Z","iopub.execute_input":"2022-08-13T16:32:32.652215Z","iopub.status.idle":"2022-08-13T16:32:32.657434Z","shell.execute_reply.started":"2022-08-13T16:32:32.652162Z","shell.execute_reply":"2022-08-13T16:32:32.656491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ntrain_original = train\ntest = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:01:10.866131Z","iopub.execute_input":"2022-08-13T17:01:10.866540Z","iopub.status.idle":"2022-08-13T17:01:11.103457Z","shell.execute_reply.started":"2022-08-13T17:01:10.866508Z","shell.execute_reply":"2022-08-13T17:01:11.102560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:01:12.261599Z","iopub.execute_input":"2022-08-13T17:01:12.262031Z","iopub.status.idle":"2022-08-13T17:01:12.292493Z","shell.execute_reply.started":"2022-08-13T17:01:12.261995Z","shell.execute_reply":"2022-08-13T17:01:12.291509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:32:32.924263Z","iopub.execute_input":"2022-08-13T16:32:32.924991Z","iopub.status.idle":"2022-08-13T16:32:32.947344Z","shell.execute_reply.started":"2022-08-13T16:32:32.924924Z","shell.execute_reply":"2022-08-13T16:32:32.946106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.fillna(0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:32:32.948895Z","iopub.execute_input":"2022-08-13T16:32:32.949727Z","iopub.status.idle":"2022-08-13T16:32:32.960610Z","shell.execute_reply.started":"2022-08-13T16:32:32.949691Z","shell.execute_reply":"2022-08-13T16:32:32.959107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_corrs():\n    train.corr()['failure'].sort_values(ascending=False)[1:].plot(kind='bar', figsize=(10, 5))\n\ntrain_corrs()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:32:32.964404Z","iopub.execute_input":"2022-08-13T16:32:32.965455Z","iopub.status.idle":"2022-08-13T16:32:33.325548Z","shell.execute_reply.started":"2022-08-13T16:32:32.965416Z","shell.execute_reply":"2022-08-13T16:32:33.324435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Loading is by far the most correlated feature with a base correlation of 0.129089","metadata":{}},{"cell_type":"code","source":"train['rounded_loading'] = train['loading'].round(-1)\ntrain_corrs()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:32:33.327949Z","iopub.execute_input":"2022-08-13T16:32:33.329283Z","iopub.status.idle":"2022-08-13T16:32:33.737727Z","shell.execute_reply.started":"2022-08-13T16:32:33.329233Z","shell.execute_reply":"2022-08-13T16:32:33.736341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Rounding has no effect. Extreme cases decrease the correlation.","metadata":{}},{"cell_type":"code","source":"train['loading_bins'] = pd.cut(train['loading'], 7, labels=False)\ntrain_corrs()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:32:33.740942Z","iopub.execute_input":"2022-08-13T16:32:33.741784Z","iopub.status.idle":"2022-08-13T16:32:34.141409Z","shell.execute_reply.started":"2022-08-13T16:32:33.741730Z","shell.execute_reply":"2022-08-13T16:32:34.140602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"float_cols = [col for col in train_original.columns if train_original[col].dtype == 'float64']\n\nbinned = pd.DataFrame()\nfor col in float_cols:\n    binned[col] = pd.cut(train_original[col], 7, labels=False)\n\nbinned['failure'] = train_original['failure']\nbinned.corr()['failure']","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:32:34.142486Z","iopub.execute_input":"2022-08-13T16:32:34.146344Z","iopub.status.idle":"2022-08-13T16:32:34.224434Z","shell.execute_reply.started":"2022-08-13T16:32:34.146302Z","shell.execute_reply":"2022-08-13T16:32:34.223252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.corr()['failure']","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:32:34.226000Z","iopub.execute_input":"2022-08-13T16:32:34.227081Z","iopub.status.idle":"2022-08-13T16:32:34.290040Z","shell.execute_reply.started":"2022-08-13T16:32:34.227047Z","shell.execute_reply":"2022-08-13T16:32:34.288637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Binning float64 columns provided no improvements.","metadata":{}},{"cell_type":"code","source":"multiplied = pd.DataFrame()\nfor col in float_cols:\n    multiplied[col] = train['loading'] * train[col]\n\nmultiplied['failure'] = train['failure']\nmultiplied.corr()['failure']","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:32:34.292214Z","iopub.execute_input":"2022-08-13T16:32:34.292576Z","iopub.status.idle":"2022-08-13T16:32:34.348698Z","shell.execute_reply.started":"2022-08-13T16:32:34.292543Z","shell.execute_reply":"2022-08-13T16:32:34.347468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"exp2 = pd.DataFrame()\nfor col in float_cols:\n    exp2[col] = train[col]**2 + (2 * train[col]) + 1\n\nexp2['failure'] = train['failure']\nexp2.corr()['failure']","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:36:13.377309Z","iopub.execute_input":"2022-08-13T16:36:13.377713Z","iopub.status.idle":"2022-08-13T16:36:13.454823Z","shell.execute_reply.started":"2022-08-13T16:36:13.377682Z","shell.execute_reply":"2022-08-13T16:36:13.453541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the material columns, take the material id as an int?","metadata":{}},{"cell_type":"code","source":"temp = pd.DataFrame()\ntemp['attribute_0'] = train['attribute_0'].str[-1].astype('int')\ntemp['attribute_1'] = train['attribute_1'].str[-1].astype('int')\ntemp['failure'] = train['failure']\ntemp.corr()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:48:48.634727Z","iopub.execute_input":"2022-08-13T16:48:48.635537Z","iopub.status.idle":"2022-08-13T16:48:48.717057Z","shell.execute_reply.started":"2022-08-13T16:48:48.635486Z","shell.execute_reply":"2022-08-13T16:48:48.715756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Using m0 to m2 as volumes","metadata":{}},{"cell_type":"code","source":"# basic cuboid\nv = train['measurement_0'] * train['measurement_1'] * train['measurement_2']\nv.corr(train['failure'])","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:03:03.879090Z","iopub.execute_input":"2022-08-13T17:03:03.879491Z","iopub.status.idle":"2022-08-13T17:03:03.889407Z","shell.execute_reply.started":"2022-08-13T17:03:03.879459Z","shell.execute_reply":"2022-08-13T17:03:03.888440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# triangular prism?? v = 0.5(bl)h\nv = v * 0.5\nv.corr(train['failure'])","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:05:53.115790Z","iopub.execute_input":"2022-08-13T17:05:53.116225Z","iopub.status.idle":"2022-08-13T17:05:53.127995Z","shell.execute_reply.started":"2022-08-13T17:05:53.116188Z","shell.execute_reply":"2022-08-13T17:05:53.126679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As expected","metadata":{}},{"cell_type":"code","source":"m_cols = [f'measurement_{num}' for num in range(18)]\n\nmeasurements = train[m_cols]\nmeasurements","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:54:18.090973Z","iopub.execute_input":"2022-08-13T16:54:18.091363Z","iopub.status.idle":"2022-08-13T16:54:18.130489Z","shell.execute_reply.started":"2022-08-13T16:54:18.091333Z","shell.execute_reply":"2022-08-13T16:54:18.129611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Calculate the difference between consecutive measurements etc.","metadata":{}},{"cell_type":"code","source":"for idx, c in enumerate(m_cols[1:]):\n    avg = (measurements[c] + measurements[m_cols[idx]]) / 2\n    print(c, avg.mean())","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:08:49.459442Z","iopub.execute_input":"2022-08-13T17:08:49.460008Z","iopub.status.idle":"2022-08-13T17:08:49.482140Z","shell.execute_reply.started":"2022-08-13T17:08:49.459940Z","shell.execute_reply":"2022-08-13T17:08:49.480898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It is again that clear that m0 -> m2 are distinct, along with m17.","metadata":{}},{"cell_type":"markdown","source":"numpy stat methods","metadata":{}},{"cell_type":"code","source":"agg_cols = m_cols[3:-1]\ntrain['m_avg'] = np.mean(train[agg_cols], axis=1)\ntrain['m_std'] = np.std(train[agg_cols], axis=1)\ntrain['m_var'] = np.var(train[agg_cols], axis=1)\ntrain['m_25q'] = np.percentile(train[agg_cols], q=25, axis=1)\ntrain['m_50q'] = np.percentile(train[agg_cols], q=50, axis=1)\ntrain['m_75q'] = np.percentile(train[agg_cols], q=75, axis=1)\ntrain_corrs()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:31:18.438878Z","iopub.execute_input":"2022-08-13T17:31:18.439310Z","iopub.status.idle":"2022-08-13T17:31:18.912490Z","shell.execute_reply.started":"2022-08-13T17:31:18.439274Z","shell.execute_reply":"2022-08-13T17:31:18.911325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['loading_perc'] = train['loading'] / train['loading'].max()\ntrain_corrs()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:28:43.440464Z","iopub.execute_input":"2022-08-13T17:28:43.440858Z","iopub.status.idle":"2022-08-13T17:28:43.862699Z","shell.execute_reply.started":"2022-08-13T17:28:43.440824Z","shell.execute_reply":"2022-08-13T17:28:43.861448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import stats\n\ntrain['m_skew'] = stats.skew(train[agg_cols], axis=1)\ntrain['m_kurtosis'] = stats.kurtosis(train[agg_cols], axis=1)\ntrain['m_entropy'] = stats.entropy(train[agg_cols], axis=1)\ntrain_corrs()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:35:19.124142Z","iopub.execute_input":"2022-08-13T17:35:19.124556Z","iopub.status.idle":"2022-08-13T17:35:19.574919Z","shell.execute_reply.started":"2022-08-13T17:35:19.124512Z","shell.execute_reply":"2022-08-13T17:35:19.573813Z"},"trusted":true},"execution_count":null,"outputs":[]}]}