{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\n\nfrom IPython.display import display_html\n\nimport scipy.stats as stats\nimport pylab\n\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\nfrom sklearn.compose import make_column_transformer\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.pipeline import make_pipeline\n\nfrom xgboost import XGBRegressor\n\nfrom sklearn.model_selection import cross_val_score\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:54.491079Z","iopub.execute_input":"2022-08-02T17:29:54.491485Z","iopub.status.idle":"2022-08-02T17:29:54.504892Z","shell.execute_reply.started":"2022-08-02T17:29:54.491451Z","shell.execute_reply":"2022-08-02T17:29:54.502967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ndft = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')\nprint(f'Train shape {df.shape}')\nprint(f'Test shape {dft.shape}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:54.508580Z","iopub.execute_input":"2022-08-02T17:29:54.509168Z","iopub.status.idle":"2022-08-02T17:29:54.848362Z","shell.execute_reply.started":"2022-08-02T17:29:54.509114Z","shell.execute_reply":"2022-08-02T17:29:54.847047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:54.849964Z","iopub.execute_input":"2022-08-02T17:29:54.851162Z","iopub.status.idle":"2022-08-02T17:29:54.867468Z","shell.execute_reply.started":"2022-08-02T17:29:54.851116Z","shell.execute_reply":"2022-08-02T17:29:54.866384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:54.869672Z","iopub.execute_input":"2022-08-02T17:29:54.870737Z","iopub.status.idle":"2022-08-02T17:29:54.915924Z","shell.execute_reply.started":"2022-08-02T17:29:54.870656Z","shell.execute_reply":"2022-08-02T17:29:54.914371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dft.head()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-02T17:29:54.917782Z","iopub.execute_input":"2022-08-02T17:29:54.918341Z","iopub.status.idle":"2022-08-02T17:29:54.950460Z","shell.execute_reply.started":"2022-08-02T17:29:54.918301Z","shell.execute_reply":"2022-08-02T17:29:54.949467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Types of columns","metadata":{}},{"cell_type":"code","source":"float_cols = dft.select_dtypes(include='float64').columns\nint_cols = dft.select_dtypes(include='int64').columns\nstring_cols = dft.select_dtypes(include='object').columns","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:54.953614Z","iopub.execute_input":"2022-08-02T17:29:54.954937Z","iopub.status.idle":"2022-08-02T17:29:54.976057Z","shell.execute_reply.started":"2022-08-02T17:29:54.954881Z","shell.execute_reply":"2022-08-02T17:29:54.974420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Column       Train Uniques      Test Uniques')\nprint()\nfor col in int_cols:\n    print(f'{col}           {df[col].nunique()}              {dft[col].nunique()}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:54.977893Z","iopub.execute_input":"2022-08-02T17:29:54.978635Z","iopub.status.idle":"2022-08-02T17:29:54.998710Z","shell.execute_reply.started":"2022-08-02T17:29:54.978586Z","shell.execute_reply":"2022-08-02T17:29:54.997233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Column       Train Uniques      Test Uniques')\nprint()\nfor col in string_cols:\n    print(f'{col}           {df[col].nunique()}              {dft[col].nunique()}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:55.000419Z","iopub.execute_input":"2022-08-02T17:29:55.001180Z","iopub.status.idle":"2022-08-02T17:29:55.024390Z","shell.execute_reply.started":"2022-08-02T17:29:55.001133Z","shell.execute_reply":"2022-08-02T17:29:55.022967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on number of unique values I am assuming all attribute columns to be categorical","metadata":{}},{"cell_type":"code","source":"# Categorical Columns\ncat_cols = ['product_code', 'attribute_0', 'attribute_1', 'attribute_2', 'attribute_3', 'failure']\n\n# Numerical Integer Columns\nnum_int_cols = ['measurement_0', 'measurement_1', 'measurement_2']\n\n# Numerical Float Columns\nnum_float_cols = float_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:55.028271Z","iopub.execute_input":"2022-08-02T17:29:55.028678Z","iopub.status.idle":"2022-08-02T17:29:55.034931Z","shell.execute_reply.started":"2022-08-02T17:29:55.028641Z","shell.execute_reply":"2022-08-02T17:29:55.033651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizing Categorical Columns","metadata":{}},{"cell_type":"code","source":"def cat_col_unique(col):\n    plt.figure(figsize=(12,5))\n    \n    plt.subplot(1,2,1)\n    plt.title('Train Data')\n    sns.barplot(df[col].value_counts().keys(), df[col].value_counts().values)\n    \n    plt.subplot(1,2,2)\n    plt.title('Test Data')\n    sns.barplot(dft[col].value_counts().keys(), dft[col].value_counts().values)\n    \n    d1 = pd.DataFrame({col: df[col].value_counts().keys(), 'Train Percentage': df[col].value_counts().values/df.shape[0]*100})\n    d2 = pd.DataFrame({col: dft[col].value_counts().keys(), 'Test Percentage': dft[col].value_counts().values/dft.shape[0]*100})\n    \n    df1_styler = d1.style.set_table_attributes(\"style='display:inline'\").set_caption('Train Set')\n    df2_styler = d2.style.set_table_attributes(\"style='display:inline'\").set_caption('Test Set')\n\n    display_html(df1_styler._repr_html_()+df2_styler._repr_html_(), raw=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:55.036467Z","iopub.execute_input":"2022-08-02T17:29:55.037611Z","iopub.status.idle":"2022-08-02T17:29:55.051266Z","shell.execute_reply.started":"2022-08-02T17:29:55.037564Z","shell.execute_reply":"2022-08-02T17:29:55.049963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Failure (Target Variable)","metadata":{}},{"cell_type":"code","source":"sns.barplot(df['failure'].value_counts().keys(), df['failure'].value_counts().values)\npd.DataFrame({'failure': df['failure'].value_counts().keys(), 'Percentage': df['failure'].value_counts().values/df.shape[0]*100})","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:55.052674Z","iopub.execute_input":"2022-08-02T17:29:55.053635Z","iopub.status.idle":"2022-08-02T17:29:55.239369Z","shell.execute_reply.started":"2022-08-02T17:29:55.053596Z","shell.execute_reply":"2022-08-02T17:29:55.238448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Dataset is imbalanced ","metadata":{}},{"cell_type":"markdown","source":"## product_code","metadata":{}},{"cell_type":"code","source":"cat_col_unique('product_code')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:55.240678Z","iopub.execute_input":"2022-08-02T17:29:55.241201Z","iopub.status.idle":"2022-08-02T17:29:55.621455Z","shell.execute_reply.started":"2022-08-02T17:29:55.241167Z","shell.execute_reply":"2022-08-02T17:29:55.620169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## attribute_0","metadata":{}},{"cell_type":"code","source":"cat_col_unique('attribute_0')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:55.623094Z","iopub.execute_input":"2022-08-02T17:29:55.623435Z","iopub.status.idle":"2022-08-02T17:29:55.926266Z","shell.execute_reply.started":"2022-08-02T17:29:55.623405Z","shell.execute_reply":"2022-08-02T17:29:55.924929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## attribute_1","metadata":{}},{"cell_type":"code","source":"cat_col_unique('attribute_1')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:55.927956Z","iopub.execute_input":"2022-08-02T17:29:55.928372Z","iopub.status.idle":"2022-08-02T17:29:56.232126Z","shell.execute_reply.started":"2022-08-02T17:29:55.928322Z","shell.execute_reply":"2022-08-02T17:29:56.230815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## attribute_2","metadata":{}},{"cell_type":"code","source":"cat_col_unique('attribute_2')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:56.233642Z","iopub.execute_input":"2022-08-02T17:29:56.234070Z","iopub.status.idle":"2022-08-02T17:29:56.528834Z","shell.execute_reply.started":"2022-08-02T17:29:56.234032Z","shell.execute_reply":"2022-08-02T17:29:56.527570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## attribute_3","metadata":{}},{"cell_type":"code","source":"cat_col_unique('attribute_3')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:56.532055Z","iopub.execute_input":"2022-08-02T17:29:56.532581Z","iopub.status.idle":"2022-08-02T17:29:56.841754Z","shell.execute_reply.started":"2022-08-02T17:29:56.532541Z","shell.execute_reply":"2022-08-02T17:29:56.840433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Observing relations in Categorical columns","metadata":{}},{"cell_type":"code","source":"cat_corr_cls = cat_cols.copy()\ncat_corr_cls.remove('failure')\n\ncat_df = df[cat_corr_cls].copy()\ncat_df = cat_df.astype('object')\nplt.figure(figsize=(13,10))\nsns.heatmap(pd.get_dummies(cat_df).corr(), annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:56.843296Z","iopub.execute_input":"2022-08-02T17:29:56.843673Z","iopub.status.idle":"2022-08-02T17:29:58.720160Z","shell.execute_reply.started":"2022-08-02T17:29:56.843638Z","shell.execute_reply":"2022-08-02T17:29:58.718759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This heatmap shows many instances of 1 and -1 indicating that there must be some definite relationship between attributes or product codes such that certain combination are always true or certain combinations are never possible","metadata":{}},{"cell_type":"code","source":"pd.get_dummies(cat_df)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-02T17:29:58.721950Z","iopub.execute_input":"2022-08-02T17:29:58.722396Z","iopub.status.idle":"2022-08-02T17:29:58.766863Z","shell.execute_reply.started":"2022-08-02T17:29:58.722356Z","shell.execute_reply":"2022-08-02T17:29:58.765903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let us examine the case of product_code = A\n\nAssuming that product_code = A, lets check the distribution(or count) of other variables","metadata":{}},{"cell_type":"code","source":"cat_cols_encoded = pd.get_dummies(cat_df)\nfor col in cat_cols_encoded:\n    cat_cols_encoded[col] = cat_cols_encoded[col] & cat_cols_encoded['product_code_A']\n\ncat_cols_encoded.sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:58.768495Z","iopub.execute_input":"2022-08-02T17:29:58.769002Z","iopub.status.idle":"2022-08-02T17:29:58.809718Z","shell.execute_reply.started":"2022-08-02T17:29:58.768956Z","shell.execute_reply":"2022-08-02T17:29:58.808320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This shows that given product_code = A, attribute_0 is always material_7 since attribute_0_material_5 = 0 and attribute_0_material_7 = 5100\n\nBasically, if product_code = A, we surely know what attribute_0 will be\n\nApplying same idea on remaining attributes, we can see that if product_code = A we can actually see a unique guranteed combination of the 4 attributes","metadata":{}},{"cell_type":"markdown","source":"So, i applied this idea to all categorical columns to find similar observations","metadata":{}},{"cell_type":"code","source":"cat_cols_encoded = pd.get_dummies(cat_df)\nsum_count_matrix_train = pd.DataFrame(columns=cat_cols_encoded.columns)\nfor col1 in cat_cols_encoded.columns:\n    for col in cat_cols_encoded.columns:\n        sum_count_matrix_train.loc[col, col1] = (cat_cols_encoded[col] & cat_cols_encoded[col1]).sum()\n        \nsum_count_matrix_train","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:58.811520Z","iopub.execute_input":"2022-08-02T17:29:58.811883Z","iopub.status.idle":"2022-08-02T17:29:58.983142Z","shell.execute_reply.started":"2022-08-02T17:29:58.811835Z","shell.execute_reply":"2022-08-02T17:29:58.981769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"If we know certain column = True, say attribute_0_material_5 is true, then if all values of rows made of attribute_1 have only one non-zero value, then we can say that knowing attribute_0_material_5  = true means we also know what attribute_1 is","metadata":{}},{"cell_type":"markdown","source":"This code implements this idea and makes a heatmap ","metadata":{}},{"cell_type":"code","source":"gr_matrix_train = sum_count_matrix_train.copy()\ngr_matrix_train = (gr_matrix_train != 0).astype(int)\n\nprod_cols = gr_matrix_train.columns[gr_matrix_train.columns.str.contains('product_code')]\natt0_cols = gr_matrix_train.columns[gr_matrix_train.columns.str.contains('attribute_0')]\natt1_cols = gr_matrix_train.columns[gr_matrix_train.columns.str.contains('attribute_1')]\natt2_cols = gr_matrix_train.columns[gr_matrix_train.columns.str.contains('attribute_2')]\natt3_cols = gr_matrix_train.columns[gr_matrix_train.columns.str.contains('attribute_3')]\n\nfor col in gr_matrix_train.columns:\n    \n    ct = 0\n    for row in prod_cols:\n        ct += gr_matrix_train[col][row]\n    gr_matrix_train.loc['product', col] = ct\n    ct = 0\n    \n    ct = 0\n    for row in att0_cols:\n        ct += gr_matrix_train[col][row]\n    gr_matrix_train.loc['attribute_0', col] = ct\n    ct = 0\n    \n    ct = 0\n    for row in att1_cols:\n        ct += gr_matrix_train[col][row]\n    gr_matrix_train.loc['attribute_1', col] = ct\n    ct = 0\n    \n    ct = 0\n    for row in att2_cols:\n        ct += gr_matrix_train[col][row]\n    gr_matrix_train.loc['attribute_2', col] = ct\n    ct = 0\n    \n    ct = 0\n    for row in att3_cols:\n        ct += gr_matrix_train[col][row]\n    gr_matrix_train.loc['attribute_3', col] = ct\n    ct = 0\n    \n    \ngr_matrix_train.drop(prod_cols, axis='rows', inplace=True)\ngr_matrix_train.drop(att0_cols, axis='rows', inplace=True)\ngr_matrix_train.drop(att1_cols, axis='rows', inplace=True)\ngr_matrix_train.drop(att2_cols, axis='rows', inplace=True)\ngr_matrix_train.drop(att3_cols, axis='rows', inplace=True)\n\ngr_matrix_train = (gr_matrix_train == 1).T\nplt.figure(figsize=(10,8))\nsns.heatmap(gr_matrix_train, cmap='flare')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:58.984784Z","iopub.execute_input":"2022-08-02T17:29:58.985144Z","iopub.status.idle":"2022-08-02T17:29:59.417618Z","shell.execute_reply.started":"2022-08-02T17:29:58.985113Z","shell.execute_reply":"2022-08-02T17:29:59.416347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From this heatmap we can infer that knowing any product_code means we surely know what attribute_0,1,2,3 are,  because for each product_code the combination of attributes is unique\n\nSo the attributes combintaions can actually tell the product_code.","metadata":{}},{"cell_type":"code","source":"cat_dft = dft[cat_corr_cls].copy()\ncat_dft = cat_dft.astype('object')\n\ncat_cols_encoded_test = pd.get_dummies(cat_dft)\nsum_count_matrix_test = pd.DataFrame(columns=cat_cols_encoded_test.columns)\nfor col1 in cat_cols_encoded_test.columns:\n    for col in cat_cols_encoded_test.columns:\n        sum_count_matrix_test.loc[col, col1] = (cat_cols_encoded_test[col] & cat_cols_encoded_test[col1]).sum()\n        \nsum_count_matrix_test","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:59.419299Z","iopub.execute_input":"2022-08-02T17:29:59.420224Z","iopub.status.idle":"2022-08-02T17:29:59.575512Z","shell.execute_reply.started":"2022-08-02T17:29:59.420184Z","shell.execute_reply":"2022-08-02T17:29:59.574109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gr_matrix_test = sum_count_matrix_test.copy()\ngr_matrix_test = (gr_matrix_test != 0).astype(int)\n\nprod_cols = gr_matrix_test.columns[gr_matrix_test.columns.str.contains('product_code')]\natt0_cols = gr_matrix_test.columns[gr_matrix_test.columns.str.contains('attribute_0')]\natt1_cols = gr_matrix_test.columns[gr_matrix_test.columns.str.contains('attribute_1')]\natt2_cols = gr_matrix_test.columns[gr_matrix_test.columns.str.contains('attribute_2')]\natt3_cols = gr_matrix_test.columns[gr_matrix_test.columns.str.contains('attribute_3')]\n\nfor col in gr_matrix_test.columns:\n    \n    ct = 0\n    for row in prod_cols:\n        ct += gr_matrix_test[col][row]\n    gr_matrix_test.loc['product', col] = ct\n    ct = 0\n    \n    ct = 0\n    for row in att0_cols:\n        ct += gr_matrix_test[col][row]\n    gr_matrix_test.loc['attribute_0', col] = ct\n    ct = 0\n    \n    ct = 0\n    for row in att1_cols:\n        ct += gr_matrix_test[col][row]\n    gr_matrix_test.loc['attribute_1', col] = ct\n    ct = 0\n    \n    ct = 0\n    for row in att2_cols:\n        ct += gr_matrix_test[col][row]\n    gr_matrix_test.loc['attribute_2', col] = ct\n    ct = 0\n    \n    ct = 0\n    for row in att3_cols:\n        ct += gr_matrix_test[col][row]\n    gr_matrix_test.loc['attribute_3', col] = ct\n    ct = 0\n    \n    \ngr_matrix_test.drop(prod_cols, axis='rows', inplace=True)\ngr_matrix_test.drop(att0_cols, axis='rows', inplace=True)\ngr_matrix_test.drop(att1_cols, axis='rows', inplace=True)\ngr_matrix_test.drop(att2_cols, axis='rows', inplace=True)\ngr_matrix_test.drop(att3_cols, axis='rows', inplace=True)\n\ngr_matrix_test = (gr_matrix_test == 1).T\nplt.figure(figsize=(10,8))\nsns.heatmap(gr_matrix_test, cmap='flare')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:29:59.577362Z","iopub.execute_input":"2022-08-02T17:29:59.577723Z","iopub.status.idle":"2022-08-02T17:29:59.994870Z","shell.execute_reply.started":"2022-08-02T17:29:59.577690Z","shell.execute_reply":"2022-08-02T17:29:59.993456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here product_code has unique combinations with remaining attributes, and attribute_3 also has unique combinations with remaining attributes and product_codes\n\nHere knowing product codes means we know remaining attributes, knowing attribute_3 means we know remaining attributes and product_codes because of unique combinations","metadata":{}},{"cell_type":"markdown","source":"# Numerical Integer Columns Distributions","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,5))\nn=1\n\nfor col in num_int_cols:\n    plt.subplot(1,3,n)\n    sns.histplot(df[col], color='blue', label='Train')\n    sns.histplot(dft[col], color='orange', label='Test')\n    plt.legend()\n    n=n+1","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-02T17:30:00.000518Z","iopub.execute_input":"2022-08-02T17:30:00.001159Z","iopub.status.idle":"2022-08-02T17:30:01.692632Z","shell.execute_reply.started":"2022-08-02T17:30:00.001101Z","shell.execute_reply":"2022-08-02T17:30:01.691290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Numerical Float Columns Distributions","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,30))\nn=1\n\ntrain_df = df.copy()\ntest_df = dft.copy()\n\ntrain_df['Source'] = 'Train'\ntest_df['Source'] = 'Test'\n\nfor col in num_float_cols:\n    plt.subplot(6,3,n)\n    sns.histplot(data=pd.concat([train_df, test_df]).reset_index(drop=True), x=col, hue='Source')\n    n=n+1","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:30:01.694230Z","iopub.execute_input":"2022-08-02T17:30:01.695408Z","iopub.status.idle":"2022-08-02T17:30:14.864785Z","shell.execute_reply.started":"2022-08-02T17:30:01.695367Z","shell.execute_reply":"2022-08-02T17:30:14.863411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Loading column is slightly skewed, rest seem to be normally distributed","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16,12))\nsns.heatmap(df[num_float_cols].corr(), annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:30:14.866872Z","iopub.execute_input":"2022-08-02T17:30:14.867327Z","iopub.status.idle":"2022-08-02T17:30:16.368143Z","shell.execute_reply.started":"2022-08-02T17:30:14.867279Z","shell.execute_reply":"2022-08-02T17:30:16.366550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Measurement 17 has correlations with measurement 5 and measurement 8","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,5))\nplt.subplot(1,2,1)\nsns.scatterplot(df['measurement_17'], df['measurement_8'], hue=df['failure'])\nplt.subplot(1,2,2)\nsns.scatterplot(df['measurement_17'], df['measurement_5'], hue=df['failure'])","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:30:16.370068Z","iopub.execute_input":"2022-08-02T17:30:16.370747Z","iopub.status.idle":"2022-08-02T17:30:18.036933Z","shell.execute_reply.started":"2022-08-02T17:30:16.370639Z","shell.execute_reply":"2022-08-02T17:30:18.035736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Missing Values","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,10))\nplt.title('Training Data Missing Values')\nsns.barplot(df.isnull().sum().values, df.isnull().sum().keys(), color='blue', orient='h')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:30:18.038454Z","iopub.execute_input":"2022-08-02T17:30:18.039581Z","iopub.status.idle":"2022-08-02T17:30:18.942057Z","shell.execute_reply.started":"2022-08-02T17:30:18.039529Z","shell.execute_reply":"2022-08-02T17:30:18.940639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,10))\nplt.title('Testing Data Missing Values')\nsns.barplot(dft.isnull().sum().values, dft.isnull().sum().keys(), color='blue', orient='h')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T17:30:18.943659Z","iopub.execute_input":"2022-08-02T17:30:18.944076Z","iopub.status.idle":"2022-08-02T17:30:19.399344Z","shell.execute_reply.started":"2022-08-02T17:30:18.944037Z","shell.execute_reply":"2022-08-02T17:30:19.398361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}